From 88ac6d8eccad207c67389ec5341d1b921c646556 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 26 Sep 2026 13:35:40 -0400 Subject: [PATCH] Move golden_al_distribution/ into infx/ GOLDEN_DIR in infx/srt_slurm/synthetic_acceptance.py and the GLM-5.3 TileRT curve path in server_tilert.sh follow the move; every doc, prompt, recipe and config-comment reference is updated. perf-changelog.yaml history is left untouched (append-only). Adds a test that the default GOLDEN_DIR resolves to the committed curves. Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/codeowner-signoff-verify-prompt.md | 4 ++-- AGENTS.md | 2 +- MODELS.md | 2 +- MODELS_zh.md | 2 +- benchmarks/multi_node/amd_utils/models.yaml | 2 +- benchmarks/multi_node/amd_utils/server_sglang.sh | 6 +++--- benchmarks/multi_node/amd_utils/server_tilert.sh | 2 +- .../sglang/b200-fp4/agentx/agg-variants.yaml | 6 +++--- .../speedbench/dsv4dspark_fp4_b300_vllm.sh | 4 ++-- configs/amd-master.yaml | 4 ++-- configs/nvidia-master.yaml | 16 ++++++++-------- docs/PR_REVIEW_CHECKLIST.md | 2 +- docs/PR_REVIEW_CHECKLIST_zh.md | 4 ++-- docs/configuration-procedures.md | 10 +++++----- docs/configuration-procedures_zh.md | 10 +++++----- .../golden_al_distribution}/README.md | 6 +++--- .../golden_al_distribution}/README_zh.md | 6 +++--- .../dsv4-pro-0813-dspark.yaml | 0 .../dsv41flash_dspark.yaml | 0 .../golden_al_distribution}/dsv4_mtp.yaml | 0 .../golden_al_distribution}/glm5.2_mtp.yaml | 0 .../golden_al_distribution}/glm5.3_mtp.yaml | 0 .../golden_al_distribution}/kimik3_dspark.yaml | 0 ...ple_method_block_rejection_sample_method.yaml | 0 .../minimaxm3_eagle3.yaml | 0 .../minimaxm3_eagle3_gqa.yaml | 0 .../golden_al_distribution}/qwen3.5_mtp.yaml | 0 .../golden_al_distribution}/qwen3.8next_mtp.yaml | 0 infx/srt_slurm/synthetic_acceptance.py | 2 +- utils/test_synthetic_acceptance.py | 7 +++++++ 30 files changed, 52 insertions(+), 45 deletions(-) rename {golden_al_distribution => infx/golden_al_distribution}/README.md (94%) rename {golden_al_distribution => infx/golden_al_distribution}/README_zh.md (94%) rename {golden_al_distribution => infx/golden_al_distribution}/dsv4-pro-0813-dspark.yaml (100%) rename {golden_al_distribution => infx/golden_al_distribution}/dsv41flash_dspark.yaml (100%) rename {golden_al_distribution => infx/golden_al_distribution}/dsv4_mtp.yaml (100%) rename {golden_al_distribution => infx/golden_al_distribution}/glm5.2_mtp.yaml (100%) rename {golden_al_distribution => infx/golden_al_distribution}/glm5.3_mtp.yaml (100%) rename {golden_al_distribution => infx/golden_al_distribution}/kimik3_dspark.yaml (100%) rename {golden_al_distribution => infx/golden_al_distribution}/kimik3_dspark_probabilistic_sample_method_block_rejection_sample_method.yaml (100%) rename {golden_al_distribution => infx/golden_al_distribution}/minimaxm3_eagle3.yaml (100%) rename {golden_al_distribution => infx/golden_al_distribution}/minimaxm3_eagle3_gqa.yaml (100%) rename {golden_al_distribution => infx/golden_al_distribution}/qwen3.5_mtp.yaml (100%) rename {golden_al_distribution => infx/golden_al_distribution}/qwen3.8next_mtp.yaml (100%) diff --git a/.github/codeowner-signoff-verify-prompt.md b/.github/codeowner-signoff-verify-prompt.md index d29eaf8bd7..7e00924e40 100644 --- a/.github/codeowner-signoff-verify-prompt.md +++ b/.github/codeowner-signoff-verify-prompt.md @@ -324,7 +324,7 @@ speculative decoding. From the PR diff, identify configs that are BOTH: downloads, or config names containing `-mtp` / `eagle`. Agentic replay does not reproduce real-world token-by-token traffic, so measured acceptance there is not representative. Per the AgentX fairness guidelines -in `golden_al_distribution/README.md` on the checked-out default branch, such configs +in `infx/golden_al_distribution/README.md` on the checked-out default branch, such configs must instead SIMULATE acceptance at the committed golden acceptance length (AL). Verify BOTH: - (a) SIMULATED ACCEPTANCE ENABLED. The launch config must pin a simulated/synthetic @@ -343,7 +343,7 @@ Verify BOTH: FAIL if an agentic spec-decode config runs real (unsimulated) acceptance. Name the config/script and line. - (b) AL VALUE MATCHES THE GOLDEN CURVE. Read the committed golden AL YAML for the - model in `golden_al_distribution/` on the default-branch checkout. Examples include + model in `infx/golden_al_distribution/` on the default-branch checkout. Examples include `qwen3.5_mtp.yaml` and `minimaxm3_eagle3.yaml`. Confirm the pinned AL equals the golden value for that model, thinking mode, and the config's `num_speculative_tokens` / MTP level (e.g. qwen3.5 thinking_on with 3 speculative tokens -> 3.39). For TRT-LLM configs, compare diff --git a/AGENTS.md b/AGENTS.md index 44c887e843..68e0a2560e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -75,7 +75,7 @@ check_env_vars IS_MULTINODE MODEL_NAME PRECISION ## SRT Slurm synthetic acceptance -- **Do not hard-code synthetic acceptance lengths in SRT recipes, master configs, or launchers.** InferenceX automatically selects the measured value from [`golden_al_distribution/`](golden_al_distribution/) for speculative AgentX throughput runs. Do not add manual `SYNTHETIC_ACCEPTANCE_LENGTH`, vLLM `synthetic_acceptance_length`, SGLang `SGLANG_SIMULATE_ACC_LEN`, or TRT-LLM `TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS` settings. +- **Do not hard-code synthetic acceptance lengths in SRT recipes, master configs, or launchers.** InferenceX automatically selects the measured value from [`infx/golden_al_distribution/`](infx/golden_al_distribution/) for speculative AgentX throughput runs. Do not add manual `SYNTHETIC_ACCEPTANCE_LENGTH`, vLLM `synthetic_acceptance_length`, SGLang `SGLANG_SIMULATE_ACC_LEN`, or TRT-LLM `TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS` settings. - Submit recipes through [`apply_srt_recipe`](runners/slurm_utils.sh). Its [`infx/srt_slurm` connector](infx/srt_slurm/synthetic_acceptance.py) applies native SRT `--set` / `--unset` overrides; calling upstream `srtctl` directly does not perform InferenceX's automatic selection. - Keep the actual speculative method, draft model, draft-token count, and relevant sampling settings explicit in the recipe. The connector combines the generation role's settings (decode, otherwise aggregated), after caller overrides, with `MODEL_PREFIX` and `THINKING_MODE` to select the golden curve. For Kimi DSpark, explicitly set `draft_sample_method` to `greedy` or `probabilistic`. - Eval-only and non-AgentX runs use real verification; the connector removes stale synthetic settings. Non-speculative roles do not receive simulation settings. `RUN_EVAL` does not disable simulation for the throughput portion. diff --git a/MODELS.md b/MODELS.md index 9e6fe11d53..528b88162c 100644 --- a/MODELS.md +++ b/MODELS.md @@ -34,7 +34,7 @@ The A/B retirements below concern standalone non-spec-decode baselines maintaine **Status: baseline retirement is not yet enacted.** Retire a redundant baseline only once replacement coverage exists for the affected model, hardware, and engine. Keep non-spec-decode configurations that contribute to the Pareto frontier; the existence of a spec-decode counterpart alone is not a reason to remove them. -**Going forward we no longer maintain separate non-spec-decode and spec-decode tracks solely as an A/B comparison.** The non-spec-decode arm existed as a neutral baseline back when acceptance length wasn't standardized. That is now solved. [`golden_al_distribution/`](golden_al_distribution/) commits one golden acceptance-length curve per model, thinking mode, and draft length, measured on the SPEED-Bench `coding` category. When speculative decoding is enabled, AgentX pins submissions to that curve through synthetic acceptance (vLLM `synthetic_acceptance_length`, SGLang `SGLANG_SIMULATE_ACC_LEN`, TensorRT-LLM `TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS`, etc). With a fair, engine-independent acceptance target in place, spec-decode results are directly comparable on their own and a dedicated non-spec-decode baseline is redundant. New models, including Kimi-K3, do not require that separate baseline from day 0. +**Going forward we no longer maintain separate non-spec-decode and spec-decode tracks solely as an A/B comparison.** The non-spec-decode arm existed as a neutral baseline back when acceptance length wasn't standardized. That is now solved. [`infx/golden_al_distribution/`](infx/golden_al_distribution/) commits one golden acceptance-length curve per model, thinking mode, and draft length, measured on the SPEED-Bench `coding` category. When speculative decoding is enabled, AgentX pins submissions to that curve through synthetic acceptance (vLLM `synthetic_acceptance_length`, SGLang `SGLANG_SIMULATE_ACC_LEN`, TensorRT-LLM `TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS`, etc). With a fair, engine-independent acceptance target in place, spec-decode results are directly comparable on their own and a dedicated non-spec-decode baseline is redundant. New models, including Kimi-K3, do not require that separate baseline from day 0. **Publish the best Pareto points, whether speculative decoding is enabled or disabled.** Recipes may disable MTP, EAGLE/EAGLE3, DSpark, or another draft method when doing so produces a better operating point, for example at high throughput. Valid non-spec-decode results remain eligible for publication under the same [North-star Pareto policy](#north-star-pareto-policy). A frontier may therefore contain both spec-decode and non-spec-decode points. diff --git a/MODELS_zh.md b/MODELS_zh.md index 14143f9d5b..ed77e52ba1 100644 --- a/MODELS_zh.md +++ b/MODELS_zh.md @@ -34,7 +34,7 @@ InferenceX-e2e 运行在数量固定且有限的 GPU 资源池上,并由一支 **状态:基线下线尚未执行。** 只有受影响的模型、硬件与引擎具备替代覆盖后,才能下线冗余基线。对帕累托前沿有贡献的非投机解码配置仍需保留;仅有对应的投机解码分支,不足以成为移除它们的理由。 -**今后我们不再仅为 A/B 对照而分别维护非投机解码与投机解码两条赛道。** 当初保留非投机解码分支,是把它当作中立基线。那时接受长度(AL)完全取决于提交方草稿头(draft head)的实际水平,导致各家投机解码数据之间无法横向比较。这一问题现已解决。[`golden_al_distribution/`](golden_al_distribution/) 为每个模型、thinking 模式与草稿长度各提交了一条黄金 AL 曲线,均在 SPEED-Bench `coding` 类别上测得。启用投机解码时,AgentX 通过合成接受(synthetic acceptance)将提交锁定到该曲线(vLLM 用 `synthetic_acceptance_length`,SGLang 用 `SGLANG_SIMULATE_ACC_LEN`,TensorRT-LLM 用 `TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS`,等等)。既然已有公平且与引擎无关的接受目标,投机解码结果本身即可直接横向比较,独立的非投机解码基线已属冗余。包括 Kimi-K3 在内的新模型,自第 0 天起即不要求维护该独立基线。 +**今后我们不再仅为 A/B 对照而分别维护非投机解码与投机解码两条赛道。** 当初保留非投机解码分支,是把它当作中立基线。那时接受长度(AL)完全取决于提交方草稿头(draft head)的实际水平,导致各家投机解码数据之间无法横向比较。这一问题现已解决。[`infx/golden_al_distribution/`](infx/golden_al_distribution/) 为每个模型、thinking 模式与草稿长度各提交了一条黄金 AL 曲线,均在 SPEED-Bench `coding` 类别上测得。启用投机解码时,AgentX 通过合成接受(synthetic acceptance)将提交锁定到该曲线(vLLM 用 `synthetic_acceptance_length`,SGLang 用 `SGLANG_SIMULATE_ACC_LEN`,TensorRT-LLM 用 `TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS`,等等)。既然已有公平且与引擎无关的接受目标,投机解码结果本身即可直接横向比较,独立的非投机解码基线已属冗余。包括 Kimi-K3 在内的新模型,自第 0 天起即不要求维护该独立基线。 **无论是否启用投机解码,都发布最优的帕累托点。** 若关闭 MTP、EAGLE/EAGLE3、DSpark 或其他草稿方法能得到更优的运行点,例如高吞吐量场景,配方可以关闭投机解码。有效的非投机解码结果仍可按相同的[北极星帕累托策略](#北极星帕累托策略)参与发布。因此,同一条前沿可以同时包含投机解码和非投机解码数据点。 diff --git a/benchmarks/multi_node/amd_utils/models.yaml b/benchmarks/multi_node/amd_utils/models.yaml index 1b871a4f9e..603b0bb80e 100644 --- a/benchmarks/multi_node/amd_utils/models.yaml +++ b/benchmarks/multi_node/amd_utils/models.yaml @@ -87,5 +87,5 @@ DeepSeek-V4-Pro-AgentX: &DeepSeek-V4-Pro-AgentX # measured clean across c4-c256; the EAGLE 3-1-4 arm remains available through # spec-decoding: mtp. Synthetic acceptance is checkpoint-specific in # server_sglang.sh and does not depend on which of the two runs: thinking-on, -# draft lengths 1-8 are wired from golden_al_distribution/dsv4-pro-0813-dspark.yaml. +# draft lengths 1-8 are wired from infx/golden_al_distribution/dsv4-pro-0813-dspark.yaml. DeepSeek-V4-Pro-0813-AgentX: *DeepSeek-V4-Pro-AgentX diff --git a/benchmarks/multi_node/amd_utils/server_sglang.sh b/benchmarks/multi_node/amd_utils/server_sglang.sh index ffb18695e9..bd6403f8c9 100755 --- a/benchmarks/multi_node/amd_utils/server_sglang.sh +++ b/benchmarks/multi_node/amd_utils/server_sglang.sh @@ -1418,7 +1418,7 @@ else # Agentic trace replay doesn't reproduce real token-by-token traffic, so # measured MTP/EAGLE acceptance there isn't representative (PR #2309 # review: https://github.com/SemiAnalysisAI/InferenceX/pull/2309#pullrequestreview-4778348624). - # Per the AgentX fairness guidelines (golden_al_distribution/README.md), + # Per the AgentX fairness guidelines (infx/golden_al_distribution/README.md), # agentic throughput benchmarks simulate acceptance at the model's # committed golden AL instead of measuring real (non-representative) # acceptance. Eval runs (RUN_EVAL / EVAL_ONLY) need real acceptance so @@ -1426,7 +1426,7 @@ else # per checkpoint, thinking mode, and draft length, including when the # supported PD draft implementation differs from the calibration engine. # Sources (thinking_on): dsv4_mtp.yaml for the original checkpoint and - # golden_al_distribution/dsv4-pro-0813-dspark.yaml for Pro-0813. + # infx/golden_al_distribution/dsv4-pro-0813-dspark.yaml for Pro-0813. DECODE_SIM_ACC_ENV="" if [[ "$DECODE_MTP_SIZE" -gt 0 ]] && { [[ "${IS_AGENTIC}" == "1" ]] || [[ "${IS_AGENTIC:-}" == "true" ]]; }; then if [[ "${EVAL_ONLY}" == "true" ]] || [[ "${RUN_EVAL}" == "true" ]]; then @@ -1453,7 +1453,7 @@ else if [[ -n "$DSV4_GOLDEN_AL" ]]; then DECODE_SIM_ACC_ENV="SGLANG_SIMULATE_ACC_LEN=${DSV4_GOLDEN_AL} SGLANG_SIMULATE_ACC_METHOD=match-expected SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token" else - echo "WARNING: agentic spec-decoding run (model=${MODEL_NAME}, algorithm=${SPEC_DECODING:-mtp}, DECODE_MTP_SIZE=${DECODE_MTP_SIZE}) has no golden AL wired in server_sglang.sh -- falling back to real (unsimulated, non-representative) acceptance. Add a case in server_sglang.sh and golden_al_distribution/ before shipping this arm. See golden_al_distribution/README.md." >&2 + echo "WARNING: agentic spec-decoding run (model=${MODEL_NAME}, algorithm=${SPEC_DECODING:-mtp}, DECODE_MTP_SIZE=${DECODE_MTP_SIZE}) has no golden AL wired in server_sglang.sh -- falling back to real (unsimulated, non-representative) acceptance. Add a case in server_sglang.sh and infx/golden_al_distribution/ before shipping this arm. See infx/golden_al_distribution/README.md." >&2 fi fi fi diff --git a/benchmarks/multi_node/amd_utils/server_tilert.sh b/benchmarks/multi_node/amd_utils/server_tilert.sh index 2efec9d8ce..b718035f0c 100644 --- a/benchmarks/multi_node/amd_utils/server_tilert.sh +++ b/benchmarks/multi_node/amd_utils/server_tilert.sh @@ -241,7 +241,7 @@ start_decode() { local extra=( ${TILERT_DECODE_EXTRA_FLAGS} ) if [[ "$SPEC_DECODING" == "mtp" && "$EVAL_ONLY" != "true" && "$RUN_EVAL" != "true" ]]; then check_env_vars MODEL_PREFIX THINKING_MODE - local curve="${WS_PATH%/benchmarks/*}/golden_al_distribution/${MODEL_PREFIX}_mtp.yaml" + local curve="${WS_PATH%/benchmarks/*}/infx/golden_al_distribution/${MODEL_PREFIX}_mtp.yaml" TILERT_SIMULATE_ACC_LEN="$("$PY" - "$curve" "$THINKING_MODE" "$DECODE_MTP_SIZE" <<'PYEOF' import sys, yaml path, thinking, tokens = sys.argv[1], sys.argv[2], int(sys.argv[3]) diff --git a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-variants.yaml b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-variants.yaml index f7b88919e4..8453d5d8a0 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-variants.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/glm5.2/sglang/b200-fp4/agentx/agg-variants.yaml @@ -125,7 +125,7 @@ base: # One TP8 worker serves prefill and decode on a single node. The sweep matrix # supplies the concurrency list; srt_agentic.sh replays every point against # this one server. Acceptance is pinned to the golden thinking-on AL for three -# speculative tokens (golden_al_distribution/glm5.2_mtp.yaml). +# speculative tokens (infx/golden_al_distribution/glm5.2_mtp.yaml). override_c1: name: agg-b200-tp8-c1-mtp roles: @@ -139,7 +139,7 @@ override_c1: # One TP8 worker serves prefill and decode on a single node. The sweep matrix # supplies the concurrency list; srt_agentic.sh replays every point against # this one server. Acceptance is pinned to the golden thinking-on AL for three -# speculative tokens (golden_al_distribution/glm5.2_mtp.yaml). +# speculative tokens (infx/golden_al_distribution/glm5.2_mtp.yaml). override_c4: name: agg-b200-tp8-c4-mtp roles: @@ -153,7 +153,7 @@ override_c4: # One TP8 worker serves prefill and decode on a single node. The sweep matrix # supplies the concurrency list; srt_agentic.sh replays every point against # this one server. Acceptance is pinned to the golden thinking-on AL for three -# speculative tokens (golden_al_distribution/glm5.2_mtp.yaml). +# speculative tokens (infx/golden_al_distribution/glm5.2_mtp.yaml). override_c8: name: agg-b200-tp8-c8-mtp roles: diff --git a/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh b/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh index 0cee4c6941..fabc3b421f 100755 --- a/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh +++ b/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh @@ -39,10 +39,10 @@ MAX_NUM_SEQS="64" # Reserve device memory for KV cache and speculative verification. GPU_MEM_UTIL="0.90" TEMPERATURE="1.0" -# MUST match the golden config: golden_al_distribution/dsv4_mtp.yaml was measured +# MUST match the golden config: infx/golden_al_distribution/dsv4_mtp.yaml was measured # with reasoning_effort=high. # The published recipe uses greedy; probabilistic won at every level on Kimi-K3 -# (golden_al_distribution/kimik3_dspark*.yaml). vLLM accepts exactly these two values +# (infx/golden_al_distribution/kimik3_dspark*.yaml). vLLM accepts exactly these two values # (vllm/config/speculative.py: DraftSampleMethod). case "$DRAFT_SAMPLE_METHOD" in greedy|probabilistic) ;; diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index e08e74208f..c7c808d706 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -654,7 +654,7 @@ dsr1-fp8-mi355x-sglang-disagg-mtp: # Kimi-K3 MXFP4 agentic-coding benchmark on MI355X via ATOM with DSpark # speculative decoding. Acceptance is pinned to the committed golden curve in -# golden_al_distribution/kimik3_dspark_probabilistic_sample_method_block_rejection_sample_method.yaml +# infx/golden_al_distribution/kimik3_dspark_probabilistic_sample_method_block_rejection_sample_method.yaml # at both draft lengths used here: 7 draft tokens -> AL 3.84 at concurrency 1 and 4, # 3 draft tokens -> AL 3.00 at concurrency 14 and 16. # Companion to kimik3-fp4-mi355x-vllm-agentic-mtp: same checkpoint, same runner. @@ -1411,7 +1411,7 @@ dsv41flash-fp4-mi355x-vllm-agentic-dspark: # and draft length (docs/PR_REVIEW_CHECKLIST.md), and a submission may not # substitute its own target. Nothing about that target is written here: # AGENTS.md forbids hard-coding an acceptance length in a master config, so -# server_tilert.sh reads golden_al_distribution/glm5.3_mtp.yaml at launch and +# server_tilert.sh reads infx/golden_al_distribution/glm5.3_mtp.yaml at launch and # fails the run if the curve or the draft length is missing. # The curve is consumed in the same units as every other framework here, and # AgentX replays run with thinking on. diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 1ecb68d8bb..8d0457a6c9 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -1502,7 +1502,7 @@ dsr1-fp8-b300-sglang-mtp: kimik3-fp4-b300-vllm-agentic-dspark: # TP8 x DCP8 with Mooncake as the external KV tier. The recipe drafts with # DSpark level 7 at conc <= 8 and stops above it; synthetic acceptance is - # pinned to the committed golden AL 3.84 (golden_al_distribution/ + # pinned to the committed golden AL 3.84 (infx/golden_al_distribution/ # kimik3_dspark_probabilistic_sample_method_block_rejection_sample_method.yaml), # and EVAL_ONLY switches to real block verification. # @@ -5177,7 +5177,7 @@ qwen3.5-fp8-h100-sglang-agentic: # qwen3.5-fp8-h100-sglang-agentic: same TP8/EP8 GPU-resident and HiCache arms, # plus SGLang EAGLE MTP (num-steps 3, eagle-topk 1, 4 draft tokens = 3 # speculative tokens) with simulated acceptance pinned to the golden AL 3.39 -# (golden_al_distribution/qwen3.5_mtp.yaml, thinking_on, K=3) -- the same value +# (infx/golden_al_distribution/qwen3.5_mtp.yaml, thinking_on, K=3) -- the same value # the GB300 Qwen3.5 AgentX srt-slurm recipes use. Image moves to # lmsysorg/sglang:v0.5.16-cu130 because SGLANG_SIMULATE_ACC_TOKEN_MODE only # exists from v0.5.16; v0.5.14 is the version the fixed-seq-len H100 MTP entry @@ -5187,7 +5187,7 @@ qwen3.5-fp8-h100-sglang-agentic: # policy that new agentic arms enable speculative decoding rather than running a # separate STP baseline (MODELS.md). SGLang EAGLE MTP (num-steps 3, eagle-topk 1, # 4 draft tokens = 3 speculative tokens) with simulated acceptance pinned to the -# golden AL 3.39 (golden_al_distribution/qwen3.5_mtp.yaml, thinking_on, K=3). +# golden AL 3.39 (infx/golden_al_distribution/qwen3.5_mtp.yaml, thinking_on, K=3). # Image is lmsysorg/sglang:v0.5.16-cu130: SGLANG_SIMULATE_ACC_TOKEN_MODE only # exists from v0.5.16, and v0.5.14 is what the fixed-seq-len H200 MTP entry runs. # Search space is the H100 AgentX shape widened for H200's 141 GB: the @@ -7020,7 +7020,7 @@ glm5.2-fp8-h200-dynamo-sglang-agentic-mtp-agg: # AgentX speculative-decoding policy in MODELS.md. SGLang EAGLE runs off # GLM-5.2's built-in nextn head (num-steps 3, # eagle-topk 1, 4 draft tokens = 3 speculative tokens), with acceptance pinned -# to the golden AL 2.99 (golden_al_distribution/glm5.2_mtp.yaml, thinking_on, +# to the golden AL 2.99 (infx/golden_al_distribution/glm5.2_mtp.yaml, thinking_on, # K=3) through SGLANG_SIMULATE_ACC_*. The pinned nightly includes FlashInfer # 0.6.18's BF16 TRTLLM MoE allocation fix for small-batch Blackwell execution # in GLM-5.2's unquantized EAGLE draft head. @@ -7048,7 +7048,7 @@ glm5.2-fp4-b300-sglang-agentic-mtp: # glm5.2-fp8-b200-sglang-agentic-mtp (#2863). Same spec-decode-only shape per # the AgentX policy (MODELS.md): SGLang EAGLE off GLM-5.2's built-in nextn head # (num-steps 3, eagle-topk 1, 4 draft tokens = 3 speculative tokens) with -# acceptance pinned to the golden AL 2.99 (golden_al_distribution/glm5.2_mtp.yaml, +# acceptance pinned to the golden AL 2.99 (infx/golden_al_distribution/glm5.2_mtp.yaml, # thinking_on, K=3), which was measured on glm-5.2-fp8, i.e. on this checkpoint. # Pinned to the 2026-09-08 cu13 dev nightly (the first that carries sgl-project/sglang#38318, the guard for the EAGLE DSA fp8 read-door crash seen on the 2026-09-07 build), the same tag the B200/B300 GLM-5.2 # and Qwen3.5 SGLang AgentX recipes moved to that day. Runs on cluster:b300-dsxe @@ -7078,7 +7078,7 @@ glm5.2-fp8-b300-sglang-agentic-mtp: # policy that agentic arms enable speculative decoding rather than running a # separate STP baseline (MODELS.md). SGLang EAGLE off GLM-5.2's built-in nextn # head (num-steps 3, eagle-topk 1, 4 draft tokens = 3 speculative tokens) with -# acceptance pinned to the golden AL 2.99 (golden_al_distribution/glm5.2_mtp.yaml, +# acceptance pinned to the golden AL 2.99 (infx/golden_al_distribution/glm5.2_mtp.yaml, # thinking_on, K=3) through SGLANG_SIMULATE_ACC_*. SGLang v0.5.16 is the first # release that reads SGLANG_SIMULATE_ACC_TOKEN_MODE. The pinned 2026-09-01 # nightly includes FlashInfer 0.6.18's BF16 TRTLLM MoE allocation fix, which is @@ -7107,7 +7107,7 @@ glm5.2-fp4-b200-sglang-agentic-mtp: # glm5.2-fp4-b200-sglang-agentic-mtp. Same spec-decode-only shape per the AgentX # policy (MODELS.md): SGLang EAGLE off GLM-5.2's built-in nextn head (num-steps # 3, eagle-topk 1, 4 draft tokens = 3 speculative tokens) with acceptance pinned -# to the golden AL 2.99 (golden_al_distribution/glm5.2_mtp.yaml, thinking_on, +# to the golden AL 2.99 (infx/golden_al_distribution/glm5.2_mtp.yaml, thinking_on, # K=3) through SGLANG_SIMULATE_ACC_*. That curve was measured on glm-5.2-fp8, # i.e. on this checkpoint. Pinned to the 2026-09-08 cu13 dev nightly (the first that carries sgl-project/sglang#38318, the guard for the EAGLE DSA fp8 read-door crash seen on the 2026-09-07 build), the same # tag the B200 Qwen3.5 FP8/FP4 SGLang AgentX recipes moved to that day. @@ -7135,7 +7135,7 @@ glm5.2-fp8-b200-sglang-agentic-mtp: # GLM-5.2 NVFP4 B200 AgentX on Dynamo + SGLang. EAGLE uses the model's # built-in nextn head, with acceptance pinned to the golden thinking-on AL in -# golden_al_distribution/glm5.2_mtp.yaml. Both arms use HiCache write-back +# infx/golden_al_distribution/glm5.2_mtp.yaml. Both arms use HiCache write-back # DRAM offload for the agentic working set. glm5.2-fp4-b200-dynamo-sglang-agentic-agg: image: lmsysorg/sglang:nightly-dev-20260910-00840301 diff --git a/docs/PR_REVIEW_CHECKLIST.md b/docs/PR_REVIEW_CHECKLIST.md index 642a096689..f5020a4b1c 100644 --- a/docs/PR_REVIEW_CHECKLIST.md +++ b/docs/PR_REVIEW_CHECKLIST.md @@ -25,7 +25,7 @@ As a PR reviewer and CODEOWNER, I have reviewed this and have: - [ ] Verified that this PR passes evals. Please link to GitHub Action workflow that shows this. - [ ] Verified that speculative decoding PRs uses chat templates to align the AL distribution to real world - [ ] Verified that every draft model and draft head is served as it ships: the draft that ships with the served checkpoint, at its stored precision, through the pinned upstream image's default handling, with the shipped and effective draft precision recorded in the additional detail section. No submission-side quantization, dtype override, checkpoint substitution, or patch may lower draft precision below that default, regardless of eval results or AL. Explicitly verified that `SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE` is not enabled in the effective recipe, including inherited settings; enabling it is prohibited going forward, and historical runs do not grant an exception. See [Draft-model precision](https://github.com/SemiAnalysisAI/InferenceX/blob/main/CONTRIBUTING.md#draft-model-precision) for what counts as the default and the MLPerf comparison. -- [ ] For agentic workloads: verified that speculative-decoding configs (EAGLE / MTP / draft models) run with simulated synthetic acceptance, with the acceptance-length value taken from the committed golden AL curve in [golden_al_distribution/](https://github.com/SemiAnalysisAI/InferenceX/tree/main/golden_al_distribution) for that model, thinking mode, and draft length. A submission may choose any supported draft length, but it may not substitute a different acceptance target. +- [ ] For agentic workloads: verified that speculative-decoding configs (EAGLE / MTP / draft models) run with simulated synthetic acceptance, with the acceptance-length value taken from the committed golden AL curve in [infx/golden_al_distribution/](https://github.com/SemiAnalysisAI/InferenceX/tree/main/infx/golden_al_distribution) for that model, thinking mode, and draft length. A submission may choose any supported draft length, but it may not substitute a different acceptance target. - [ ] Verified against the current [MODELS.md](https://github.com/SemiAnalysisAI/InferenceX/blob/main/MODELS.md) that this PR does not submit a deprecated model, scenario, or model-scenario combination. - [ ] Verified that the model architecture isn't changed with benchmark hacks like using --hf-overrides to skipping indexer for every x layers on models that don't natively support this. As a general rule, we won't accept optimizations that reduces the number of model architecture FLOPs. Anything that makes that same computation run faster is fair game; target/verifier FLOPs at lower precisions is fine, given that the config passes private evals, but this does not permit lowering draft-model or draft-head precision below what ships. As an general north star princple, we should only use optimizations which is used in production by customers that care about accuracy - [ ] If an company claims that they support vLLM/SGLang as first class LLM inference engines on their hardware, I have verified that the respective vLLM submission made using upstream https://hub.docker.com/u/vllm docker repo, upstream SGLang https://hub.docker.com/u/lmsysorg docker repo. The only exceptions are for new hardware, such as MI455X UALoE72, Vera Rubin NVL72, Rubin NVL8, etc., and for new model architectures where there is an actual reason why vLLM/SGLang does not fundamentally support them yet as supported by vLLM/SGLang community maintainers diff --git a/docs/PR_REVIEW_CHECKLIST_zh.md b/docs/PR_REVIEW_CHECKLIST_zh.md index 8479e2128b..3c5f6180aa 100644 --- a/docs/PR_REVIEW_CHECKLIST_zh.md +++ b/docs/PR_REVIEW_CHECKLIST_zh.md @@ -27,7 +27,7 @@ As a PR reviewer and CODEOWNER, I have reviewed this and have: - [ ] Verified that this PR passes evals. Please link to GitHub Action workflow that shows this. - [ ] Verified that speculative decoding PRs uses chat templates to align the AL distribution to real world - [ ] Verified that every draft model and draft head is served as it ships: the draft that ships with the served checkpoint, at its stored precision, through the pinned upstream image's default handling, with the shipped and effective draft precision recorded in the additional detail section. No submission-side quantization, dtype override, checkpoint substitution, or patch may lower draft precision below that default, regardless of eval results or AL. Explicitly verified that `SGLANG_NVFP4_CKPT_FP8_NEXTN_MOE` is not enabled in the effective recipe, including inherited settings; enabling it is prohibited going forward, and historical runs do not grant an exception. See [Draft-model precision](https://github.com/SemiAnalysisAI/InferenceX/blob/main/CONTRIBUTING.md#draft-model-precision) for what counts as the default and the MLPerf comparison. -- [ ] For agentic workloads: verified that speculative-decoding configs (EAGLE / MTP / draft models) run with simulated synthetic acceptance, with the acceptance-length value taken from the committed golden AL curve in [golden_al_distribution/](https://github.com/SemiAnalysisAI/InferenceX/tree/main/golden_al_distribution) for that model, thinking mode, and draft length. A submission may choose any supported draft length, but it may not substitute a different acceptance target. +- [ ] For agentic workloads: verified that speculative-decoding configs (EAGLE / MTP / draft models) run with simulated synthetic acceptance, with the acceptance-length value taken from the committed golden AL curve in [infx/golden_al_distribution/](https://github.com/SemiAnalysisAI/InferenceX/tree/main/infx/golden_al_distribution) for that model, thinking mode, and draft length. A submission may choose any supported draft length, but it may not substitute a different acceptance target. - [ ] Verified against the current [MODELS.md](https://github.com/SemiAnalysisAI/InferenceX/blob/main/MODELS.md) that this PR does not submit a deprecated model, scenario, or model-scenario combination. - [ ] Verified that the model architecture isn't changed with benchmark hacks like using --hf-overrides to skipping indexer for every x layers on models that don't natively support this. As a general rule, we won't accept optimizations that reduces the number of model architecture FLOPs. Anything that makes that same computation run faster is fair game; target/verifier FLOPs at lower precisions is fine, given that the config passes private evals, but this does not permit lowering draft-model or draft-head precision below what ships. As an general north star princple, we should only use optimizations which is used in production by customers that care about accuracy - [ ] If an company claims that they support vLLM/SGLang as first class LLM inference engines on their hardware, I have verified that the respective vLLM submission made using upstream https://hub.docker.com/u/vllm docker repo, upstream SGLang https://hub.docker.com/u/lmsysorg docker repo. The only exceptions are for new hardware, such as MI455X UALoE72, Vera Rubin NVL72, Rubin NVL8, etc., and for new model architectures where there is an actual reason why vLLM/SGLang does not fundamentally support them yet as supported by vLLM/SGLang community maintainers @@ -52,7 +52,7 @@ Signed: `FILL_IN_GITHUB_USERNAME` 4. 已确认该 PR 通过了 evals(准确性评测),并附上能证明这一点的 GitHub Action 工作流链接。 5. 已确认投机解码(speculative decoding)PR 使用 chat template,使接受长度(AL)分布与真实场景对齐。 6. 已确认所有 draft 模型和 draft head 均按其发布时的默认方式运行:使用随所服务 checkpoint 一同发布的 draft,保持其存储精度,并采用锁定上游镜像的默认加载处理;已在 Additional detail section 中记录 draft 的发布精度与实际运行精度。无论 eval 结果或 AL 如何,提交方不得通过量化、dtype 覆盖、替换 checkpoint 或打补丁将 draft 精度降到该默认值以下。关于"默认"的定义及 MLPerf 对照见[Draft 模型精度](../CONTRIBUTING_zh.md#draft-模型精度)。 -7. 对 agentic 工作负载:已确认投机解码配置(EAGLE / MTP / draft 模型)启用了模拟合成接受(simulated synthetic acceptance),且接受长度(AL)取值来自 [golden_al_distribution/](https://github.com/SemiAnalysisAI/InferenceX/tree/main/golden_al_distribution) 中该模型、thinking 模式与 draft 长度对应的已提交黄金 AL 曲线。提交可选择任意受支持的 draft 长度,但不得替换为其他接受目标。 +7. 对 agentic 工作负载:已确认投机解码配置(EAGLE / MTP / draft 模型)启用了模拟合成接受(simulated synthetic acceptance),且接受长度(AL)取值来自 [infx/golden_al_distribution/](https://github.com/SemiAnalysisAI/InferenceX/tree/main/infx/golden_al_distribution) 中该模型、thinking 模式与 draft 长度对应的已提交黄金 AL 曲线。提交可选择任意受支持的 draft 长度,但不得替换为其他接受目标。 8. 已确认此 PR 对照最新版 [MODELS.md](https://github.com/SemiAnalysisAI/InferenceX/blob/main/MODELS.md),未提交已弃用的模型、场景或模型场景组合。 9. 已确认模型架构未被基准测试 hack 更改,例如在不原生支持的模型上使用 `--hf-overrides` 每 x 层跳过 indexer。一般规则:不接受减少模型架构 FLOPs 的优化;让同样的计算跑得更快没有问题;target/verifier 的更低精度 FLOPs 也可以,前提是该配置通过私有 evals,但这不允许将 draft 模型或 draft head 的精度降到其发布默认值以下。北极星原则:只使用在意准确性的客户在生产中实际使用的优化。 10. 如果公司声称在其硬件上将 vLLM/SGLang 作为一等 LLM 推理引擎支持,已确认相应 vLLM 提交使用上游 [vLLM docker 仓库](https://hub.docker.com/u/vllm)、SGLang 提交使用上游 [lmsysorg docker 仓库](https://hub.docker.com/u/lmsysorg)。唯一例外:新硬件(如 MI455X UALoE72、Vera Rubin NVL72、Rubin NVL8 等),以及经 vLLM/SGLang 社区维护者确认上游尚未从根本上支持的新模型架构。 diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index 6492af02d0..7156acf1de 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -304,7 +304,7 @@ B300 uses the same minimum capture size at c1/c2/c4. Its c1 CI comparison reduce The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use `vllm/vllm-openai:deepseekv41-flash-0909` at TP4 on Blackwell SKUs with native five-token DSpark, -probabilistic drafting. Throughput uses the [committed golden AL](../golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking on and five draft tokens, with synthetic rejection sampling and adaptive verification disabled. Accuracy evals retain real block rejection and adaptive verification. `--engram-config '{"cpu_offload":true}'` +probabilistic drafting. Throughput uses the [committed golden AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking on and five draft tokens, with synthetic rejection sampling and adaptive verification disabled. Accuracy evals retain real block rejection and adaptive verification. `--engram-config '{"cpu_offload":true}'` stores Engram embedding tables in pinned host DRAM accessed through UVA; `kv-offloading: none` describes the separate, GPU-resident KV cache. MXFP4 expert weights determine the recipe's `precision: fp4` label. @@ -356,7 +356,7 @@ for diagnostics. GPU sweep and eval validation is pending. `dsv41flash-fp4-h200-vllm-agentic-dspark` is the H200 AgentX arm of the DeepSeek-V4.1-Flash recipe. It shares `vllm/vllm-openai:deepseekv41-flash-0909` and the text-only serving script with the Blackwell arms: `deepseek_v41` tokenizer and parsers, -1M context, native five-token DSpark with probabilistic drafting. Throughput uses the [committed golden AL](../golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking on and five draft tokens, with synthetic rejection sampling and adaptive verification disabled. Accuracy evals retain real block rejection and adaptive verification. +1M context, native five-token DSpark with probabilistic drafting. Throughput uses the [committed golden AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking on and five draft tokens, with synthetic rejection sampling and adaptive verification disabled. Accuracy evals retain real block rejection and adaptive verification. The arm runs **TP8**, not the upstream TP4. Upstream verifies TP4 on one GB200 NVL4 tray and states that the same layout becomes TP8 per role on 8-GPU nodes, which is what an @@ -397,7 +397,7 @@ port is occupied; serving, replay, metrics, and eval share that endpoint. ### DeepSeek-V4.1-Flash DSpark on H100 -Throughput uses the [committed golden AL](../golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking on and five draft tokens, with synthetic rejection sampling and adaptive verification disabled. Accuracy evals retain real block rejection and adaptive verification. +Throughput uses the [committed golden AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking on and five draft tokens, with synthetic rejection sampling and adaptive verification disabled. Accuracy evals retain real block rejection and adaptive verification. `dsv41flash-fp4-h100-vllm-agentic-dspark` is the H100 AgentX arm of the DeepSeek-V4.1-Flash recipe, added after the H200 arm and deliberately separate from @@ -527,7 +527,7 @@ inspect startup's actual resident/huge-page counts before claiming a benefit. DSpark is the checkpoint's own bundled draft. SGLang exposes no EAGLE or MTP path and no `--speculative-num-steps` knob for it; the recipes pass `--speculative-algorithm DSPARK --speculative-dspark-block-size 5`. Throughput uses the same -[committed golden AL](../golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking +[committed golden AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking on and five draft tokens through `SGLANG_SIMULATE_ACC_LEN` with `match-expected` and `real-draft-token`; accuracy evals keep real verification. Thinking is off by default in SGLang for this model, so the scripts set `SGLANG_DEFAULT_THINKING=1` and @@ -672,7 +672,7 @@ A configuration is ready for sweep only when the executable files agree, the exa ## DeepSeek-V4.1-Flash on MI355X -The `dsv41flash-fp4-mi355x-vllm-agentic-dspark` recipe extends [#2958](https://github.com/SemiAnalysisAI/InferenceX/pull/2958) to MI355X AgentX: TP4 and TP2, concurrency 1–128, native five-token DSpark. Throughput uses the [committed golden AL](../golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking on and five draft tokens, with synthetic rejection sampling and adaptive verification disabled. Accuracy evals retain real block rejection but, unlike the CUDA arms, also keep adaptive verification disabled: it trims verification requests on device, which the ROCm `DeepseekV4IndexerBackend` does not support, and the engine refused to start with it enabled ([run 34651830283](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34651830283)). FP4 describes the MXFP4 experts; the checkpoint also contains MXFP8 weights. +The `dsv41flash-fp4-mi355x-vllm-agentic-dspark` recipe extends [#2958](https://github.com/SemiAnalysisAI/InferenceX/pull/2958) to MI355X AgentX: TP4 and TP2, concurrency 1–128, native five-token DSpark. Throughput uses the [committed golden AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking on and five draft tokens, with synthetic rejection sampling and adaptive verification disabled. Accuracy evals retain real block rejection but, unlike the CUDA arms, also keep adaptive verification disabled: it trims verification requests on device, which the ROCm `DeepseekV4IndexerBackend` does not support, and the engine refused to start with it enabled ([run 34651830283](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34651830283)). FP4 describes the MXFP4 experts; the checkpoint also contains MXFP8 weights. Follow the AMD overrides in the merged [upstream recipe #968](https://github.com/vllm-project/recipes/pull/968): `VLLM_ROCM_USE_AITER=1`, `VLLM_ROCM_USE_AITER_MOE=1`, and `--moe-backend aiter`. The generic AITER selector lets vLLM pick the CK a8w4 experts, matching the DSV4-Pro MI355X recipe. The recipe pins `semianalysis_cc_traces_weka_062126` (the unfiltered corpus) via `WEKA_LOADER_OVERRIDE`. KV stays GPU-resident. Engram stayed on GPU under the upstream AMD defaults until [vllm-project/vllm#57491](https://github.com/vllm-project/vllm/pull/57491) widened the two `is_cuda()` gates to `is_cuda_alike()`. From that commit on, ROCm resolves an `EngramConfig` and `cpu_offload` defaults to on through `VLLM_PLE_CPU_OFFLOAD`, so the recipe sets `--engram-config` explicitly rather than leaning on that default. TP=2 always offloads, since the tables need 94.4 GiB per rank there; TP=4 keeps them resident through concurrency 64, where the KV pool is not the constraint, and offloads only at 128. The recipe likewise trims `--max-num-batched-tokens` only above concurrency 32, to 8192 at TP=2 c64 and TP=4 c128 and to 4096 at TP=2 c128, because the sparse-attention indexer and its companion per-rank buffers grow at roughly 4.4 MiB per batched token. Where that chunk falls below six times the API-server default of 1024 sequences, `--max-num-seqs` is capped at the graph-capture shape: DSpark verifies 1+5 tokens per sequence, and at 4096 against 1024 sequences the engram projection faults during profiling. The rule in every case is to spend device memory on KV only at the concurrencies that ran short of it, leaving the validated low-concurrency settings alone. Images built before that merge still reject the option on ROCm. The MI355X launcher uses the shared HF cache and mounts this model's repository at `/ix`, and exports `INFMAX_CONTAINER_WORKSPACE=/ix` so AgentX dependencies and outputs resolve inside that mount. diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 6afc3283b5..d974e0f236 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -248,7 +248,7 @@ B300 在 c1/c2/c4 使用相同的最小捕获范围。其 c1 CI 对比中,请 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 `vllm/vllm-openai:deepseekv41-flash-0909`,在 Blackwell SKU 上采用 TP4、原生五 token DSpark、 -概率采样草稿。吞吐测试使用[已提交的黄金 AL](../golden_al_distribution/dsv41flash_dspark.yaml):thinking 开启、五个草稿 token 对应 3.51,采用合成拒绝采样并关闭自适应验证。准确率 eval 保留真实块拒绝采样和自适应验证。 +概率采样草稿。吞吐测试使用[已提交的黄金 AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml):thinking 开启、五个草稿 token 对应 3.51,采用合成拒绝采样并关闭自适应验证。准确率 eval 保留真实块拒绝采样和自适应验证。 `--engram-config '{"cpu_offload":true}'` 将 Engram 嵌入表放在固定页主机 DRAM 中,通过 UVA 访问;`kv-offloading: none` 描述的是另行保留在 GPU 上的 KV cache。 专家权重为 MXFP4,因此配方标记为 `precision: fp4`。 @@ -292,7 +292,7 @@ eval 使用真实 acceptance,并保留检查点随附的 draft 和 `dsml_v41` `dsv41flash-fp4-h200-vllm-agentic-dspark` 是 DeepSeek-V4.1-Flash 配方的 H200 AgentX 分支。它与 Blackwell 分支共用 `vllm/vllm-openai:deepseekv41-flash-0909` 和纯文本服务 -脚本:`deepseek_v41` tokenizer 和解析器、1M 上下文、原生五 token DSpark(概率采样草稿)。吞吐测试使用[已提交的黄金 AL](../golden_al_distribution/dsv41flash_dspark.yaml):thinking 开启、五个草稿 token 对应 3.51,采用合成拒绝采样并关闭自适应验证。准确率 eval 保留真实块拒绝采样和自适应验证。 +脚本:`deepseek_v41` tokenizer 和解析器、1M 上下文、原生五 token DSpark(概率采样草稿)。吞吐测试使用[已提交的黄金 AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml):thinking 开启、五个草稿 token 对应 3.51,采用合成拒绝采样并关闭自适应验证。准确率 eval 保留真实块拒绝采样和自适应验证。 该分支使用 **TP8**,而非上游的 TP4。上游在一个 GB200 NVL4 tray 上验证 TP4,并说明在 8 GPU 节点上同一布局每个角色变为 TP8,而 H200 DGXC 节点正是 8 GPU 节点。 @@ -326,7 +326,7 @@ launcher 为该配方将仓库挂载到 `/ix`,避免在 `/workspace` 下创建 ### H100 上的 DeepSeek-V4.1-Flash DSpark -吞吐测试使用[已提交的黄金 AL](../golden_al_distribution/dsv41flash_dspark.yaml):thinking 开启、五个草稿 token 对应 3.51,采用合成拒绝采样并关闭自适应验证。准确率 eval 保留真实块拒绝采样和自适应验证。 +吞吐测试使用[已提交的黄金 AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml):thinking 开启、五个草稿 token 对应 3.51,采用合成拒绝采样并关闭自适应验证。准确率 eval 保留真实块拒绝采样和自适应验证。 `dsv41flash-fp4-h100-vllm-agentic-dspark` 是 DeepSeek-V4.1-Flash 配方的 H100 AgentX 分支,在 H200 分支之后加入,并有意与其分开。H100 **不在**上游硬件表中(该表列出 @@ -437,7 +437,7 @@ GB200 的主机表布局为 `per_rank`:计算节点内核通过 `madvise` 启 DSpark 是检查点自带的草稿模型。SGLang 对它不提供 EAGLE 或 MTP 路径,也没有 `--speculative-num-steps` 参数;配方传入 `--speculative-algorithm DSPARK --speculative-dspark-block-size 5`。吞吐测试通过 `SGLANG_SIMULATE_ACC_LEN`(`match-expected`、 -`real-draft-token`)使用同一[已提交的黄金 AL](../golden_al_distribution/dsv41flash_dspark.yaml): +`real-draft-token`)使用同一[已提交的黄金 AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml): thinking 开启、五个草稿 token 对应 3.51;准确率 eval 保留真实验证。SGLang 对该模型默认关闭 thinking,因此脚本设置 `SGLANG_DEFAULT_THINKING=1` 与 `SGLANG_DSV41_REASONING_EFFORT=high`, 以测量黄金 AL 所采集的 thinking 开启状态。 @@ -576,7 +576,7 @@ python -m pytest utils/matrix_logic/ -v ## MI355X 上的 DeepSeek-V4.1-Flash -配方 `dsv41flash-fp4-mi355x-vllm-agentic-dspark` 将 [#2958](https://github.com/SemiAnalysisAI/InferenceX/pull/2958) 扩展至 MI355X AgentX:TP4 与 TP2、并发 1–128、原生五 token DSpark。吞吐测试使用[已提交的黄金 AL](../golden_al_distribution/dsv41flash_dspark.yaml):thinking 开启、五个草稿 token 对应 3.51,采用合成拒绝采样并关闭自适应验证。准确率 eval 保留真实块拒绝采样,但与 CUDA 分支不同,同样关闭自适应验证:它会在设备端裁剪验证请求,而 ROCm 的 `DeepseekV4IndexerBackend` 不支持该操作,启用后引擎拒绝启动([运行 34651830283](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34651830283))。FP4 表示 MXFP4 专家权重;检查点还包含 MXFP8 权重。 +配方 `dsv41flash-fp4-mi355x-vllm-agentic-dspark` 将 [#2958](https://github.com/SemiAnalysisAI/InferenceX/pull/2958) 扩展至 MI355X AgentX:TP4 与 TP2、并发 1–128、原生五 token DSpark。吞吐测试使用[已提交的黄金 AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml):thinking 开启、五个草稿 token 对应 3.51,采用合成拒绝采样并关闭自适应验证。准确率 eval 保留真实块拒绝采样,但与 CUDA 分支不同,同样关闭自适应验证:它会在设备端裁剪验证请求,而 ROCm 的 `DeepseekV4IndexerBackend` 不支持该操作,启用后引擎拒绝启动([运行 34651830283](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34651830283))。FP4 表示 MXFP4 专家权重;检查点还包含 MXFP8 权重。 遵循已合并的[上游配方 #968](https://github.com/vllm-project/recipes/pull/968) 中的 AMD 设置:`VLLM_ROCM_USE_AITER=1`、`VLLM_ROCM_USE_AITER_MOE=1` 和 `--moe-backend aiter`。通用 AITER 选择器允许 vLLM 选择 CK a8w4 专家内核,与 DSV4-Pro MI355X 配方一致。配方通过 `WEKA_LOADER_OVERRIDE` 固定使用完整语料 `semianalysis_cc_traces_weka_062126`。KV 驻留 GPU。在 [vllm-project/vllm#57491](https://github.com/vllm-project/vllm/pull/57491) 将两处 `is_cuda()` 判断放宽为 `is_cuda_alike()` 之前,Engram 按上游 AMD 默认设置常驻 GPU。自该提交起,ROCm 会解析 `EngramConfig`,且 `cpu_offload` 经由 `VLLM_PLE_CPU_OFFLOAD` 默认开启,因此配方显式设置 `--engram-config`,而不依赖该默认值。TP=2 始终下放,因为此时表每 rank 需 94.4 GiB;TP=4 在并发 64 及以下保持常驻(此时 KV 池并非瓶颈),仅在并发 128 时下放。同样地,配方仅在并发高于 32 时调低 `--max-num-batched-tokens`:TP=2 c64 与 TP=4 c128 为 8192,TP=2 c128 为 4096,因为稀疏注意力 indexer 及其配套的每 rank 缓冲区按每个批量 token 约 4.4 MiB 增长。当该分块低于 API server 默认 1024 序列所需的六倍时,`--max-num-seqs` 会被限制为 CUDA graph 捕获的规模:DSpark 每序列验证 1+5 个 token,4096 对 1024 序列会使 engram 投影在 profiling 阶段崩溃。所有情况下的原则一致:只在确实出现 KV 不足的并发点上把设备内存让给 KV,保持低并发处已验证的设置不变。早于该合并的镜像在 ROCm 上仍会拒绝该选项。MI355X launcher 使用共享 HF 缓存,并将此模型的仓库挂载至 `/ix`,同时导出 `INFMAX_CONTAINER_WORKSPACE=/ix`,确保 AgentX 依赖与输出路径位于该挂载中。 diff --git a/golden_al_distribution/README.md b/infx/golden_al_distribution/README.md similarity index 94% rename from golden_al_distribution/README.md rename to infx/golden_al_distribution/README.md index e464fbba52..ae43da91b8 100644 --- a/golden_al_distribution/README.md +++ b/infx/golden_al_distribution/README.md @@ -74,10 +74,10 @@ This policy follows the same broad principle as MLPerf Inference: prescribe the ## How a golden AL curve is collected -The push-button [`speedbench-al.yml`](../.github/workflows/speedbench-al.yml) workflow, introduced in [InferenceX#1650](https://github.com/SemiAnalysisAI/InferenceX/pull/1650) and extended to additional MTP and EAGLE3 models in [InferenceX#1706](https://github.com/SemiAnalysisAI/InferenceX/pull/1706), performs the following process. It superseded the early manually assembled reference in [InferenceX#1592](https://github.com/SemiAnalysisAI/InferenceX/pull/1592), making the exact commands, logs, outputs, and generated YAML auditable from one run. +The push-button [`speedbench-al.yml`](../../.github/workflows/speedbench-al.yml) workflow, introduced in [InferenceX#1650](https://github.com/SemiAnalysisAI/InferenceX/pull/1650) and extended to additional MTP and EAGLE3 models in [InferenceX#1706](https://github.com/SemiAnalysisAI/InferenceX/pull/1706), performs the following process. It superseded the early manually assembled reference in [InferenceX#1592](https://github.com/SemiAnalysisAI/InferenceX/pull/1592), making the exact commands, logs, outputs, and generated YAML auditable from one run. 1. A maintainer dispatches the workflow with a model, model prefix, vLLM image, draft lengths (normally 1–8), thinking modes, `category=coding`, and `output-len=4096`. -2. The workflow launches the model on a B300 runner and selects the matching collector under [`benchmarks/single_node/speedbench/`](../benchmarks/single_node/speedbench/). +2. The workflow launches the model on a B300 runner and selects the matching collector under [`benchmarks/single_node/speedbench/`](../../benchmarks/single_node/speedbench/). 3. For every `(thinking mode, draft length)` cell, the collector starts a clean vLLM server with real MTP or EAGLE3 decoding and the model's production sampling/chat-template settings. 4. The collector snapshots vLLM's cumulative accepted-token and verification-draft counters, runs every prompt in the SPEED-Bench Qualitative `coding` category through `vllm bench serve`, and snapshots the counters again. 5. It computes the mean acceptance length as: @@ -146,7 +146,7 @@ Before accepting an updated curve, reviewers should verify: - [vLLM synthetic acceptance support](https://github.com/vllm-project/vllm/pull/40662) - [ATOM forced acceptance-length support](https://github.com/ROCm/ATOM/pull/1948) - [InferenceX synthetic-acceptance tracking issue](https://github.com/SemiAnalysisAI/InferenceX/issues/1651) -- [InferenceX SPEED-Bench workflow](../.github/workflows/speedbench-al.yml) +- [InferenceX SPEED-Bench workflow](../../.github/workflows/speedbench-al.yml) - [InferenceX early reference-alignment PR](https://github.com/SemiAnalysisAI/InferenceX/pull/1592) - [InferenceX initial AL collector PR](https://github.com/SemiAnalysisAI/InferenceX/pull/1650) - [InferenceX multi-model AL collectors PR](https://github.com/SemiAnalysisAI/InferenceX/pull/1706) diff --git a/golden_al_distribution/README_zh.md b/infx/golden_al_distribution/README_zh.md similarity index 94% rename from golden_al_distribution/README_zh.md rename to infx/golden_al_distribution/README_zh.md index b59c4874f3..33a3b70fbe 100644 --- a/golden_al_distribution/README_zh.md +++ b/infx/golden_al_distribution/README_zh.md @@ -74,10 +74,10 @@ python -m atom.entrypoints.openai_server \ ## 黄金 AL 曲线如何收集 -一键触发的 [`speedbench-al.yml`](../.github/workflows/speedbench-al.yml) 工作流最初由 [InferenceX#1650](https://github.com/SemiAnalysisAI/InferenceX/pull/1650) 引入,随后在 [InferenceX#1706](https://github.com/SemiAnalysisAI/InferenceX/pull/1706) 中扩展到更多 MTP 和 EAGLE3 模型。它取代了 [InferenceX#1592](https://github.com/SemiAnalysisAI/InferenceX/pull/1592) 中早期手工整理的参考值,使精确命令、日志、输出和生成的 YAML 都可以从同一次运行中审计。其流程如下: +一键触发的 [`speedbench-al.yml`](../../.github/workflows/speedbench-al.yml) 工作流最初由 [InferenceX#1650](https://github.com/SemiAnalysisAI/InferenceX/pull/1650) 引入,随后在 [InferenceX#1706](https://github.com/SemiAnalysisAI/InferenceX/pull/1706) 中扩展到更多 MTP 和 EAGLE3 模型。它取代了 [InferenceX#1592](https://github.com/SemiAnalysisAI/InferenceX/pull/1592) 中早期手工整理的参考值,使精确命令、日志、输出和生成的 YAML 都可以从同一次运行中审计。其流程如下: 1. 维护者触发工作流,指定模型、模型前缀、vLLM 镜像、草稿长度(通常为 1–8)、思考模式、`category=coding` 和 `output-len=4096`。 -2. 工作流在 B300 runner 上启动模型,并选择 [`benchmarks/single_node/speedbench/`](../benchmarks/single_node/speedbench/) 下对应的收集脚本。 +2. 工作流在 B300 runner 上启动模型,并选择 [`benchmarks/single_node/speedbench/`](../../benchmarks/single_node/speedbench/) 下对应的收集脚本。 3. 对每个“思考模式 × 草稿长度”组合,收集脚本使用真实 MTP 或 EAGLE3 解码以及该模型的生产采样和聊天模板设置,启动一个干净的 vLLM 服务。 4. 收集脚本读取 vLLM 累计的已接受 token 和验证草稿计数器,通过 `vllm bench serve` 运行 SPEED-Bench Qualitative `coding` 类别中的全部提示词,然后再次读取计数器。 5. 按以下公式计算平均接受长度: @@ -146,7 +146,7 @@ gh workflow run speedbench-al.yml \ - [vLLM 合成接受支持](https://github.com/vllm-project/vllm/pull/40662) - [ATOM 强制接受长度支持](https://github.com/ROCm/ATOM/pull/1948) - [InferenceX 合成接受跟踪 issue](https://github.com/SemiAnalysisAI/InferenceX/issues/1651) -- [InferenceX SPEED-Bench 工作流](../.github/workflows/speedbench-al.yml) +- [InferenceX SPEED-Bench 工作流](../../.github/workflows/speedbench-al.yml) - [InferenceX 早期参考值对齐 PR](https://github.com/SemiAnalysisAI/InferenceX/pull/1592) - [InferenceX 初始 AL 收集器 PR](https://github.com/SemiAnalysisAI/InferenceX/pull/1650) - [InferenceX 多模型 AL 收集器 PR](https://github.com/SemiAnalysisAI/InferenceX/pull/1706) diff --git a/golden_al_distribution/dsv4-pro-0813-dspark.yaml b/infx/golden_al_distribution/dsv4-pro-0813-dspark.yaml similarity index 100% rename from golden_al_distribution/dsv4-pro-0813-dspark.yaml rename to infx/golden_al_distribution/dsv4-pro-0813-dspark.yaml diff --git a/golden_al_distribution/dsv41flash_dspark.yaml b/infx/golden_al_distribution/dsv41flash_dspark.yaml similarity index 100% rename from golden_al_distribution/dsv41flash_dspark.yaml rename to infx/golden_al_distribution/dsv41flash_dspark.yaml diff --git a/golden_al_distribution/dsv4_mtp.yaml b/infx/golden_al_distribution/dsv4_mtp.yaml similarity index 100% rename from golden_al_distribution/dsv4_mtp.yaml rename to infx/golden_al_distribution/dsv4_mtp.yaml diff --git a/golden_al_distribution/glm5.2_mtp.yaml b/infx/golden_al_distribution/glm5.2_mtp.yaml similarity index 100% rename from golden_al_distribution/glm5.2_mtp.yaml rename to infx/golden_al_distribution/glm5.2_mtp.yaml diff --git a/golden_al_distribution/glm5.3_mtp.yaml b/infx/golden_al_distribution/glm5.3_mtp.yaml similarity index 100% rename from golden_al_distribution/glm5.3_mtp.yaml rename to infx/golden_al_distribution/glm5.3_mtp.yaml diff --git a/golden_al_distribution/kimik3_dspark.yaml b/infx/golden_al_distribution/kimik3_dspark.yaml similarity index 100% rename from golden_al_distribution/kimik3_dspark.yaml rename to infx/golden_al_distribution/kimik3_dspark.yaml diff --git a/golden_al_distribution/kimik3_dspark_probabilistic_sample_method_block_rejection_sample_method.yaml b/infx/golden_al_distribution/kimik3_dspark_probabilistic_sample_method_block_rejection_sample_method.yaml similarity index 100% rename from golden_al_distribution/kimik3_dspark_probabilistic_sample_method_block_rejection_sample_method.yaml rename to infx/golden_al_distribution/kimik3_dspark_probabilistic_sample_method_block_rejection_sample_method.yaml diff --git a/golden_al_distribution/minimaxm3_eagle3.yaml b/infx/golden_al_distribution/minimaxm3_eagle3.yaml similarity index 100% rename from golden_al_distribution/minimaxm3_eagle3.yaml rename to infx/golden_al_distribution/minimaxm3_eagle3.yaml diff --git a/golden_al_distribution/minimaxm3_eagle3_gqa.yaml b/infx/golden_al_distribution/minimaxm3_eagle3_gqa.yaml similarity index 100% rename from golden_al_distribution/minimaxm3_eagle3_gqa.yaml rename to infx/golden_al_distribution/minimaxm3_eagle3_gqa.yaml diff --git a/golden_al_distribution/qwen3.5_mtp.yaml b/infx/golden_al_distribution/qwen3.5_mtp.yaml similarity index 100% rename from golden_al_distribution/qwen3.5_mtp.yaml rename to infx/golden_al_distribution/qwen3.5_mtp.yaml diff --git a/golden_al_distribution/qwen3.8next_mtp.yaml b/infx/golden_al_distribution/qwen3.8next_mtp.yaml similarity index 100% rename from golden_al_distribution/qwen3.8next_mtp.yaml rename to infx/golden_al_distribution/qwen3.8next_mtp.yaml diff --git a/infx/srt_slurm/synthetic_acceptance.py b/infx/srt_slurm/synthetic_acceptance.py index 9e91531eb7..c22e4103c0 100644 --- a/infx/srt_slurm/synthetic_acceptance.py +++ b/infx/srt_slurm/synthetic_acceptance.py @@ -17,7 +17,7 @@ import yaml -GOLDEN_DIR = Path(__file__).resolve().parents[2] / "golden_al_distribution" +GOLDEN_DIR = Path(__file__).resolve().parents[1] / "golden_al_distribution" ENGINES = { "sglang": "sglang", "sglang-disagg": "sglang", diff --git a/utils/test_synthetic_acceptance.py b/utils/test_synthetic_acceptance.py index a3ae8f92f2..04b0b4ae6a 100644 --- a/utils/test_synthetic_acceptance.py +++ b/utils/test_synthetic_acceptance.py @@ -12,6 +12,7 @@ import yaml from infx.srt_slurm.synthetic_acceptance import ( + GOLDEN_DIR, build_overrides, plan_commands, selected_recipes, @@ -535,3 +536,9 @@ def test_sglang_conflicting_algorithm_aliases_fail(golden_dir: Path) -> None: } with pytest.raises(ValueError, match="Conflicting speculative-algorithm and speculative-algo"): build_overrides(recipe, "dynamo-sglang", ENV, golden_dir=golden_dir) + + +def test_default_golden_dir_holds_committed_curves() -> None: + # Launchers rely on the default; the tests above all pass a temp directory. + assert GOLDEN_DIR == ROOT / "infx/golden_al_distribution" + assert (GOLDEN_DIR / "qwen3.5_mtp.yaml").is_file()