diff --git a/benchmarks/single_node/agentic/qwen3.8next_fp4_mi355x_sglang_mtp.sh b/benchmarks/single_node/agentic/qwen3.8next_fp4_mi355x_sglang_mtp.sh new file mode 100755 index 0000000000..7c8c7fb616 --- /dev/null +++ b/benchmarks/single_node/agentic/qwen3.8next_fp4_mi355x_sglang_mtp.sh @@ -0,0 +1,145 @@ +#!/usr/bin/env bash +set -euo pipefail +set -x + +# Qwen3.8-Flash-Next Quark MXFP4 AgentX on MI355X with SGLang native NEXTN +# MTP. Based on the upstream MI355X balanced FP8 recipe plus the cookbook's +# "NEXTN / MTP" speculative card (--speculative-algorithm NEXTN, 3 steps, +# eagle-topk 1, 4 draft tokens), the same shape the B200/B300/H200 +# qwen3.8next SGLang AgentX arms run: +# https://docs.sglang.io/cookbook/autoregressive/Qwen/Qwen3.8-Flash-Next +# Checkpoint: https://huggingface.co/amd/Qwen3.8-Flash-Next-Quark-MXFP4 +# Quantization is read from the checkpoint (Quark MXFP4 MoE; BF16 PLE). The +# model ships its own multi-step-trained MTP head, so NEXTN needs no external +# drafter. Per the AgentX policy (MODELS.md) agentic recipes run with +# speculative decoding only: throughput pins acceptance to the committed +# golden AL, evals retain real target-model verification. + +source "$(dirname "$0")/../../benchmark_lib.sh" + +export EVAL_FRAMEWORK="lm-eval" +check_env_vars \ + MODEL TP CONC EP_SIZE KV_OFFLOADING \ + TOTAL_CPU_DRAM_GB RESULT_DIR DURATION +require_agentic_kv_offload_none + +if [[ "$EP_SIZE" != 1 ]]; then + echo "Error: this recipe supports EP_SIZE=1 only" >&2 + exit 1 +fi + +# Let HF validate/resume downloads even when a local directory is nonempty. +if [[ -n "${MODEL_PATH:-}" && "$MODEL_PATH" != "$MODEL" ]]; then + hf download "$MODEL" --local-dir "$MODEL_PATH" +else + hf download "$MODEL" + export MODEL_PATH="$MODEL" +fi + +rocm-smi || true +amd-smi || true + +export WEKA_LOADER_OVERRIDE=semianalysis_cc_traces_weka_062126_256k +resolve_trace_source +install_agentic_deps + +export AIPERF_SERVER_METRICS_URLS="http://localhost:${PORT}/metrics" +export AIPERF_REQUIRED_SERVER_METRIC_PREFIX="sglang:" + +mkdir -p "$RESULT_DIR" +SERVER_LOG="$RESULT_DIR/server.log" +SERVER_PID="" +cleanup_agentic_services() { + local exit_code=$? + trap - EXIT INT TERM + set +e + stop_background_process_tree "$SERVER_PID" "SGLang server" 60 + exit "$exit_code" +} +trap cleanup_agentic_services EXIT +trap 'exit 130' INT +trap 'exit 143' TERM + +SCHEDULER_RECV_INTERVAL=${SCHEDULER_RECV_INTERVAL:-30} +MAX_RUNNING_REQUESTS=$((2 * CONC)) +CUDA_GRAPH_MAX_BS="$CONC" +[ "$CUDA_GRAPH_MAX_BS" -gt 64 ] && CUDA_GRAPH_MAX_BS=64 + +TOKENIZER_ARGS=() +if [ "$TP" -ge 4 ]; then + TOKENIZER_ARGS=(--tokenizer-worker-num 6) +fi + +export PYTHONNOUSERSITE=1 +export SGLANG_USE_AITER=1 +export SGLANG_USE_AITER_UNIFIED_ATTN=1 +export AITER_FLYDSL_FORCE=1 +export SGLANG_MAMBA_SSM_DTYPE=bfloat16 +export SGLANG_TIMEOUT_KEEP_ALIVE=1800 + +if [ "${EVAL_ONLY:-false}" != "true" ]; then + # golden_al_distribution/qwen3.8next_mtp.yaml: + # qwen3.8-flash-next-fp8.thinking_on[3] = 2.32. --speculative-num-steps 3 + # with 4 draft tokens is 3 speculative tokens per verification step, i.e. + # the MTP=3 cell; AgentX replays run with thinking on. Same value as the + # B200/B300/H200 qwen3.8next SGLang arms. EVAL_ONLY leaves simulated + # acceptance off so evals score real verification. + export SGLANG_SIMULATE_ACC_LEN=2.32 + export SGLANG_SIMULATE_ACC_METHOD=match-expected + export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token +fi + +SGLANG_CMD=( + python3 -m sglang.launch_server + --model-path "$MODEL_PATH" + --served-model-name "$MODEL" + --host 0.0.0.0 + --port "$PORT" + --trust-remote-code + --tp-size "$TP" + --ep-size "$EP_SIZE" + --attention-backend aiter + --page-size 32 + --kv-cache-dtype auto + --chunked-prefill-size 16384 + --watchdog-timeout 1200 + # MTP: leave non-static headroom for the NEXTN draft head's verification + # batch and AITER spec-decode workspaces. The cookbook STP cell runs 0.9; + # the B300 NVFP4 MTP sibling runs 0.80 and the H200 FP8 one 0.85. The + # ~126 GiB checkpoint is ~16 GB/GPU across TP8 on 288 GB parts, so 0.85 + # still leaves a ~229 GB/GPU static share for weights plus KV. + --mem-fraction-static 0.85 + --model-loader-extra-config '{"enable_multithread_load": true}' + # NEXTN silently resets --max-running-requests to 48 when it is unset, so + # this must stay explicit and sized to the AgentX concurrency. + --max-running-requests "$MAX_RUNNING_REQUESTS" + # Decode-specific spelling as the MI355X Qwen3.5/DeepSeek-V4/GLM-5.2 SGLang + # MTP arms use; recent SGLang splits --cuda-graph-max-bs into + # decode/prefill variants and rejects the old prefix as ambiguous. + --cuda-graph-max-bs-decode "$CUDA_GRAPH_MAX_BS" + --speculative-algorithm NEXTN + --speculative-num-steps 3 + --speculative-eagle-topk 1 + --speculative-num-draft-tokens 4 + --stream-interval 50 + --scheduler-recv-interval "$SCHEDULER_RECV_INTERVAL" + "${TOKENIZER_ARGS[@]}" + --tokenizer-path "$MODEL" + --enable-metrics + --enable-cache-report +) + +printf '%q ' "${SGLANG_CMD[@]}" | tee "$RESULT_DIR/sglang_command.txt" +printf '\n' | tee -a "$RESULT_DIR/sglang_command.txt" +"${SGLANG_CMD[@]}" > "$SERVER_LOG" 2>&1 & +SERVER_PID=$! + +wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" + +if [ "${EVAL_ONLY:-false}" = "true" ]; then + run_eval --port "$PORT" +else + build_replay_cmd "$RESULT_DIR" + REPLAY_CMD+=" --apply-chat-template" + run_agentic_replay_and_write_outputs "$RESULT_DIR" +fi diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index d6f6a2441d..43e417ffa3 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -329,6 +329,25 @@ qwen3.5-fp4-mi355x-sglang-agentic-mtp: - { tp: 2, ep: 1, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 4, 8, 12, 16, 20] } - { tp: 2, ep: 1, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [20, 24, 28, 32, 36, 40] } +# Quark MXFP4 MoE checkpoint with BF16 PLE, SGLang native NEXTN MTP +# (3 steps, eagle-topk 1, 4 draft tokens) pinned to the golden AL 2.32 +# (golden_al_distribution/qwen3.8next_mtp.yaml, thinking_on, K=3). Spec-decode +# only, per the AgentX policy; same shape as the B200/B300/H200 qwen3.8next +# SGLang AgentX arms. +qwen3.8next-fp4-mi355x-sglang-agentic-mtp: + image: lmsysorg/sglang-rocm:qwen38flashnext@sha256:f2e3928cd5be1d7bf9bfb783d57e9f5b35baa8e09d4248e5fc44da43c85c35e4 + model: amd/Qwen3.8-Flash-Next-Quark-MXFP4 + model-prefix: qwen3.8next + runner: cluster:mi355x-amds + precision: fp4 + framework: sglang + multinode: false + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - { tp: 8, ep: 1, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 4, 8, 12, 16] } + qwen3.5-fp4-mi355x-sglang-disagg: image: lmsysorg/sglang-rocm:v0.5.12.post1-rocm720-mi35x-20260523 model: amd/Qwen3.5-397B-A17B-MXFP4 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index fb73af4e77..4bdd744c94 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -7807,3 +7807,23 @@ - "Use FLASH_ATTN for the target model, select the Humming MXFP8 dense-linear backend, and enable Marlin atomic reduction for the MXFP8 MoE path." - "Use 0.95 GPU memory utilization for resident serving while retaining 0.90 for Mooncake DRAM offload, and remove the Mooncake concurrency-14 point." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3108 + +- config-keys: + - qwen3.8next-fp4-mi355x-sglang-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Add Qwen3.8-Flash-Next Quark MXFP4 AgentX on MI355X with SGLang, TP8/EP1 and concurrency 1/4/8/12/16." + - "Use the public AMD checkpoint, a digest-pinned upstream ROCm image, and no KV offloading." + - "Run with SGLang native NEXTN MTP (--speculative-algorithm NEXTN, 3 steps, eagle-topk 1, 4 draft tokens), the cookbook speculative card and the shape of the B200/B300/H200 qwen3.8next SGLang AgentX arms; spec-decode only per the AgentX policy. Throughput pins acceptance to the golden AL 2.32 (golden_al_distribution/qwen3.8next_mtp.yaml, thinking_on, K=3) through SGLANG_SIMULATE_ACC_*; EVAL_ONLY keeps real verification. --mem-fraction-static 0.85 (cookbook STP cell: 0.9) leaves headroom for the draft head, and the graph cap uses --cuda-graph-max-bs-decode like the other MI355X SGLang MTP arms." + - "使用 SGLang 原生 NEXTN MTP(3 步、eagle-topk 1、4 草稿 token),与 B200/B300/H200 的 qwen3.8next SGLang AgentX 配方同形;按 AgentX 政策仅运行投机解码。吞吐通过 SGLANG_SIMULATE_ACC_* 固定黄金 AL 2.32,评测保留真实验证;--mem-fraction-static 0.85,图捕获上限使用 --cuda-graph-max-bs-decode。" + - "Mount this recipe outside /workspace to preserve the image's SGLang and AITER packages." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3129 + +- config-keys: + - qwen3.8next-fp4-mi355x-sglang-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Translate this recipe's pinned Docker image to an Enroot-compatible digest URI on cold-cache imports; keep the original image identity and squash cache key." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3129 diff --git a/runners/launch_mi355x-amds.sh b/runners/launch_mi355x-amds.sh index d7e95711e4..1c4136ec5c 100644 --- a/runners/launch_mi355x-amds.sh +++ b/runners/launch_mi355x-amds.sh @@ -262,6 +262,14 @@ else PARTITION="compute" SQUASH_FILE="/var/lib/squash/$(echo "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" LOCK_FILE="${SQUASH_FILE}.lock" + ENROOT_IMAGE_URI="docker://$IMAGE" + # Enroot 3.x treats Docker's @digest as a username separator. Keep the + # image/cache identity pinned, but use its manifest-reference syntax. + if [[ "$MODEL" == "amd/Qwen3.8-Flash-Next-Quark-MXFP4" && + "$FRAMEWORK" == "sglang" && + "$IMAGE" == lmsysorg/sglang-rocm:*@sha256:* ]]; then + ENROOT_IMAGE_URI="docker://registry-1.docker.io#lmsysorg/sglang-rocm:${IMAGE##*@}" + fi export GPU_COUNT="${GPU_COUNT:-${TP:?TP must be set}}" @@ -279,7 +287,7 @@ else echo 'Squash file already exists and is valid, skipping import' else rm -f \"$SQUASH_FILE\" - enroot import -o \"$SQUASH_FILE\" docker://$IMAGE + enroot import -o \"$SQUASH_FILE\" \"$ENROOT_IMAGE_URI\" fi " @@ -316,6 +324,16 @@ else esac fi + # The Qwen bring-up image keeps SGLang and AITER under /workspace. + # Do not hide those packages with the repository bind mount. + if [[ "$MODEL" == "amd/Qwen3.8-Flash-Next-Quark-MXFP4" && "$FRAMEWORK" == "sglang" ]]; then + CONTAINER_REPO=/ix + export INFMAX_CONTAINER_WORKSPACE="$CONTAINER_REPO" + case "${RESULT_DIR:-}" in + /workspace/*) export RESULT_DIR="/ix/${RESULT_DIR#/workspace/}" ;; + esac + fi + SCRIPT_BASE="${EXP_NAME%%_*}_${PRECISION}_mi355x" SCRIPT_FW="benchmarks/single_node/${SCENARIO_SUBDIR:-fixed_seq_len/}${SCRIPT_BASE}_${FRAMEWORK}${SPEC_SUFFIX}.sh" SCRIPT_FALLBACK="benchmarks/single_node/${SCENARIO_SUBDIR:-fixed_seq_len/}${SCRIPT_BASE}${FRAMEWORK_SUFFIX}${SPEC_SUFFIX}.sh" diff --git a/runners/test_slurm_utils.py b/runners/test_slurm_utils.py index 9a12f274ac..d2a8c09a2a 100644 --- a/runners/test_slurm_utils.py +++ b/runners/test_slurm_utils.py @@ -414,21 +414,23 @@ def test_eval_only_acceptance_rewrite_allows_non_speculative_recipe( @pytest.mark.parametrize( - ("model", "prefix", "mount", "cache"), + ("model", "prefix", "mount", "cache", "framework", "spec"), [ - ("deepseek-ai/DeepSeek-V4.1-Flash", "dsv41flash", "/ix", "/it-share/hf-hub-cache/"), - ("deepseek-ai/DeepSeek-V4-Pro", "dsv4", "/workspace", "/it-share/hf-hub-cache/"), + ("deepseek-ai/DeepSeek-V4.1-Flash", "dsv41flash", "/ix", "/it-share/hf-hub-cache/", "vllm", "mtp"), + ("deepseek-ai/DeepSeek-V4-Pro", "dsv4", "/workspace", "/it-share/hf-hub-cache/", "vllm", "mtp"), + ("amd/Qwen3.8-Flash-Next-Quark-MXFP4", "qwen3.8next", "/ix", "/var/lib/hf-hub-cache/", "sglang", "mtp"), ], ) def test_mi355x_agentic_model_mount_and_routing( tmp_path: Path, model: str, prefix: str, mount: str, cache: str, + framework: str, spec: str, ) -> None: capture = tmp_path / "launch.txt" env = { **os.environ, "IS_MULTINODE": "false", "MODEL": model, - "EXP_NAME": f"{prefix}_tp4_conc1", "FRAMEWORK": "vllm", - "PRECISION": "fp4", "SPEC_DECODING": "mtp", + "EXP_NAME": f"{prefix}_tp4_conc1", "FRAMEWORK": framework, + "PRECISION": "fp4", "SPEC_DECODING": spec, "SCENARIO_SUBDIR": "agentic/", "TP": "4", "GPU_COUNT": "4", "RUNNER_NAME": "mi355x-amds_01", "IMAGE": "test/image:mock", "GITHUB_WORKSPACE": str(REPO_ROOT), "HF_HUB_CACHE": "/mnt/hf_hub_cache/", @@ -463,5 +465,83 @@ def test_mi355x_agentic_model_mount_and_routing( f"--container-mounts={REPO_ROOT}:{mount}/,{cache}:/mnt/hf_hub_cache/," "/it-share/aiperf-cache/:/aiperf_mmap_cache" ) in args - script = f"benchmarks/single_node/agentic/{prefix}_fp4_mi355x_vllm_mtp.sh" + suffix = "_mtp" if spec == "mtp" else "" + script = f"benchmarks/single_node/agentic/{prefix}_fp4_mi355x_{framework}{suffix}.sh" assert args[-2] == script + + +@pytest.mark.parametrize( + ("model", "framework", "image", "cache_valid", "expected_uri"), + [ + ("amd/Qwen3.8-Flash-Next-Quark-MXFP4", "sglang", + "lmsysorg/sglang-rocm:fixture@sha256:abc123", False, + "docker://registry-1.docker.io#lmsysorg/sglang-rocm:sha256:abc123"), + ("amd/Qwen3.8-Flash-Next-Quark-MXFP4", "sglang", + "lmsysorg/sglang-rocm:fixture@sha256:abc123", True, None), + ("unrelated/model", "sglang", + "lmsysorg/sglang-rocm:fixture@sha256:abc123", False, + "docker://lmsysorg/sglang-rocm:fixture@sha256:abc123"), + ("amd/Qwen3.8-Flash-Next-Quark-MXFP4", "vllm", + "lmsysorg/sglang-rocm:fixture@sha256:abc123", False, + "docker://lmsysorg/sglang-rocm:fixture@sha256:abc123"), + ("amd/Qwen3.8-Flash-Next-Quark-MXFP4", "sglang", + "lmsysorg/sglang-rocm:fixture", False, + "docker://lmsysorg/sglang-rocm:fixture"), + ("amd/Qwen3.8-Flash-Next-Quark-MXFP4", "sglang", + "other/image:fixture@sha256:abc123", False, + "docker://other/image:fixture@sha256:abc123"), + ], +) +def test_mi355x_import_executes_scoped_digest_uri( + tmp_path: Path, model: str, framework: str, image: str, + cache_valid: bool, expected_uri: str | None, +) -> None: + capture = tmp_path / "enroot.txt" + env = { + **os.environ, + "IS_MULTINODE": "false", "MODEL": model, "FRAMEWORK": framework, + "EXP_NAME": "qwen3.8next_tp8_conc1", "PRECISION": "fp4", + "SPEC_DECODING": "mtp", "SCENARIO_SUBDIR": "agentic/", "TP": "8", + "RUNNER_NAME": "mi355x-amds_01", "IMAGE": image, + "GITHUB_WORKSPACE": str(REPO_ROOT), "HF_HUB_CACHE": "/mnt/hf_hub_cache/", + "RESULT_DIR": "/workspace/results", "CAPTURE": str(capture), + "CACHE_ROOT": str(tmp_path), "CACHE_STATUS": "0" if cache_valid else "1", + } + result = subprocess.run( + ["bash", "-c", r''' + salloc() { :; } + squeue() { echo 123; } + scancel() { :; } + docker() { :; } + unsquashfs() { return "$CACHE_STATUS"; } + enroot() { printf '%s\n' "$@" > "$CAPTURE"; } + export -f docker unsquashfs enroot + srun() { + shift + if [[ "$1" == bash ]]; then + # Simulate the remote filesystem in scratch, then execute the + # launcher's actual import shell, including cache selection. + local remote_command="$3" + remote_command="${remote_command//\/var\/lib\/squash/$CACHE_ROOT}" + bash -c "$remote_command" + else + printf '%s\n' "$@" > "$CAPTURE.launch" + fi + } + source runners/launch_mi355x-amds.sh + '''], cwd=REPO_ROOT, env=env, capture_output=True, text=True, check=False, + ) + assert result.returncode == 0, result.stderr + if expected_uri is None: + assert not capture.exists() + assert "skipping import" in result.stdout + else: + args = capture.read_text().splitlines() + assert args[:2] == ["import", "-o"] + assert args[3:] == [expected_uri] + if image == "lmsysorg/sglang-rocm:fixture@sha256:abc123": + assert Path(args[2]).name == "lmsysorg_sglang-rocm_fixture_sha256_abc123.sqsh" + assert any( + arg.startswith("--container-image=/var/lib/squash/") + for arg in Path(f"{capture}.launch").read_text().splitlines() + )