Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
145 changes: 145 additions & 0 deletions benchmarks/single_node/agentic/qwen3.8next_fp4_mi355x_sglang_mtp.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,145 @@
#!/usr/bin/env bash
set -euo pipefail
set -x

# Qwen3.8-Flash-Next Quark MXFP4 AgentX on MI355X with SGLang native NEXTN
# MTP. Based on the upstream MI355X balanced FP8 recipe plus the cookbook's
# "NEXTN / MTP" speculative card (--speculative-algorithm NEXTN, 3 steps,
# eagle-topk 1, 4 draft tokens), the same shape the B200/B300/H200
# qwen3.8next SGLang AgentX arms run:
# https://docs.sglang.io/cookbook/autoregressive/Qwen/Qwen3.8-Flash-Next
# Checkpoint: https://huggingface.co/amd/Qwen3.8-Flash-Next-Quark-MXFP4
# Quantization is read from the checkpoint (Quark MXFP4 MoE; BF16 PLE). The
# model ships its own multi-step-trained MTP head, so NEXTN needs no external
# drafter. Per the AgentX policy (MODELS.md) agentic recipes run with
# speculative decoding only: throughput pins acceptance to the committed
# golden AL, evals retain real target-model verification.

source "$(dirname "$0")/../../benchmark_lib.sh"

export EVAL_FRAMEWORK="lm-eval"
check_env_vars \
MODEL TP CONC EP_SIZE KV_OFFLOADING \
TOTAL_CPU_DRAM_GB RESULT_DIR DURATION
require_agentic_kv_offload_none

if [[ "$EP_SIZE" != 1 ]]; then
echo "Error: this recipe supports EP_SIZE=1 only" >&2
exit 1
fi

# Let HF validate/resume downloads even when a local directory is nonempty.
if [[ -n "${MODEL_PATH:-}" && "$MODEL_PATH" != "$MODEL" ]]; then
hf download "$MODEL" --local-dir "$MODEL_PATH"
else
hf download "$MODEL"
export MODEL_PATH="$MODEL"
fi

rocm-smi || true
amd-smi || true

export WEKA_LOADER_OVERRIDE=semianalysis_cc_traces_weka_062126_256k
resolve_trace_source
install_agentic_deps

export AIPERF_SERVER_METRICS_URLS="http://localhost:${PORT}/metrics"
export AIPERF_REQUIRED_SERVER_METRIC_PREFIX="sglang:"

mkdir -p "$RESULT_DIR"
SERVER_LOG="$RESULT_DIR/server.log"
SERVER_PID=""
cleanup_agentic_services() {
local exit_code=$?
trap - EXIT INT TERM
set +e
stop_background_process_tree "$SERVER_PID" "SGLang server" 60
exit "$exit_code"
}
trap cleanup_agentic_services EXIT
trap 'exit 130' INT
trap 'exit 143' TERM

SCHEDULER_RECV_INTERVAL=${SCHEDULER_RECV_INTERVAL:-30}
MAX_RUNNING_REQUESTS=$((2 * CONC))
CUDA_GRAPH_MAX_BS="$CONC"
[ "$CUDA_GRAPH_MAX_BS" -gt 64 ] && CUDA_GRAPH_MAX_BS=64

TOKENIZER_ARGS=()
if [ "$TP" -ge 4 ]; then
TOKENIZER_ARGS=(--tokenizer-worker-num 6)
fi

export PYTHONNOUSERSITE=1
export SGLANG_USE_AITER=1
export SGLANG_USE_AITER_UNIFIED_ATTN=1
export AITER_FLYDSL_FORCE=1
export SGLANG_MAMBA_SSM_DTYPE=bfloat16
export SGLANG_TIMEOUT_KEEP_ALIVE=1800

if [ "${EVAL_ONLY:-false}" != "true" ]; then
# golden_al_distribution/qwen3.8next_mtp.yaml:
# qwen3.8-flash-next-fp8.thinking_on[3] = 2.32. --speculative-num-steps 3
# with 4 draft tokens is 3 speculative tokens per verification step, i.e.
# the MTP=3 cell; AgentX replays run with thinking on. Same value as the
# B200/B300/H200 qwen3.8next SGLang arms. EVAL_ONLY leaves simulated
# acceptance off so evals score real verification.
export SGLANG_SIMULATE_ACC_LEN=2.32
export SGLANG_SIMULATE_ACC_METHOD=match-expected
export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token
fi

SGLANG_CMD=(
python3 -m sglang.launch_server
--model-path "$MODEL_PATH"
--served-model-name "$MODEL"
--host 0.0.0.0
--port "$PORT"
--trust-remote-code
--tp-size "$TP"
--ep-size "$EP_SIZE"
--attention-backend aiter
--page-size 32
--kv-cache-dtype auto
--chunked-prefill-size 16384
--watchdog-timeout 1200
# MTP: leave non-static headroom for the NEXTN draft head's verification
# batch and AITER spec-decode workspaces. The cookbook STP cell runs 0.9;
# the B300 NVFP4 MTP sibling runs 0.80 and the H200 FP8 one 0.85. The
# ~126 GiB checkpoint is ~16 GB/GPU across TP8 on 288 GB parts, so 0.85
# still leaves a ~229 GB/GPU static share for weights plus KV.
--mem-fraction-static 0.85
--model-loader-extra-config '{"enable_multithread_load": true}'
# NEXTN silently resets --max-running-requests to 48 when it is unset, so
# this must stay explicit and sized to the AgentX concurrency.
--max-running-requests "$MAX_RUNNING_REQUESTS"
# Decode-specific spelling as the MI355X Qwen3.5/DeepSeek-V4/GLM-5.2 SGLang
# MTP arms use; recent SGLang splits --cuda-graph-max-bs into
# decode/prefill variants and rejects the old prefix as ambiguous.
--cuda-graph-max-bs-decode "$CUDA_GRAPH_MAX_BS"
--speculative-algorithm NEXTN
--speculative-num-steps 3
--speculative-eagle-topk 1
--speculative-num-draft-tokens 4
--stream-interval 50
--scheduler-recv-interval "$SCHEDULER_RECV_INTERVAL"
"${TOKENIZER_ARGS[@]}"
--tokenizer-path "$MODEL"
--enable-metrics
--enable-cache-report
)

printf '%q ' "${SGLANG_CMD[@]}" | tee "$RESULT_DIR/sglang_command.txt"
printf '\n' | tee -a "$RESULT_DIR/sglang_command.txt"
"${SGLANG_CMD[@]}" > "$SERVER_LOG" 2>&1 &
SERVER_PID=$!

wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID"

if [ "${EVAL_ONLY:-false}" = "true" ]; then
run_eval --port "$PORT"
else
build_replay_cmd "$RESULT_DIR"
REPLAY_CMD+=" --apply-chat-template"
run_agentic_replay_and_write_outputs "$RESULT_DIR"
fi
19 changes: 19 additions & 0 deletions configs/amd-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -329,6 +329,25 @@ qwen3.5-fp4-mi355x-sglang-agentic-mtp:
- { tp: 2, ep: 1, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 4, 8, 12, 16, 20] }
- { tp: 2, ep: 1, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [20, 24, 28, 32, 36, 40] }

# Quark MXFP4 MoE checkpoint with BF16 PLE, SGLang native NEXTN MTP
# (3 steps, eagle-topk 1, 4 draft tokens) pinned to the golden AL 2.32
# (golden_al_distribution/qwen3.8next_mtp.yaml, thinking_on, K=3). Spec-decode
# only, per the AgentX policy; same shape as the B200/B300/H200 qwen3.8next
# SGLang AgentX arms.
qwen3.8next-fp4-mi355x-sglang-agentic-mtp:
image: lmsysorg/sglang-rocm:qwen38flashnext@sha256:f2e3928cd5be1d7bf9bfb783d57e9f5b35baa8e09d4248e5fc44da43c85c35e4
model: amd/Qwen3.8-Flash-Next-Quark-MXFP4
model-prefix: qwen3.8next
runner: cluster:mi355x-amds
precision: fp4
framework: sglang
multinode: false
scenarios:
agentic-coding:
- dram-utilization: 0.80
search-space:
- { tp: 8, ep: 1, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 4, 8, 12, 16] }

qwen3.5-fp4-mi355x-sglang-disagg:
image: lmsysorg/sglang-rocm:v0.5.12.post1-rocm720-mi35x-20260523
model: amd/Qwen3.5-397B-A17B-MXFP4
Expand Down
20 changes: 20 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -7807,3 +7807,23 @@
- "Use FLASH_ATTN for the target model, select the Humming MXFP8 dense-linear backend, and enable Marlin atomic reduction for the MXFP8 MoE path."
- "Use 0.95 GPU memory utilization for resident serving while retaining 0.90 for Mooncake DRAM offload, and remove the Mooncake concurrency-14 point."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3108

- config-keys:
- qwen3.8next-fp4-mi355x-sglang-agentic-mtp
scenario-type:
- agentic-coding
description:
- "Add Qwen3.8-Flash-Next Quark MXFP4 AgentX on MI355X with SGLang, TP8/EP1 and concurrency 1/4/8/12/16."
- "Use the public AMD checkpoint, a digest-pinned upstream ROCm image, and no KV offloading."
- "Run with SGLang native NEXTN MTP (--speculative-algorithm NEXTN, 3 steps, eagle-topk 1, 4 draft tokens), the cookbook speculative card and the shape of the B200/B300/H200 qwen3.8next SGLang AgentX arms; spec-decode only per the AgentX policy. Throughput pins acceptance to the golden AL 2.32 (golden_al_distribution/qwen3.8next_mtp.yaml, thinking_on, K=3) through SGLANG_SIMULATE_ACC_*; EVAL_ONLY keeps real verification. --mem-fraction-static 0.85 (cookbook STP cell: 0.9) leaves headroom for the draft head, and the graph cap uses --cuda-graph-max-bs-decode like the other MI355X SGLang MTP arms."
- "使用 SGLang 原生 NEXTN MTP(3 步、eagle-topk 1、4 草稿 token),与 B200/B300/H200 的 qwen3.8next SGLang AgentX 配方同形;按 AgentX 政策仅运行投机解码。吞吐通过 SGLANG_SIMULATE_ACC_* 固定黄金 AL 2.32,评测保留真实验证;--mem-fraction-static 0.85,图捕获上限使用 --cuda-graph-max-bs-decode。"
- "Mount this recipe outside /workspace to preserve the image's SGLang and AITER packages."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3129

- config-keys:
- qwen3.8next-fp4-mi355x-sglang-agentic-mtp
scenario-type:
- agentic-coding
description:
- "Translate this recipe's pinned Docker image to an Enroot-compatible digest URI on cold-cache imports; keep the original image identity and squash cache key."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3129
20 changes: 19 additions & 1 deletion runners/launch_mi355x-amds.sh
Original file line number Diff line number Diff line change
Expand Up @@ -262,6 +262,14 @@ else
PARTITION="compute"
SQUASH_FILE="/var/lib/squash/$(echo "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh"
LOCK_FILE="${SQUASH_FILE}.lock"
ENROOT_IMAGE_URI="docker://$IMAGE"
# Enroot 3.x treats Docker's @digest as a username separator. Keep the
# image/cache identity pinned, but use its manifest-reference syntax.
if [[ "$MODEL" == "amd/Qwen3.8-Flash-Next-Quark-MXFP4" &&
"$FRAMEWORK" == "sglang" &&
"$IMAGE" == lmsysorg/sglang-rocm:*@sha256:* ]]; then
ENROOT_IMAGE_URI="docker://registry-1.docker.io#lmsysorg/sglang-rocm:${IMAGE##*@}"
fi

export GPU_COUNT="${GPU_COUNT:-${TP:?TP must be set}}"

Expand All @@ -279,7 +287,7 @@ else
echo 'Squash file already exists and is valid, skipping import'
else
rm -f \"$SQUASH_FILE\"
enroot import -o \"$SQUASH_FILE\" docker://$IMAGE
enroot import -o \"$SQUASH_FILE\" \"$ENROOT_IMAGE_URI\"
fi
"

Expand Down Expand Up @@ -316,6 +324,16 @@ else
esac
fi

# The Qwen bring-up image keeps SGLang and AITER under /workspace.
# Do not hide those packages with the repository bind mount.
if [[ "$MODEL" == "amd/Qwen3.8-Flash-Next-Quark-MXFP4" && "$FRAMEWORK" == "sglang" ]]; then
CONTAINER_REPO=/ix
export INFMAX_CONTAINER_WORKSPACE="$CONTAINER_REPO"
case "${RESULT_DIR:-}" in
/workspace/*) export RESULT_DIR="/ix/${RESULT_DIR#/workspace/}" ;;
esac
fi

SCRIPT_BASE="${EXP_NAME%%_*}_${PRECISION}_mi355x"
SCRIPT_FW="benchmarks/single_node/${SCENARIO_SUBDIR:-fixed_seq_len/}${SCRIPT_BASE}_${FRAMEWORK}${SPEC_SUFFIX}.sh"
SCRIPT_FALLBACK="benchmarks/single_node/${SCENARIO_SUBDIR:-fixed_seq_len/}${SCRIPT_BASE}${FRAMEWORK_SUFFIX}${SPEC_SUFFIX}.sh"
Expand Down
92 changes: 86 additions & 6 deletions runners/test_slurm_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -414,21 +414,23 @@ def test_eval_only_acceptance_rewrite_allows_non_speculative_recipe(


@pytest.mark.parametrize(
("model", "prefix", "mount", "cache"),
("model", "prefix", "mount", "cache", "framework", "spec"),
[
("deepseek-ai/DeepSeek-V4.1-Flash", "dsv41flash", "/ix", "/it-share/hf-hub-cache/"),
("deepseek-ai/DeepSeek-V4-Pro", "dsv4", "/workspace", "/it-share/hf-hub-cache/"),
("deepseek-ai/DeepSeek-V4.1-Flash", "dsv41flash", "/ix", "/it-share/hf-hub-cache/", "vllm", "mtp"),
("deepseek-ai/DeepSeek-V4-Pro", "dsv4", "/workspace", "/it-share/hf-hub-cache/", "vllm", "mtp"),
("amd/Qwen3.8-Flash-Next-Quark-MXFP4", "qwen3.8next", "/ix", "/var/lib/hf-hub-cache/", "sglang", "mtp"),
],
)
def test_mi355x_agentic_model_mount_and_routing(
tmp_path: Path, model: str, prefix: str, mount: str, cache: str,
framework: str, spec: str,
) -> None:
capture = tmp_path / "launch.txt"
env = {
**os.environ,
"IS_MULTINODE": "false", "MODEL": model,
"EXP_NAME": f"{prefix}_tp4_conc1", "FRAMEWORK": "vllm",
"PRECISION": "fp4", "SPEC_DECODING": "mtp",
"EXP_NAME": f"{prefix}_tp4_conc1", "FRAMEWORK": framework,
"PRECISION": "fp4", "SPEC_DECODING": spec,
"SCENARIO_SUBDIR": "agentic/", "TP": "4", "GPU_COUNT": "4",
"RUNNER_NAME": "mi355x-amds_01", "IMAGE": "test/image:mock",
"GITHUB_WORKSPACE": str(REPO_ROOT), "HF_HUB_CACHE": "/mnt/hf_hub_cache/",
Expand Down Expand Up @@ -463,5 +465,83 @@ def test_mi355x_agentic_model_mount_and_routing(
f"--container-mounts={REPO_ROOT}:{mount}/,{cache}:/mnt/hf_hub_cache/,"
"/it-share/aiperf-cache/:/aiperf_mmap_cache"
) in args
script = f"benchmarks/single_node/agentic/{prefix}_fp4_mi355x_vllm_mtp.sh"
suffix = "_mtp" if spec == "mtp" else ""
script = f"benchmarks/single_node/agentic/{prefix}_fp4_mi355x_{framework}{suffix}.sh"
assert args[-2] == script


@pytest.mark.parametrize(
("model", "framework", "image", "cache_valid", "expected_uri"),
[
("amd/Qwen3.8-Flash-Next-Quark-MXFP4", "sglang",
"lmsysorg/sglang-rocm:fixture@sha256:abc123", False,
"docker://registry-1.docker.io#lmsysorg/sglang-rocm:sha256:abc123"),
("amd/Qwen3.8-Flash-Next-Quark-MXFP4", "sglang",
"lmsysorg/sglang-rocm:fixture@sha256:abc123", True, None),
("unrelated/model", "sglang",
"lmsysorg/sglang-rocm:fixture@sha256:abc123", False,
"docker://lmsysorg/sglang-rocm:fixture@sha256:abc123"),
("amd/Qwen3.8-Flash-Next-Quark-MXFP4", "vllm",
"lmsysorg/sglang-rocm:fixture@sha256:abc123", False,
"docker://lmsysorg/sglang-rocm:fixture@sha256:abc123"),
("amd/Qwen3.8-Flash-Next-Quark-MXFP4", "sglang",
"lmsysorg/sglang-rocm:fixture", False,
"docker://lmsysorg/sglang-rocm:fixture"),
("amd/Qwen3.8-Flash-Next-Quark-MXFP4", "sglang",
"other/image:fixture@sha256:abc123", False,
"docker://other/image:fixture@sha256:abc123"),
],
)
def test_mi355x_import_executes_scoped_digest_uri(
tmp_path: Path, model: str, framework: str, image: str,
cache_valid: bool, expected_uri: str | None,
) -> None:
capture = tmp_path / "enroot.txt"
env = {
**os.environ,
"IS_MULTINODE": "false", "MODEL": model, "FRAMEWORK": framework,
"EXP_NAME": "qwen3.8next_tp8_conc1", "PRECISION": "fp4",
"SPEC_DECODING": "mtp", "SCENARIO_SUBDIR": "agentic/", "TP": "8",
"RUNNER_NAME": "mi355x-amds_01", "IMAGE": image,
"GITHUB_WORKSPACE": str(REPO_ROOT), "HF_HUB_CACHE": "/mnt/hf_hub_cache/",
"RESULT_DIR": "/workspace/results", "CAPTURE": str(capture),
"CACHE_ROOT": str(tmp_path), "CACHE_STATUS": "0" if cache_valid else "1",
}
result = subprocess.run(
["bash", "-c", r'''
salloc() { :; }
squeue() { echo 123; }
scancel() { :; }
docker() { :; }
unsquashfs() { return "$CACHE_STATUS"; }
enroot() { printf '%s\n' "$@" > "$CAPTURE"; }
export -f docker unsquashfs enroot
srun() {
shift
if [[ "$1" == bash ]]; then
# Simulate the remote filesystem in scratch, then execute the
# launcher's actual import shell, including cache selection.
local remote_command="$3"
remote_command="${remote_command//\/var\/lib\/squash/$CACHE_ROOT}"
bash -c "$remote_command"
else
printf '%s\n' "$@" > "$CAPTURE.launch"
fi
}
source runners/launch_mi355x-amds.sh
'''], cwd=REPO_ROOT, env=env, capture_output=True, text=True, check=False,
)
assert result.returncode == 0, result.stderr
if expected_uri is None:
assert not capture.exists()
assert "skipping import" in result.stdout
else:
args = capture.read_text().splitlines()
assert args[:2] == ["import", "-o"]
assert args[3:] == [expected_uri]
if image == "lmsysorg/sglang-rocm:fixture@sha256:abc123":
assert Path(args[2]).name == "lmsysorg_sglang-rocm_fixture_sha256_abc123.sqsh"
assert any(
arg.startswith("--container-image=/var/lib/squash/")
for arg in Path(f"{capture}.launch").read_text().splitlines()
)
Loading