Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
26 commits
Select commit Hold shift + click to select a range
e868f8c
[AMD][MI355X] Add DSv4 FP4 agentic MTP recipe with DPA tuning
seungrokj Jul 17, 2026
281087a
fix(changelog): add pr-link for dsv4-fp4-mi355x-sglang-agentic-mtp entry
seungrokj Jul 17, 2026
04529c7
fix(changelog): trim verbose description for agentic-mtp entry
seungrokj Jul 17, 2026
83530a7
fix(config): remove duplicate dsv4-fp4-mi355x-sglang-agentic-hicache …
seungrokj Jul 17, 2026
e10fa87
fix(validation): add run-eval/eval-only fields to agentic matrix entries
seungrokj Jul 17, 2026
85f425a
fix(agentic-mtp): tune DPA memory, prefill chunk, and batch sizing
seungrokj Jul 18, 2026
090cd59
fix(agentic-mtp): revert mem_fraction_static override, narrow sweep t…
seungrokj Jul 18, 2026
441b580
fix(agentic-mtp): stop dividing CUDA_GRAPH_MAX_BS by TP for DPA
seungrokj Jul 18, 2026
e162550
fix(agentic-mtp): restore MAX_RUNNING_REQUESTS to 2*CONC
seungrokj Jul 18, 2026
522b3d3
fix(agentic-mtp): reduce chunked prefill back to 8K per scheduler
seungrokj Jul 18, 2026
9d2810a
fix(agentic-mtp): restore GPU_MAX_HW_QUEUES=5 for ROCm overlap
seungrokj Jul 18, 2026
7a41734
fix(agentic-mtp): clean up script to match ref, expand sweep concurre…
seungrokj Jul 19, 2026
15a8fea
fix(validation): remove run_eval/eval_only from MultiNodeAgenticMatri…
seungrokj Jul 19, 2026
7d1713f
fix(agentic-mtp): add MAX_RUNNING_REQUESTS=2*CONC with comment
seungrokj Jul 19, 2026
a68e8bd
fix(agentic-mtp): drop c52 from sweep to avoid GPU OOM
seungrokj Jul 19, 2026
c9d41c1
feat(agentic-mtp): add cache metrics capture and eval-only support
seungrokj Jul 19, 2026
9facbfe
fix(agentic-mtp): disable default thinking to avoid context overflow
seungrokj Jul 20, 2026
3729411
fix(agentic-mtp): also disable reasoning_effort=high
seungrokj Jul 20, 2026
cc5888d
fix(agentic-mtp): re-enable thinking mode for SWE-bench quality
seungrokj Jul 21, 2026
f4293c6
fix(agentic-mtp): replace prefill-delayer with dp-attention-local-con…
seungrokj Jul 21, 2026
0b79406
merge origin/main into amd/agentx_dsv4_sgl_mtp_0717 and resolve perf-…
seungrokj Jul 21, 2026
ed29f29
Merge branch 'main' into amd/agentx_dsv4_sgl_mtp_0717
seungrokj Jul 21, 2026
ad0e95d
exclude mia1-p01-g09,mia1-p01-g11 from MI355X salloc
seungrokj Jul 21, 2026
712e17a
Merge branch 'main' of https://github.com/SemiAnalysisAI/InferenceX i…
seungrokj Jul 22, 2026
79eafda
[AgentX] DSv4 FP4 MI355X SGLang MTP agentic: drop conc-48 from sweep
seungrokj Jul 22, 2026
6839630
Merge branch 'main' into amd/agentx_dsv4_sgl_mtp_0717
seungrokj Jul 22, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
217 changes: 217 additions & 0 deletions benchmarks/single_node/agentic/dsv4_fp4_mi355x_sglang_mtp.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,217 @@
#!/usr/bin/env bash
set -eo pipefail
set -x

# Agentic trace replay benchmark for DeepSeek-V4-Pro FP4 on MI355X using SGLang.
#
# KV_OFFLOADING=dram requires KV_OFFLOAD_BACKEND=hicache.
#
# Required env vars:
# MODEL, TP, CONC, KV_OFFLOADING, TOTAL_CPU_DRAM_GB, RESULT_DIR
#
# KV_OFFLOADING=dram requires one of these.
# KV_OFFLOAD_BACKEND=hicache.

source "$(dirname "$0")/../../benchmark_lib.sh"

check_env_vars MODEL TP CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION EP_SIZE DP_ATTENTION

if [[ -n "$SLURM_JOB_ID" ]]; then
echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME"
fi

# ROCR/HIP visibility under slurm cgroups.
if [ -n "$ROCR_VISIBLE_DEVICES" ]; then
export HIP_VISIBLE_DEVICES="$ROCR_VISIBLE_DEVICES"
fi

if [[ -n "$MODEL_PATH" ]]; then
if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then
hf download "$MODEL" --local-dir "$MODEL_PATH"
fi
else
hf download "$MODEL"
export MODEL_PATH="$MODEL"
fi
rocm-smi || true
amd-smi || true

# ---- Resolve traces and install deps ----------------------------------------
resolve_trace_source
install_agentic_deps

# ---- Server config ----------------------------------------------------------
SERVER_LOG="$RESULT_DIR/server.log"
mkdir -p "$RESULT_DIR"

CACHE_ARGS=()
if agentic_kv_offload_enabled; then
# HiCache config — https://lmsysorg.mintlify.app/cookbook/autoregressive/DeepSeek/DeepSeek-V4
case "$KV_OFFLOAD_BACKEND" in
hicache)
HICACHE_RATIO=4
HICACHE_WRITE_POLICY="write_through"
HICACHE_IO_BACKEND="direct"
HICACHE_MEM_LAYOUT="page_first_direct"
CACHE_ARGS=(
--enable-hierarchical-cache
--hicache-ratio "$HICACHE_RATIO"
--hicache-write-policy "$HICACHE_WRITE_POLICY"
--hicache-io-backend "$HICACHE_IO_BACKEND"
--hicache-mem-layout "$HICACHE_MEM_LAYOUT"
)
echo "HiCache DSv4 CPU tier: ratio=$HICACHE_RATIO, write_policy=$HICACHE_WRITE_POLICY, io_backend=$HICACHE_IO_BACKEND, mem_layout=$HICACHE_MEM_LAYOUT"
;;
*)
echo "Error: unsupported KV_OFFLOAD_BACKEND '$KV_OFFLOAD_BACKEND' (expected: hicache)" >&2
exit 1
;;
esac
fi
# ---- Client config ----------------------------------------------------------
export AIPERF_HTTP_TCP_USER_TIMEOUT=1000000

# ---- LLM server config ----------------------------------------------------------
USE_SGLANG_ROUTER=false
SGLANG_BACKEND_PORT="$PORT"
ROUTER_LOG="$RESULT_DIR/router.log"
MEM_FRACTION_STATIC=0.90
CHUNKED_PREFILL_SIZE=8192
PARALLEL_ARGS=(--tensor-parallel-size "$TP")
if [ "$DP_ATTENTION" = "true" ]; then
USE_SGLANG_ROUTER=true
export AIPERF_HTTP_X_SMG_ROUTING_KEY_FROM_CORRELATION_ID=true
SGLANG_BACKEND_PORT=$((PORT + 1))
SGLANG_ROUTER_METRICS_PORT=$((PORT + 10000))
SGLANG_ROUTER_CMD=(python3 -m sglang_router.launch_router)

export SGLANG_SHARED_EXPERT_TP1=1
export SGLANG_DP_SHARED_EXPERT_LOCAL=1
export SGLANG_DP_USE_GATHERV=1
export SGLANG_DP_USE_REDUCE_SCATTER=1
export GPU_MAX_HW_QUEUES=5

CHUNKED_PREFILL_SIZE=$((8192 * TP))
PARALLEL_ARGS+=(
--dp "$TP"
--enable-dp-attention
--enable-dp-attention-local-control-broadcast
)
fi

if [ "$EP_SIZE" -gt 1 ]; then
PARALLEL_ARGS+=(--ep-size "$EP_SIZE")
fi
# AgentX concurrency counts live session trees, not individual requests.
# Allow subagent fan-out to exceed CONC without clipping request bursts.
MAX_RUNNING_REQUESTS=$((2 * CONC))
CUDA_GRAPH_MAX_BS=$CONC
[ "$CUDA_GRAPH_MAX_BS" -gt 128 ] && CUDA_GRAPH_MAX_BS=128
# Simulated acceptance-length (AL) settings.
# openai.BadRequestError: Error code: 400 - {'object': 'error', 'message': "The input (3620936 tokens) is longer than the model's context length (1048576 tokens).", 'type': 'BadRequestError', 'param': None, 'code': 400}
export SGLANG_DEFAULT_THINKING=1
export SGLANG_DSV4_REASONING_EFFORT=high
export SGLANG_SIMULATE_ACC_LEN=2.49
export SGLANG_SIMULATE_ACC_METHOD=match-expected
export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token

export SGLANG_USE_ROCM700A=0
export SGLANG_HACK_FLASHMLA_BACKEND=unified_kv_triton
export AITER_BF16_FP8_MOE_BOUND=0

export SGLANG_ENABLE_UNIFIED_RADIX_TREE=1
export SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS=1

METRICS_ARGS=(--enable-metrics)
SPEC_ARGS=(
--speculative-algorithm EAGLE
--speculative-num-steps 3
--speculative-eagle-topk 1
--speculative-num-draft-tokens 4
)

SGLANG_CMD=(
python3 -m sglang.launch_server
--model-path "$MODEL_PATH"
--served-model-name "$MODEL"
--host 0.0.0.0
--port "$SGLANG_BACKEND_PORT"
--trust-remote-code
"${PARALLEL_ARGS[@]}"
--attention-backend compressed
--cuda-graph-max-bs-decode "$CUDA_GRAPH_MAX_BS"
--max-running-requests "$MAX_RUNNING_REQUESTS"
--mem-fraction-static "$MEM_FRACTION_STATIC"
--swa-full-tokens-ratio 0.10
--page-size 256
--kv-cache-dtype fp8_e4m3
--chunked-prefill-size "$CHUNKED_PREFILL_SIZE"
--disable-shared-experts-fusion
--tool-call-parser deepseekv4
--reasoning-parser deepseek-v4
--chat-template "$(dirname "$0")/../chat_templates/deepseek_v4_thinking.jinja"
--watchdog-timeout 1800
"${METRICS_ARGS[@]}"
"${SPEC_ARGS[@]}"
"${CACHE_ARGS[@]}"
)

printf '%q ' "${SGLANG_CMD[@]}" | tee "$RESULT_DIR/sglang_command.txt"
printf '\n' | tee -a "$RESULT_DIR/sglang_command.txt"

{
echo "=== SGLANG_* env vars at launch ==="
env | grep -E '^SGLANG_' | sort
echo "==================================="
} | tee "$SERVER_LOG"

echo "Starting SGLang server for MI355X..."
"${SGLANG_CMD[@]}" >> "$SERVER_LOG" 2>&1 &
SERVER_PID=$!
echo "Server PID: $SERVER_PID"

capture_cache_metrics() {
{
echo "=== SGLang cache metrics snapshot $(date --iso-8601=seconds) ==="
curl -fsS "http://localhost:$SGLANG_BACKEND_PORT/metrics" 2>/dev/null \
| grep -E '^(sglang:(cache_hit_rate|cached_tokens_total|prompt_tokens_total|hicache_host_used_tokens|hicache_host_total_tokens|token_usage|num_requests_running|num_requests_waiting))' \
|| true
echo "============================================================"
} >> "$SERVER_LOG"
}

wait_for_server_ready --port "$SGLANG_BACKEND_PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID"

if [ "$USE_SGLANG_ROUTER" = "true" ]; then
echo "Starting SGLang router on port $PORT for $TP DP ranks..."
"${SGLANG_ROUTER_CMD[@]}" \
--worker-urls "http://localhost:$SGLANG_BACKEND_PORT" \
--policy consistent_hashing \
--request-id-headers x-correlation-id \
--dp-aware \
--host 0.0.0.0 \
--port "$PORT" \
--prometheus-host 127.0.0.1 \
--prometheus-port "$SGLANG_ROUTER_METRICS_PORT" \
--connect-timeout-secs 900 \
--request-timeout-secs 14400 \
--disable-health-check \
--disable-retries > "$ROUTER_LOG" 2>&1 &
ROUTER_PID=$!
echo "Router PID: $ROUTER_PID"
wait_for_server_ready --port "$PORT" --server-log "$ROUTER_LOG" --server-pid "$ROUTER_PID"
fi
# ---- Run benchmark ----------------------------------------------------------

if [ "${#METRICS_ARGS[@]}" -gt 0 ]; then
capture_cache_metrics
trap capture_cache_metrics EXIT
fi

if [ "${EVAL_ONLY}" = "true" ]; then
run_eval --port "$PORT"
else
build_replay_cmd "$RESULT_DIR"
REPLAY_CMD+=" --server-metrics http://localhost:$SGLANG_BACKEND_PORT/metrics"
run_agentic_replay_and_write_outputs "$RESULT_DIR"
fi
14 changes: 14 additions & 0 deletions configs/amd-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -1568,6 +1568,20 @@ dsv4-fp4-mi355x-sglang-agentic-hicache:
- { tp: 8, dp-attn: true, kv-offloading: none, conc-list: [16, 32, 48, 64] }
- { tp: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [64] }

dsv4-fp4-mi355x-sglang-agentic-mtp:
image: lmsysorg/sglang-rocm:v0.5.15.post1-rocm720-mi35x-20260714
model: deepseek-ai/DeepSeek-V4-Pro
model-prefix: dsv4
runner: cluster:mi355x-amds
precision: fp4
framework: sglang
multinode: false
scenarios:
agentic-coding:
- dram-utilization: 0.80
search-space:
- { tp: 8, dp-attn: true, kv-offloading: none, conc-list: [4, 8, 16, 32], spec-decoding: mtp }

# MiniMax-M3 MXFP8 MI355X recipe:
# https://github.com/vllm-project/recipes/commit/2a3728ed9892debfd767a72a58ebc90b33f186e5
# MXFP8 runs from TP=4 on gfx950; block size 128 is mandatory for MSA.
Expand Down
7 changes: 6 additions & 1 deletion perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -4986,6 +4986,12 @@
- "HiCache spills evicted prefixes to host DRAM and restores them at C2C bandwidth instead of recomputing; sizing follows the qwen3.5-fp8-b300-sglang-agentic-hicache recipe (GLM-5.2 is plain GQA: one host pool per rank, GB-based --hicache-size)"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2280

- config-keys:
- dsv4-fp4-mi355x-sglang-agentic-mtp
description:
- "Add DSv4 FP4 MI355X SGLang agentic MTP recipe (separate from hicache config)"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2254

- config-keys:
- glm5.2-fp4-b300-sglang-agentic
description:
Expand Down Expand Up @@ -5013,7 +5019,6 @@
- "LMCache MP server (lmcache_driven transfer mode) + LMCacheMPConnector per PR #2153/#2231; L1 pool derated to 75% of TOTAL_CPU_DRAM_GB; PYTORCH_ALLOC_CONF=expandable_segments:True dropped on the lmcache arms only (VMM allocations cannot be CUDA-IPC-exported to the LMCache server, same failure mode as cuMem on B200); gpu-memory-utilization 0.92 on lmcache DEP8 (matches the official DEP8 derate) and 0.94 on lmcache TP4/DEP4 (at 0.96 the PR #2232 bring-up sweeps hit torch-pool, DeepGEMM-JIT, and cuBLAS-workspace OOMs; 23 lmcache points validated green in runs 29463061871/29535333851 after the derate)"
- "LMCache ladders offset the official arms: TP4 +4 conc [32, 36, 40, 44] vs SimpleCPU [28, 32, 36, 40]; DEP4 +4 conc [36, 44, 52, 60, 68, 76] vs SimpleCPU [32, 40, 48, 56, 64, 72]; DEP8 +8 conc capped at 208 [72, 104, 120, 136, 152, 168, 184, 200] vs GPU-resident [64, 96, 112, 128, 144, 160, 176, 192, 224]"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2232


- config-keys:
- dsv4-fp4-b200-vllm-agentic
Expand Down
2 changes: 1 addition & 1 deletion runners/launch_mi355x-amds.sh
Original file line number Diff line number Diff line change
Expand Up @@ -268,7 +268,7 @@ else
export GPU_COUNT="${GPU_COUNT:-${TP:?TP must be set}}"

set -x
salloc --partition=$PARTITION --gres=gpu:$GPU_COUNT --exclusive --cpus-per-task=128 --time=500 --no-shell --job-name="$RUNNER_NAME"
salloc --partition=$PARTITION --gres=gpu:$GPU_COUNT --exclusive --cpus-per-task=128 --time=500 --no-shell --job-name="$RUNNER_NAME" --exclude=mia1-p01-g09,mia1-p01-g11
JOB_ID=$(squeue --name="$RUNNER_NAME" -h -o %A | head -n1)

srun --jobid=$JOB_ID bash -c "docker stop \$(docker ps -a -q)"
Expand Down
Loading