Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
209 changes: 91 additions & 118 deletions benchmarks/single_node/agentic/qwen3.5_fp8_h200_mtp.sh
Original file line number Diff line number Diff line change
Expand Up @@ -2,44 +2,34 @@
set -euo pipefail
set -x

# Agentic trace replay benchmark for Qwen3.5 FP8 on H200 using SGLang with MTP
# speculative decoding. First Qwen3.5 AgentX recipe on H200; it is spec-decode
# only, per the AgentX policy that new agentic arms ship with speculative
# decoding enabled rather than as an STP/MTP A/B (MODELS.md).
#
# Structure follows the proven agentic/qwen3.5_fp8_h100.sh replay path
# (HiCache host-DRAM offload, the multi_tokenizer cached_tokens_details patch,
# aiperf-driven trace replay). H200's 141 GB HBM3e is roomier than H100's 80 GB,
# so --mem-fraction-static is 0.8 rather than 0.75, matching
# fixed_seq_len/qwen3.5_fp8_h200_mtp.sh. Attention backend stays flashinfer
# (sm_90); the trtllm_mha path is Blackwell-only.
#
# Speculative decoding mirrors fixed_seq_len/qwen3.5_fp8_h100_mtp.sh:
# SGLANG_ENABLE_SPEC_V2=1 with --speculative-algorithm EAGLE, 3 steps, eagle-topk
# 1 and 4 draft tokens, i.e. 3 speculative tokens per verification step.
#
# Throughput runs pin acceptance to the committed golden AL through SGLang's
# simulated-acceptance path; the EVAL_ONLY accuracy run leaves it off and keeps
# real verification. See the SGLANG_SIMULATE_ACC_* block.
#
# Required env vars:
# MODEL, TP, CONC, KV_OFFLOADING, TOTAL_CPU_DRAM_GB, RESULT_DIR
#
# KV_OFFLOADING=dram requires KV_OFFLOAD_BACKEND=hicache.

source "$(dirname "$0")/../../benchmark_lib.sh"

check_env_vars MODEL TP CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION EP_SIZE

SCHEDULER_RECV_INTERVAL=${SCHEDULER_RECV_INTERVAL:-10}
# Agentic trace replay benchmark for Qwen3.5 FP8 on H200 with SGLang MTP.

SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../../benchmark_lib.sh"

check_env_vars \
MODEL MODEL_PREFIX TP CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR \
DURATION EP_SIZE EVAL_ONLY SPEC_DECODING

if ! require_agentic_kv_offload_backend hicache; then
echo "Error: the H200 MTP recipe requires KV_OFFLOADING=dram with HiCache" >&2
exit 1
fi

if [ "$TP" -ne 8 ] || [ "$EP_SIZE" -ne 1 ] || [ "$SPEC_DECODING" != "mtp" ]; then
echo "Error: the H200 MTP recipe requires TP=8, EP_SIZE=1, and SPEC_DECODING=mtp" >&2
exit 1
fi

if [ "$TOTAL_CPU_DRAM_GB" -lt 1200 ]; then
echo "Error: TP8 requires at least 1200 GB host DRAM; generated budget is ${TOTAL_CPU_DRAM_GB} GB" >&2
exit 1
fi

if [[ -n "${SLURM_JOB_ID:-}" ]]; then
echo "JOB $SLURM_JOB_ID running on ${SLURMD_NODENAME:-unknown}"
fi

# `hf download` creates the target dir if missing and is itself idempotent.
# When MODEL_PATH is unset (stand-alone runs), fall back to the HF_HUB_CACHE
# Either way, MODEL_PATH is what the server is launched with.
if [[ -n "${MODEL_PATH:-}" ]]; then
if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then
hf download "$MODEL" --local-dir "$MODEL_PATH"
Expand All @@ -50,134 +40,117 @@ else
fi
nvidia-smi

# ---- Resolve traces and install deps ----------------------------------------
# Keep the 256k-capped with-subagents corpus the H100 Qwen3.5 AgentX recipe
# uses (470 traces, max in+out <= 256k). The unfiltered corpus has requests up
# to ~1M proxy tokens that the server would reject; H200's extra HBM raises the
# context ceiling but not past 256k for this model at TP8.
export WEKA_LOADER_OVERRIDE=semianalysis_cc_traces_weka_with_subagents_256k

resolve_trace_source
install_agentic_deps

# ---- Server config ----------------------------------------------------------
agentic_pip_install --no-deps --force-reinstall flashinfer_python==0.6.17
agentic_pip_install \
--no-deps --force-reinstall flashinfer-cubin==0.6.17 \
--index-url https://flashinfer.ai/whl
agentic_pip_install \
--no-deps --force-reinstall flashinfer-jit-cache==0.6.17+cu130 \
--index-url https://flashinfer.ai/whl/cu130

SERVER_LOG="$RESULT_DIR/server.log"
mkdir -p "$RESULT_DIR"

CACHE_ARGS=()
if require_agentic_kv_offload_backend hicache; then
# HiCache extends RadixAttention, so do not pass --disable-radix-cache.
# Hybrid GDN/Mamba allocates one KV and one Mamba host pool per rank.
REQUESTED_HICACHE_TOTAL_GB="${HICACHE_TOTAL_CPU_DRAM_GB:-$TOTAL_CPU_DRAM_GB}"
if [ "$REQUESTED_HICACHE_TOTAL_GB" -gt "$TOTAL_CPU_DRAM_GB" ]; then
echo "Error: requested HiCache pool ${REQUESTED_HICACHE_TOTAL_GB} GB exceeds configured capacity ${TOTAL_CPU_DRAM_GB} GB" >&2
exit 1
fi
TOTAL_CPU_DRAM_GB="$REQUESTED_HICACHE_TOTAL_GB"
HICACHE_HOST_POOL_COUNT="${HICACHE_HOST_POOL_COUNT:-2}"
HICACHE_WRITE_POLICY="${HICACHE_WRITE_POLICY:-write_through_selective}"
MAX_HICACHE_SIZE_GB=$((TOTAL_CPU_DRAM_GB / TP / HICACHE_HOST_POOL_COUNT))
HICACHE_SIZE_GB="${HICACHE_SIZE_GB:-$MAX_HICACHE_SIZE_GB}"
if [ "$HICACHE_SIZE_GB" -gt "$MAX_HICACHE_SIZE_GB" ]; then
echo "Error: HICACHE_SIZE_GB=$HICACHE_SIZE_GB exceeds configured per-pool limit $MAX_HICACHE_SIZE_GB" >&2
exit 1
fi
if [ "$HICACHE_SIZE_GB" -lt 1 ]; then
echo "Error: computed HICACHE_SIZE_GB=$HICACHE_SIZE_GB from TOTAL_CPU_DRAM_GB=$TOTAL_CPU_DRAM_GB, TP=$TP, HICACHE_HOST_POOL_COUNT=$HICACHE_HOST_POOL_COUNT" >&2
exit 1
fi
echo "HiCache CPU pool: ${HICACHE_SIZE_GB} GB per rank per host pool across TP=${TP}, host_pool_count=${HICACHE_HOST_POOL_COUNT}"
CACHE_ARGS=(
--page-size 64
--enable-hierarchical-cache
--hicache-size "$HICACHE_SIZE_GB"
--hicache-io-backend kernel
--hicache-mem-layout page_first
--hicache-write-policy "$HICACHE_WRITE_POLICY"
)
fi

echo "Starting SGLang server..."
export TORCH_CUDA_ARCH_LIST=9.0
export PYTHONNOUSERSITE=1
export PYTHONUNBUFFERED=1
export FLASHINFER_DISABLE_VERSION_CHECK=1
export FLASHINFER_WORKSPACE_BASE=/tmp/flashinfer-cache
export SGL_ENABLE_JIT_DEEPGEMM=false
export SGLANG_ENABLE_FLASHINFER_GEMM=true
export SGLANG_ENABLE_SPEC_V2=1

# 3 speculative tokens per step (num-steps 3, eagle-topk 1, 4 draft tokens),
# the same MTP shape as the fixed-seq-len Qwen3.5 recipes.
SPEC_ARGS=(
--speculative-algorithm EAGLE
--speculative-algorithm NEXTN
--speculative-num-steps 3
--speculative-eagle-topk 1
--speculative-num-draft-tokens 4
)

# AgentX pins acceptance to the committed golden AL so submissions are compared
# on system performance at a fixed acceptance target rather than on draft-head
# quality (golden_al_distribution/README.md). 3.39 is the Qwen3.5 MTP curve at
# num_speculative_tokens=3, thinking_on (golden_al_distribution/qwen3.5_mtp.yaml)
# -- the same value the GB300 Qwen3.5 AgentX srt-slurm recipes pin.
# SGLANG_SIMULATE_ACC_TOKEN_MODE landed in SGLang v0.5.16, which is why this
# recipe pins that image rather than the non-MTP agentic sibling's v0.5.12.
#
# EVAL_ONLY leaves simulated acceptance off: it commits drafted tokens
# regardless of the target logits, so generated text is wrong and the eval would
# score ~0.
if [ "${EVAL_ONLY:-false}" != "true" ]; then
if [ "$EVAL_ONLY" != "true" ]; then
export SGLANG_SIMULATE_ACC_LEN=3.39
export SGLANG_SIMULATE_ACC_METHOD=match-expected
export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token
fi

SGLANG_MULTI_TOKENIZER=/sgl-workspace/sglang/python/sglang/srt/managers/multi_tokenizer_mixin.py
if ! sed -n '/elif isinstance(output, BatchStrOutput):/,/input_token_logprobs_val=_extract_field_by_index/p' "$SGLANG_MULTI_TOKENIZER" \
| grep -q 'cached_tokens_details=_extract_field_by_index'; then
sed -i '/elif isinstance(output, BatchStrOutput):/,/input_token_logprobs_val=_extract_field_by_index/ {
/cached_tokens=_extract_field_by_index(output, "cached_tokens", i),/a\
cached_tokens_details=_extract_field_by_index(\
output, "cached_tokens_details", i\
),
}' "$SGLANG_MULTI_TOKENIZER"
fi
CACHE_ARGS=(
--page-size 64
--enable-hierarchical-cache
--hicache-ratio 0.9
--hicache-io-backend kernel
--hicache-mem-layout page_first_direct
--hicache-write-policy write_back
)

{ set +x; } 2>/dev/null
SGLANG_CMD=(
python3 -m sglang.launch_server
--model-path="$MODEL_PATH" --served-model-name="$MODEL"
--host=0.0.0.0
--port="$PORT"
--served-model-name "Qwen/Qwen3.5-397B-A17B-FP8"
--model-path "$MODEL_PATH"
--served-model-name "$MODEL"
--host 0.0.0.0
--port "$PORT"
--trust-remote-code
--tensor-parallel-size="$TP"
--data-parallel-size=1
--expert-parallel-size="$EP_SIZE"
--tensor-parallel-size "$TP"
--data-parallel-size 1
--expert-parallel-size "$EP_SIZE"
--quantization fp8
--kv-cache-dtype fp8_e4m3
--mamba-ssm-dtype bfloat16
--mamba-scheduler-strategy extra_buffer
--mamba-track-interval 8192
--attention-backend flashinfer
--enable-flashinfer-allreduce-fusion
# --cuda-graph-max-bs "$CONC"
# --max-running-requests "$CONC"
# --max-prefill-tokens 8192
# --chunked-prefill-size 8192
--mem-fraction-static 0.8
--cuda-graph-max-bs 32
--max-running-requests 128
--max-prefill-tokens 16384
--chunked-prefill-size 16384
--mem-fraction-static 0.78
--max-mamba-cache-size 360
--allow-auto-truncate
--stream-interval 50
--scheduler-recv-interval "$SCHEDULER_RECV_INTERVAL"
--scheduler-recv-interval 10
--tokenizer-worker-num 6
--tokenizer-path "$MODEL"
--enable-cache-report
--enable-symm-mem
--enable-metrics
"${SPEC_ARGS[@]}"
"${CACHE_ARGS[@]}"
"${SPEC_ARGS[@]}"
)

printf '%q ' "${SGLANG_CMD[@]}" | tee "$RESULT_DIR/sglang_command.txt"
printf '\n' | tee -a "$RESULT_DIR/sglang_command.txt"
"${SGLANG_CMD[@]}" > "$SERVER_LOG" 2>&1 &

SERVER_PID=""
cleanup_agentic_services() {
local exit_code=$?
trap - EXIT INT TERM
set +e
stop_background_process_tree "$SERVER_PID" "SGLang server" 60
exit "$exit_code"
}
trap cleanup_agentic_services EXIT
trap 'exit 130' INT
trap 'exit 143' TERM

{
echo "=== SGLANG_* env vars at launch ==="
env | grep -E '^SGLANG_' | sort
echo "==================================="
} > "$SERVER_LOG"

"${SGLANG_CMD[@]}" >> "$SERVER_LOG" 2>&1 &
SERVER_PID=$!
echo "Server PID: $SERVER_PID"

wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID"

if [ "${EVAL_ONLY}" = "true" ]; then
if [ "$EVAL_ONLY" = "true" ]; then
run_eval --port "$PORT"
else
build_replay_cmd "$RESULT_DIR"
REPLAY_CMD+=" --apply-chat-template"
REPLAY_CMD+=" --server-metrics http://localhost:$PORT/metrics"
run_agentic_replay_and_write_outputs "$RESULT_DIR"
fi
16 changes: 16 additions & 0 deletions configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -7186,6 +7186,22 @@ qwen3.5-fp8-h200-sglang-agentic-mtp:
- { tp: 8, ep: 8, spec-decoding: mtp, kv-offloading: none, conc-list: [4, 8, 12, 16] }
- { tp: 8, ep: 8, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [4, 8, 12, 16] }

# H200 AgentX MTP frontier with DRAM HiCache. This is intentionally an MTP-only
# submission; the model's non-speculative AgentX arm is not included.
qwen3.5-fp8-h200-sglang-agentic-hicache-mtp:
image: lmsysorg/sglang:nightly-dev-cu13-20260813-273d978b
model: Qwen/Qwen3.5-397B-A17B-FP8
model-prefix: qwen3.5
runner: cluster:h200-dgxc
precision: fp8
framework: sglang
multinode: false
scenarios:
agentic-coding:
- dram-utilization: 0.80
search-space:
- { tp: 8, ep: 1, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [2, 4, 8, 10, 12, 16, 20, 24] }


qwen3.5-fp8-h100-sglang-agentic-mtp:
image: lmsysorg/sglang:v0.5.16-cu130
Expand Down
9 changes: 9 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -5953,3 +5953,12 @@
- "Use native EAGLE MTP (3 steps, top-k 1, 4 draft tokens) and golden synthetic acceptance length 2.49 for throughput; eval retains real verification."
- "Follow the official SGLang DeepSeek-V4 Blackwell recipe, require nonempty SGLang server metrics, keep pooled AgentX connections alive, let AIPerf own HiCache warmup, and reserve transient MoE workspace at DEP8 c512."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2577

- config-keys:
- qwen3.5-fp8-h200-sglang-agentic-hicache-mtp
scenario-type:
- agentic-coding
description:
- "Add Qwen3.5 FP8 H200 SGLang AgentX MTP points with DRAM HiCache"
- "Use TP8/EP1 concurrency 2 through 24, published FlashInfer 0.6.17 CUDA 13 wheels, and golden MTP acceptance length 3.39"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX
2 changes: 1 addition & 1 deletion runners/launch_h200-dgxc-slurm.sh
Original file line number Diff line number Diff line change
Expand Up @@ -350,7 +350,7 @@ EOF
find . -name '.nfs*' -delete 2>/dev/null || true

else
SQUASH_FILE="/data/gharunners/containers/$(echo "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh"
SQUASH_FILE="/data/containers/$(echo "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh"

# Convert pyxis image format (nvcr.io#path) to docker format (nvcr.io/path) for enroot import
DOCKER_IMAGE=$(echo "$IMAGE" | sed 's/#/\//g')
Expand Down
Loading