diff --git a/benchmarks/single_node/agentic/qwen3.5_fp8_h200_mtp.sh b/benchmarks/single_node/agentic/qwen3.5_fp8_h200_mtp.sh index 62982a397..b6f6d7eeb 100755 --- a/benchmarks/single_node/agentic/qwen3.5_fp8_h200_mtp.sh +++ b/benchmarks/single_node/agentic/qwen3.5_fp8_h200_mtp.sh @@ -2,44 +2,34 @@ set -euo pipefail set -x -# Agentic trace replay benchmark for Qwen3.5 FP8 on H200 using SGLang with MTP -# speculative decoding. First Qwen3.5 AgentX recipe on H200; it is spec-decode -# only, per the AgentX policy that new agentic arms ship with speculative -# decoding enabled rather than as an STP/MTP A/B (MODELS.md). -# -# Structure follows the proven agentic/qwen3.5_fp8_h100.sh replay path -# (HiCache host-DRAM offload, the multi_tokenizer cached_tokens_details patch, -# aiperf-driven trace replay). H200's 141 GB HBM3e is roomier than H100's 80 GB, -# so --mem-fraction-static is 0.8 rather than 0.75, matching -# fixed_seq_len/qwen3.5_fp8_h200_mtp.sh. Attention backend stays flashinfer -# (sm_90); the trtllm_mha path is Blackwell-only. -# -# Speculative decoding mirrors fixed_seq_len/qwen3.5_fp8_h100_mtp.sh: -# SGLANG_ENABLE_SPEC_V2=1 with --speculative-algorithm EAGLE, 3 steps, eagle-topk -# 1 and 4 draft tokens, i.e. 3 speculative tokens per verification step. -# -# Throughput runs pin acceptance to the committed golden AL through SGLang's -# simulated-acceptance path; the EVAL_ONLY accuracy run leaves it off and keeps -# real verification. See the SGLANG_SIMULATE_ACC_* block. -# -# Required env vars: -# MODEL, TP, CONC, KV_OFFLOADING, TOTAL_CPU_DRAM_GB, RESULT_DIR -# -# KV_OFFLOADING=dram requires KV_OFFLOAD_BACKEND=hicache. - -source "$(dirname "$0")/../../benchmark_lib.sh" - -check_env_vars MODEL TP CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION EP_SIZE - -SCHEDULER_RECV_INTERVAL=${SCHEDULER_RECV_INTERVAL:-10} +# Agentic trace replay benchmark for Qwen3.5 FP8 on H200 with SGLang MTP. + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +source "$SCRIPT_DIR/../../benchmark_lib.sh" + +check_env_vars \ + MODEL MODEL_PREFIX TP CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR \ + DURATION EP_SIZE EVAL_ONLY SPEC_DECODING + +if ! require_agentic_kv_offload_backend hicache; then + echo "Error: the H200 MTP recipe requires KV_OFFLOADING=dram with HiCache" >&2 + exit 1 +fi + +if [ "$TP" -ne 8 ] || [ "$EP_SIZE" -ne 1 ] || [ "$SPEC_DECODING" != "mtp" ]; then + echo "Error: the H200 MTP recipe requires TP=8, EP_SIZE=1, and SPEC_DECODING=mtp" >&2 + exit 1 +fi + +if [ "$TOTAL_CPU_DRAM_GB" -lt 1200 ]; then + echo "Error: TP8 requires at least 1200 GB host DRAM; generated budget is ${TOTAL_CPU_DRAM_GB} GB" >&2 + exit 1 +fi if [[ -n "${SLURM_JOB_ID:-}" ]]; then echo "JOB $SLURM_JOB_ID running on ${SLURMD_NODENAME:-unknown}" fi -# `hf download` creates the target dir if missing and is itself idempotent. -# When MODEL_PATH is unset (stand-alone runs), fall back to the HF_HUB_CACHE -# Either way, MODEL_PATH is what the server is launched with. if [[ -n "${MODEL_PATH:-}" ]]; then if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then hf download "$MODEL" --local-dir "$MODEL_PATH" @@ -50,134 +40,117 @@ else fi nvidia-smi -# ---- Resolve traces and install deps ---------------------------------------- -# Keep the 256k-capped with-subagents corpus the H100 Qwen3.5 AgentX recipe -# uses (470 traces, max in+out <= 256k). The unfiltered corpus has requests up -# to ~1M proxy tokens that the server would reject; H200's extra HBM raises the -# context ceiling but not past 256k for this model at TP8. export WEKA_LOADER_OVERRIDE=semianalysis_cc_traces_weka_with_subagents_256k - resolve_trace_source install_agentic_deps -# ---- Server config ---------------------------------------------------------- +agentic_pip_install --no-deps --force-reinstall flashinfer_python==0.6.17 +agentic_pip_install \ + --no-deps --force-reinstall flashinfer-cubin==0.6.17 \ + --index-url https://flashinfer.ai/whl +agentic_pip_install \ + --no-deps --force-reinstall flashinfer-jit-cache==0.6.17+cu130 \ + --index-url https://flashinfer.ai/whl/cu130 + SERVER_LOG="$RESULT_DIR/server.log" mkdir -p "$RESULT_DIR" -CACHE_ARGS=() -if require_agentic_kv_offload_backend hicache; then - # HiCache extends RadixAttention, so do not pass --disable-radix-cache. - # Hybrid GDN/Mamba allocates one KV and one Mamba host pool per rank. - REQUESTED_HICACHE_TOTAL_GB="${HICACHE_TOTAL_CPU_DRAM_GB:-$TOTAL_CPU_DRAM_GB}" - if [ "$REQUESTED_HICACHE_TOTAL_GB" -gt "$TOTAL_CPU_DRAM_GB" ]; then - echo "Error: requested HiCache pool ${REQUESTED_HICACHE_TOTAL_GB} GB exceeds configured capacity ${TOTAL_CPU_DRAM_GB} GB" >&2 - exit 1 - fi - TOTAL_CPU_DRAM_GB="$REQUESTED_HICACHE_TOTAL_GB" - HICACHE_HOST_POOL_COUNT="${HICACHE_HOST_POOL_COUNT:-2}" - HICACHE_WRITE_POLICY="${HICACHE_WRITE_POLICY:-write_through_selective}" - MAX_HICACHE_SIZE_GB=$((TOTAL_CPU_DRAM_GB / TP / HICACHE_HOST_POOL_COUNT)) - HICACHE_SIZE_GB="${HICACHE_SIZE_GB:-$MAX_HICACHE_SIZE_GB}" - if [ "$HICACHE_SIZE_GB" -gt "$MAX_HICACHE_SIZE_GB" ]; then - echo "Error: HICACHE_SIZE_GB=$HICACHE_SIZE_GB exceeds configured per-pool limit $MAX_HICACHE_SIZE_GB" >&2 - exit 1 - fi - if [ "$HICACHE_SIZE_GB" -lt 1 ]; then - echo "Error: computed HICACHE_SIZE_GB=$HICACHE_SIZE_GB from TOTAL_CPU_DRAM_GB=$TOTAL_CPU_DRAM_GB, TP=$TP, HICACHE_HOST_POOL_COUNT=$HICACHE_HOST_POOL_COUNT" >&2 - exit 1 - fi - echo "HiCache CPU pool: ${HICACHE_SIZE_GB} GB per rank per host pool across TP=${TP}, host_pool_count=${HICACHE_HOST_POOL_COUNT}" - CACHE_ARGS=( - --page-size 64 - --enable-hierarchical-cache - --hicache-size "$HICACHE_SIZE_GB" - --hicache-io-backend kernel - --hicache-mem-layout page_first - --hicache-write-policy "$HICACHE_WRITE_POLICY" - ) -fi - -echo "Starting SGLang server..." +export TORCH_CUDA_ARCH_LIST=9.0 export PYTHONNOUSERSITE=1 +export PYTHONUNBUFFERED=1 +export FLASHINFER_DISABLE_VERSION_CHECK=1 +export FLASHINFER_WORKSPACE_BASE=/tmp/flashinfer-cache +export SGL_ENABLE_JIT_DEEPGEMM=false +export SGLANG_ENABLE_FLASHINFER_GEMM=true export SGLANG_ENABLE_SPEC_V2=1 -# 3 speculative tokens per step (num-steps 3, eagle-topk 1, 4 draft tokens), -# the same MTP shape as the fixed-seq-len Qwen3.5 recipes. SPEC_ARGS=( - --speculative-algorithm EAGLE + --speculative-algorithm NEXTN --speculative-num-steps 3 --speculative-eagle-topk 1 --speculative-num-draft-tokens 4 ) -# AgentX pins acceptance to the committed golden AL so submissions are compared -# on system performance at a fixed acceptance target rather than on draft-head -# quality (golden_al_distribution/README.md). 3.39 is the Qwen3.5 MTP curve at -# num_speculative_tokens=3, thinking_on (golden_al_distribution/qwen3.5_mtp.yaml) -# -- the same value the GB300 Qwen3.5 AgentX srt-slurm recipes pin. -# SGLANG_SIMULATE_ACC_TOKEN_MODE landed in SGLang v0.5.16, which is why this -# recipe pins that image rather than the non-MTP agentic sibling's v0.5.12. -# -# EVAL_ONLY leaves simulated acceptance off: it commits drafted tokens -# regardless of the target logits, so generated text is wrong and the eval would -# score ~0. -if [ "${EVAL_ONLY:-false}" != "true" ]; then +if [ "$EVAL_ONLY" != "true" ]; then export SGLANG_SIMULATE_ACC_LEN=3.39 export SGLANG_SIMULATE_ACC_METHOD=match-expected export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token fi -SGLANG_MULTI_TOKENIZER=/sgl-workspace/sglang/python/sglang/srt/managers/multi_tokenizer_mixin.py -if ! sed -n '/elif isinstance(output, BatchStrOutput):/,/input_token_logprobs_val=_extract_field_by_index/p' "$SGLANG_MULTI_TOKENIZER" \ - | grep -q 'cached_tokens_details=_extract_field_by_index'; then - sed -i '/elif isinstance(output, BatchStrOutput):/,/input_token_logprobs_val=_extract_field_by_index/ { - /cached_tokens=_extract_field_by_index(output, "cached_tokens", i),/a\ - cached_tokens_details=_extract_field_by_index(\ - output, "cached_tokens_details", i\ - ), - }' "$SGLANG_MULTI_TOKENIZER" -fi +CACHE_ARGS=( + --page-size 64 + --enable-hierarchical-cache + --hicache-ratio 0.9 + --hicache-io-backend kernel + --hicache-mem-layout page_first_direct + --hicache-write-policy write_back +) -{ set +x; } 2>/dev/null SGLANG_CMD=( python3 -m sglang.launch_server - --model-path="$MODEL_PATH" --served-model-name="$MODEL" - --host=0.0.0.0 - --port="$PORT" - --served-model-name "Qwen/Qwen3.5-397B-A17B-FP8" + --model-path "$MODEL_PATH" + --served-model-name "$MODEL" + --host 0.0.0.0 + --port "$PORT" --trust-remote-code - --tensor-parallel-size="$TP" - --data-parallel-size=1 - --expert-parallel-size="$EP_SIZE" + --tensor-parallel-size "$TP" + --data-parallel-size 1 + --expert-parallel-size "$EP_SIZE" --quantization fp8 --kv-cache-dtype fp8_e4m3 --mamba-ssm-dtype bfloat16 + --mamba-scheduler-strategy extra_buffer + --mamba-track-interval 8192 --attention-backend flashinfer - --enable-flashinfer-allreduce-fusion - # --cuda-graph-max-bs "$CONC" - # --max-running-requests "$CONC" - # --max-prefill-tokens 8192 - # --chunked-prefill-size 8192 - --mem-fraction-static 0.8 + --cuda-graph-max-bs 32 + --max-running-requests 128 + --max-prefill-tokens 16384 + --chunked-prefill-size 16384 + --mem-fraction-static 0.78 + --max-mamba-cache-size 360 + --allow-auto-truncate --stream-interval 50 - --scheduler-recv-interval "$SCHEDULER_RECV_INTERVAL" + --scheduler-recv-interval 10 --tokenizer-worker-num 6 - --tokenizer-path "$MODEL" + --enable-cache-report + --enable-symm-mem --enable-metrics - "${SPEC_ARGS[@]}" "${CACHE_ARGS[@]}" + "${SPEC_ARGS[@]}" ) + printf '%q ' "${SGLANG_CMD[@]}" | tee "$RESULT_DIR/sglang_command.txt" printf '\n' | tee -a "$RESULT_DIR/sglang_command.txt" -"${SGLANG_CMD[@]}" > "$SERVER_LOG" 2>&1 & + +SERVER_PID="" +cleanup_agentic_services() { + local exit_code=$? + trap - EXIT INT TERM + set +e + stop_background_process_tree "$SERVER_PID" "SGLang server" 60 + exit "$exit_code" +} +trap cleanup_agentic_services EXIT +trap 'exit 130' INT +trap 'exit 143' TERM + +{ + echo "=== SGLANG_* env vars at launch ===" + env | grep -E '^SGLANG_' | sort + echo "===================================" +} > "$SERVER_LOG" + +"${SGLANG_CMD[@]}" >> "$SERVER_LOG" 2>&1 & SERVER_PID=$! echo "Server PID: $SERVER_PID" wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" -if [ "${EVAL_ONLY}" = "true" ]; then +if [ "$EVAL_ONLY" = "true" ]; then run_eval --port "$PORT" else build_replay_cmd "$RESULT_DIR" + REPLAY_CMD+=" --apply-chat-template" + REPLAY_CMD+=" --server-metrics http://localhost:$PORT/metrics" run_agentic_replay_and_write_outputs "$RESULT_DIR" fi diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index caa0a7e08..f0f770e07 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -7186,6 +7186,22 @@ qwen3.5-fp8-h200-sglang-agentic-mtp: - { tp: 8, ep: 8, spec-decoding: mtp, kv-offloading: none, conc-list: [4, 8, 12, 16] } - { tp: 8, ep: 8, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [4, 8, 12, 16] } +# H200 AgentX MTP frontier with DRAM HiCache. This is intentionally an MTP-only +# submission; the model's non-speculative AgentX arm is not included. +qwen3.5-fp8-h200-sglang-agentic-hicache-mtp: + image: lmsysorg/sglang:nightly-dev-cu13-20260813-273d978b + model: Qwen/Qwen3.5-397B-A17B-FP8 + model-prefix: qwen3.5 + runner: cluster:h200-dgxc + precision: fp8 + framework: sglang + multinode: false + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - { tp: 8, ep: 1, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [2, 4, 8, 10, 12, 16, 20, 24] } + qwen3.5-fp8-h100-sglang-agentic-mtp: image: lmsysorg/sglang:v0.5.16-cu130 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 04ba9feb5..1bd6cf219 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5953,3 +5953,12 @@ - "Use native EAGLE MTP (3 steps, top-k 1, 4 draft tokens) and golden synthetic acceptance length 2.49 for throughput; eval retains real verification." - "Follow the official SGLang DeepSeek-V4 Blackwell recipe, require nonempty SGLang server metrics, keep pooled AgentX connections alive, let AIPerf own HiCache warmup, and reserve transient MoE workspace at DEP8 c512." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2577 + +- config-keys: + - qwen3.5-fp8-h200-sglang-agentic-hicache-mtp + scenario-type: + - agentic-coding + description: + - "Add Qwen3.5 FP8 H200 SGLang AgentX MTP points with DRAM HiCache" + - "Use TP8/EP1 concurrency 2 through 24, published FlashInfer 0.6.17 CUDA 13 wheels, and golden MTP acceptance length 3.39" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX diff --git a/runners/launch_h200-dgxc-slurm.sh b/runners/launch_h200-dgxc-slurm.sh index f904d8bc2..a0ff8b14b 100755 --- a/runners/launch_h200-dgxc-slurm.sh +++ b/runners/launch_h200-dgxc-slurm.sh @@ -350,7 +350,7 @@ EOF find . -name '.nfs*' -delete 2>/dev/null || true else - SQUASH_FILE="/data/gharunners/containers/$(echo "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" + SQUASH_FILE="/data/containers/$(echo "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh" # Convert pyxis image format (nvcr.io#path) to docker format (nvcr.io/path) for enroot import DOCKER_IMAGE=$(echo "$IMAGE" | sed 's/#/\//g')