Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
219 changes: 219 additions & 0 deletions benchmarks/single_node/agentic/qwen3.8next_fp8_h100_mtp.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,219 @@
#!/usr/bin/env bash
set -euo pipefail
set -x

# Agentic trace replay benchmark for Qwen3.8-Flash-Next FP8 on H100 using
# SGLang with MTP speculative decoding. Day-zero recipe; SGLang is the
# plan-of-record engine for this model (MODELS.md), and it is spec-decode only,
# per the AgentX policy that new agentic arms ship with speculative decoding
# enabled rather than as an STP/MTP A/B.
#
# H100 is Hopper, so this arm is FP8 (Qwen/Qwen3.8-Flash-Next-FP8, 172.8 GiB)
# rather than the NVFP4 checkpoint the Blackwell arms use: NVFP4 needs SM100
# tensor cores. The SGLang cookbook does not offer H100 at all, so this recipe
# is the H200 arm adjusted for the smaller part rather than a verified command:
# * TP8/EP8 instead of the cookbook's TP4/EP4. At TP4 the 172.8 GiB
# checkpoint is ~43 GiB per rank of an 80 GB card, which leaves too little
# for the 256k-capped agentic traces. TP8 halves that to ~22 GiB.
# * --mem-fraction-static 0.75 rather than 0.85, matching the Qwen3.5 H100
# sibling: 80 GB HBM3 has far less slack than H200's 141 GB HBM3e.
#
# Structure follows the proven H100 MTP AgentX replay path (HiCache host-DRAM
# offload, the multi_tokenizer cached_tokens_details patch, aiperf-driven trace
# replay). Attention stays on the flashinfer linear-attention backends (sm_90);
# the trtllm_mha path is Blackwell-only.
#
# Speculative decoding is SGLANG_ENABLE_SPEC_V2=1 with NEXTN, 3 steps,
# eagle-topk 1 and 4 draft tokens, i.e. 3 speculative tokens per verification
# step, matching every other Qwen3.8-Flash-Next arm.
#
# Throughput runs pin acceptance to the committed golden AL through SGLang's
# simulated-acceptance path; the EVAL_ONLY accuracy run leaves it off and keeps
# real verification. See the SGLANG_SIMULATE_ACC_* block.
#
# Required env vars:
# MODEL, TP, CONC, KV_OFFLOADING, TOTAL_CPU_DRAM_GB, RESULT_DIR
#
# KV_OFFLOADING=dram requires KV_OFFLOAD_BACKEND=hicache.

source "$(dirname "$0")/../../benchmark_lib.sh"

check_env_vars MODEL TP CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION EP_SIZE

SCHEDULER_RECV_INTERVAL=${SCHEDULER_RECV_INTERVAL:-10}

if [[ -n "${SLURM_JOB_ID:-}" ]]; then
echo "JOB $SLURM_JOB_ID running on ${SLURMD_NODENAME:-unknown}"
fi

# `hf download` creates the target dir if missing and is itself idempotent.
# When MODEL_PATH is unset (stand-alone runs), fall back to the HF_HUB_CACHE
# Either way, MODEL_PATH is what the server is launched with.
if [[ -n "${MODEL_PATH:-}" ]]; then
if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then
hf download "$MODEL" --local-dir "$MODEL_PATH"
fi
else
hf download "$MODEL"
export MODEL_PATH="$MODEL"
fi
nvidia-smi

# ---- Resolve traces and install deps ----------------------------------------
# Keep the 256k-capped with-subagents corpus the H100 Qwen3.5 AgentX recipe
# uses (470 traces, max in+out <= 256k). The unfiltered corpus has requests up
# to ~1M proxy tokens that the server would reject, and H100's 80 GB is the
# tightest part in this set, so the capped corpus matters most here.
export WEKA_LOADER_OVERRIDE=semianalysis_cc_traces_weka_with_subagents_256k

resolve_trace_source
install_agentic_deps

# ---- Server config ----------------------------------------------------------
SERVER_LOG="$RESULT_DIR/server.log"
mkdir -p "$RESULT_DIR"

CACHE_ARGS=()
if require_agentic_kv_offload_backend hicache; then
# HiCache extends RadixAttention, so do not pass --disable-radix-cache.
# Hybrid GDN/Mamba allocates one KV and one Mamba host pool per rank.
REQUESTED_HICACHE_TOTAL_GB="${HICACHE_TOTAL_CPU_DRAM_GB:-$TOTAL_CPU_DRAM_GB}"
if [ "$REQUESTED_HICACHE_TOTAL_GB" -gt "$TOTAL_CPU_DRAM_GB" ]; then
echo "Error: requested HiCache pool ${REQUESTED_HICACHE_TOTAL_GB} GB exceeds configured capacity ${TOTAL_CPU_DRAM_GB} GB" >&2
exit 1
fi
TOTAL_CPU_DRAM_GB="$REQUESTED_HICACHE_TOTAL_GB"
HICACHE_HOST_POOL_COUNT="${HICACHE_HOST_POOL_COUNT:-2}"
HICACHE_WRITE_POLICY="${HICACHE_WRITE_POLICY:-write_through_selective}"
MAX_HICACHE_SIZE_GB=$((TOTAL_CPU_DRAM_GB / TP / HICACHE_HOST_POOL_COUNT))
HICACHE_SIZE_GB="${HICACHE_SIZE_GB:-$MAX_HICACHE_SIZE_GB}"
if [ "$HICACHE_SIZE_GB" -gt "$MAX_HICACHE_SIZE_GB" ]; then
echo "Error: HICACHE_SIZE_GB=$HICACHE_SIZE_GB exceeds configured per-pool limit $MAX_HICACHE_SIZE_GB" >&2
exit 1
fi
if [ "$HICACHE_SIZE_GB" -lt 1 ]; then
echo "Error: computed HICACHE_SIZE_GB=$HICACHE_SIZE_GB from TOTAL_CPU_DRAM_GB=$TOTAL_CPU_DRAM_GB, TP=$TP, HICACHE_HOST_POOL_COUNT=$HICACHE_HOST_POOL_COUNT" >&2
exit 1
fi
echo "HiCache CPU pool: ${HICACHE_SIZE_GB} GB per rank per host pool across TP=${TP}, host_pool_count=${HICACHE_HOST_POOL_COUNT}"
CACHE_ARGS=(
--page-size 64
--enable-hierarchical-cache
--hicache-size "$HICACHE_SIZE_GB"
--hicache-io-backend kernel
--hicache-mem-layout page_first
--hicache-write-policy "$HICACHE_WRITE_POLICY"
)
fi

echo "Starting SGLang server..."
export PYTHONNOUSERSITE=1
export SGLANG_ENABLE_SPEC_V2=1

# 3 speculative tokens per step (num-steps 3, eagle-topk 1, 4 draft tokens),
# the same MTP shape as the fixed-seq-len Qwen3.5 recipes.
SPEC_ARGS=(
--speculative-algorithm NEXTN
--speculative-num-steps 3
--speculative-eagle-topk 1
--speculative-num-draft-tokens 4
)

# AgentX pins acceptance to the committed golden AL so submissions are compared
# on system performance at a fixed acceptance target rather than on draft-head
# quality (golden_al_distribution/README.md). 3.39 is the Qwen3.5 MTP curve at
# num_speculative_tokens=3, thinking_on (golden_al_distribution/qwen3.5_mtp.yaml)
# -- the same value the GB300 Qwen3.5 AgentX srt-slurm recipes pin.
Comment on lines +124 to +126

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 Stale copy-pasted comment block from the Qwen3.5 H100 sibling still claims the pinned golden AL is "3.39 ... the Qwen3.5 MTP curve" from golden_al_distribution/qwen3.5_mtp.yaml, and that the image is pinned for v0.5.16 vs the sibling's v0.5.12, even though this script actually exports SGLANG_SIMULATE_ACC_LEN=2.32 from golden_al_distribution/qwen3.8next_mtp.yaml a few lines below and uses a differently-named image tag (qwen38flashnext) with no version pin claim.

Extended reasoning...

A reviewer or future engineer auditing/reusing this recipe reads the outer comment, believes the pinned acceptance length is 3.39 for a Qwen3.5-style curve, and either miscopies that stale value into a new arm or is confused when the actual exported value (2.32) disagrees with the documentation, since the code correctly uses 2.32 but the surrounding prose was never updated after copy-paste from the qwen3.5_fp8_h100_mtp.sh template.

Verification: nit. The candidate is factually correct and the contradiction is on the changed lines. The outer comment (diff lines 114-122) states "3.39 is the Qwen3.5 MTP curve at num_speculative_tokens=3, thinking_on (golden_al_distribution/qwen3.5_mtp.yaml)" and claims the image is pinned because "SGLANG_SIMULATE_ACC_TOKEN_MODE landed in SGLang v0.5.16 ... rather than the non-MTP agentic sibling's…

# SGLANG_SIMULATE_ACC_TOKEN_MODE landed in SGLang v0.5.16, which is why this
# recipe pins that image rather than the non-MTP agentic sibling's v0.5.12.
#
# EVAL_ONLY leaves simulated acceptance off: it commits drafted tokens
# regardless of the target logits, so generated text is wrong and the eval would
# score ~0.
if [ "${EVAL_ONLY:-false}" != "true" ]; then
# golden_al_distribution/qwen3.8next_mtp.yaml:
# qwen3.8-flash-next-fp8.thinking_on[3] = 2.32.
# --speculative-num-steps 3 with 4 draft tokens is 3 speculative tokens
# per verification step, i.e. the MTP=3 cell. AgentX replays run with
# thinking on, so the thinking_on row is the right one.
export SGLANG_SIMULATE_ACC_LEN=2.32
export SGLANG_SIMULATE_ACC_METHOD=match-expected
export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token
fi

SGLANG_MULTI_TOKENIZER=/sgl-workspace/sglang/python/sglang/srt/managers/multi_tokenizer_mixin.py
if ! sed -n '/elif isinstance(output, BatchStrOutput):/,/input_token_logprobs_val=_extract_field_by_index/p' "$SGLANG_MULTI_TOKENIZER" \
| grep -q 'cached_tokens_details=_extract_field_by_index'; then
sed -i '/elif isinstance(output, BatchStrOutput):/,/input_token_logprobs_val=_extract_field_by_index/ {
/cached_tokens=_extract_field_by_index(output, "cached_tokens", i),/a\
cached_tokens_details=_extract_field_by_index(\
output, "cached_tokens_details", i\
),
}' "$SGLANG_MULTI_TOKENIZER"
fi

{ set +x; } 2>/dev/null
# AgentX concurrency counts live session trees rather than individual HTTP
# requests. Leave room for subagent fan-out, and do not spend HBM capturing
# graphs above the batch sizes that stay useful for this long-context workload.
# The Qwen3.5 H200 template left both flags commented out, so neither variable
# existed; NEXTN silently caps --max-running-requests at 48 when it is unset.
MAX_RUNNING_REQUESTS=$((2 * CONC))
CUDA_GRAPH_MAX_BS="$CONC"
if [ "$CUDA_GRAPH_MAX_BS" -gt 64 ]; then
CUDA_GRAPH_MAX_BS=64
fi

SGLANG_CMD=(
python3 -m sglang.launch_server
--model-path "$MODEL_PATH"
--served-model-name "$MODEL"
--host 0.0.0.0
--port "$PORT"
--trust-remote-code
# Verified flags from the SGLang cookbook playground for this model on
# H200 / FP8 / low latency / single node, adjusted for H100's 80 GB.
# NVFP4 is greyed out for Hopper, so FP8 is the whole surface here.
--tp-size "$TP"
--ep-size "$EP_SIZE"
--dp-size 1
--mem-fraction-static 0.75
--chunked-prefill-size 8192
--linear-attn-prefill-backend flashinfer
--linear-attn-decode-backend flashinfer
# float32, not the cookbook's bfloat16. With NEXTN enabled the GDN linear
# attention backend routes verification through flashinfer's
# gated_delta_rule_mtp, which asserts initial_state.dtype == torch.float32
# and aborts CUDA graph capture on a bf16 SSM state:
# AssertionError: initial_state must be float32, got torch.bfloat16
# flashinfer/gdn_decode.py:761, via gdn_backend.py target_verify
# The cookbook command pairs bfloat16 with NEXTN, but this flashinfer build
# rejects that combination, and the state dtype is the half that can move.
--mamba-ssm-dtype float32
"${SPEC_ARGS[@]}"
--reasoning-parser auto
# NEXTN silently resets --max-running-requests to 48 when it is unset, so
# this must stay explicit and sized to the AgentX concurrency.
--max-running-requests "$MAX_RUNNING_REQUESTS"
--cuda-graph-max-bs "$CUDA_GRAPH_MAX_BS"
--stream-interval 50
--scheduler-recv-interval "$SCHEDULER_RECV_INTERVAL"
--tokenizer-worker-num 6
--tokenizer-path "$MODEL"
--enable-metrics
"${CACHE_ARGS[@]}"
)
printf '%q ' "${SGLANG_CMD[@]}" | tee "$RESULT_DIR/sglang_command.txt"
printf '\n' | tee -a "$RESULT_DIR/sglang_command.txt"
"${SGLANG_CMD[@]}" > "$SERVER_LOG" 2>&1 &
SERVER_PID=$!
echo "Server PID: $SERVER_PID"

wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID"

if [ "${EVAL_ONLY}" = "true" ]; then
run_eval --port "$PORT"
else
build_replay_cmd "$RESULT_DIR"
run_agentic_replay_and_write_outputs "$RESULT_DIR"
fi
19 changes: 19 additions & 0 deletions configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -7253,6 +7253,25 @@ qwen3.5-fp8-h100-sglang-agentic-mtp:
- { tp: 8, ep: 8, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [4, 8, 12, 16] }



# Qwen3.8-Flash-Next FP8 AgentX on H100 via SGLang with native NEXTN MTP.
# Day-zero recipe. H100 is Hopper, so FP8: NVFP4 needs SM100 tensor cores. The
# SGLang cookbook does not list H100, so this mirrors the H200 arm adjusted for
# the smaller part: TP8/EP8 rather than the cookbook's TP4/EP4, since 172.8 GiB
# at TP4 leaves too little of an 80 GB card for the 256k-capped traces.
qwen3.8next-fp8-h100-sglang-agentic-mtp:
image: lmsysorg/sglang:qwen38flashnext
model: Qwen/Qwen3.8-Flash-Next-FP8
model-prefix: qwen3.8next
runner: cluster:h100-dgxc
precision: fp8
framework: sglang
multinode: false
scenarios:
agentic-coding:
- dram-utilization: 0.8
search-space:
- { tp: 8, ep: 8, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 4, 8, 12, 16] }
qwen3.5-fp4-b200-trt:
image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc18
model: nvidia/Qwen3.5-397B-A17B-NVFP4
Expand Down
11 changes: 11 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -6526,3 +6526,14 @@
- "Pin throughput runs to the committed golden thinking_on acceptance length of 2.32 at three speculative tokens; eval-only runs keep real target verification."
- "Keep the bfloat16 Mamba SSM state the cookbook specifies: SGLang requires it on SM100 or newer whenever the flashinfer linear-attention decode backend is selected."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2758

- config-keys:
- qwen3.8next-fp8-h100-sglang-agentic-mtp
scenario-type:
- agentic-coding
description:
- "Add the day-zero Qwen3.8-Flash-Next FP8 AgentX recipe on H100 with SGLang native NEXTN MTP at TP8 with EP8 and concurrency 1/4/8/12/16."
- "Mirror the H200 arm, adjusted for the smaller part: TP8 with EP8 rather than TP4 with EP4, and memory fraction 0.75 rather than 0.85."
- "Use a float32 Mamba SSM state, as Hopper's flashinfer verify kernel requires, unlike the bfloat16 the Blackwell arms must use."
- "Pin throughput runs to the committed golden thinking_on acceptance length of 2.32 at three speculative tokens; eval-only runs keep real target verification."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2756