Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
49 changes: 36 additions & 13 deletions benchmarks/single_node/agentic/dsv4_fp4_b200_sglang_mtp.sh
Original file line number Diff line number Diff line change
Expand Up @@ -67,11 +67,15 @@ if require_agentic_kv_offload_backend hicache; then
# DeepSeek V4 HiCache currently rejects --hicache-size and supports
# capacity control only through a host/device token-capacity ratio.
# DSv4 exposes capacity as a host/device token ratio rather than bytes.
# B200 ratio=8 stays below the configured host-memory capacity for the
# currently supported TP8 shape.
DEFAULT_HICACHE_RATIO=8
# DEP8 shards the host pools and fits ratio=8 on NScale. The replicated
# TP8 pools need a lower ratio: 2.75 allocates about 121 GiB per rank and
# leaves startup headroom on the 1.7 TiB NScale hosts.
DEFAULT_HICACHE_RATIO=2.75
if [ "$DP_ATTENTION" = "true" ]; then
DEFAULT_HICACHE_RATIO=8
fi
HICACHE_RATIO="${HICACHE_RATIO:-$DEFAULT_HICACHE_RATIO}"
if [ "$HICACHE_RATIO" -gt "$DEFAULT_HICACHE_RATIO" ]; then
if awk -v ratio="$HICACHE_RATIO" -v max="$DEFAULT_HICACHE_RATIO" 'BEGIN { exit !(ratio > max) }'; then
echo "Error: HICACHE_RATIO=$HICACHE_RATIO exceeds configured limit $DEFAULT_HICACHE_RATIO" >&2
exit 1
fi
Expand All @@ -93,6 +97,7 @@ SGLANG_BACKEND_PORT="$PORT"
ROUTER_LOG="$RESULT_DIR/router.log"
if [ "$DP_ATTENTION" = "true" ]; then
USE_SGLANG_ROUTER=true
ROUTER_POLICY_ARGS=()
export AIPERF_HTTP_X_SMG_ROUTING_KEY_FROM_CORRELATION_ID=true
SGLANG_BACKEND_PORT=$((PORT + 1))
SGLANG_ROUTER_METRICS_PORT=$((PORT + 10000))
Expand All @@ -103,15 +108,29 @@ PARALLEL_ARGS=(--tp "$TP")
METRICS_ARGS=(--enable-metrics --enable-cache-report)
CHUNKED_PREFILL_SIZE=8192
SWA_FULL_TOKENS_RATIO=0.1
MEM_FRACTION_STATIC=0.90
if [ "$DP_ATTENTION" = "true" ]; then
export SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1
export SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1
export SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320

# Leave HBM headroom for the FP4 indexer's context-dependent workspace.
MEM_FRACTION_STATIC=0.88
PREFILL_DECODE_INTERVAL=24

# Keep DP admission and session routing uniform across the DEP8 curve.
PARALLEL_ARGS+=(--load-balance-method total_requests)
METRICS_ARGS+=(--load-snapshot-publish-interval 1)
export AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID=true
if [ "$CONC" -eq 160 ]; then
PREFILL_DECODE_INTERVAL=20
ROUTER_POLICY_ARGS+=(--balance-abs-threshold 32)
fi

PARALLEL_ARGS+=(
--dp "$TP"
--tokenizer-worker-num "$TP"
--enable-prefill-delayer
--prefill-decode-interval 10
--prefill-decode-interval "$PREFILL_DECODE_INTERVAL"
--enable-dp-attention
--enable-dp-attention-local-control-broadcast
--incremental-streaming-output
Expand All @@ -123,9 +142,9 @@ if [ "$DP_ATTENTION" = "true" ]; then
--disable-shared-experts-fusion
--disable-flashinfer-autotune
)
# SGLang divides this global budget by dp_size. Keep the tuned 8192-token
# per-rank budget for every DP-attention topology.
CHUNKED_PREFILL_SIZE=$((8192 * TP))
# SGLang divides this global budget by dp_size. Keep 6144 tokens per rank
# for every DP-attention profile so the FP4 indexer retains HBM headroom.
CHUNKED_PREFILL_SIZE=$((6144 * TP))
SWA_FULL_TOKENS_RATIO=0.02
else
PARALLEL_ARGS+=(
Expand All @@ -138,14 +157,16 @@ fi
# The B200-specialized image deadlocks immediately after weight loading when
# forced through the B300 compressed-attention/page-size overrides.
# DeepGEMM's DSv4 indexer needs a multi-GiB temporary allocation at long
# contexts. Leave the same HBM headroom used by the B300 recipe so a nearly
# full GPU KV cache does not OOM while HiCache is spilling to host memory.
MEM_FRACTION_STATIC=0.88
# contexts. The selected fractions preserve the measured indexer and CUDA
# graph headroom while HiCache spills to host memory.

# AgentX concurrency counts live session trees, not individual requests.
# Allow subagent fan-out to exceed CONC without clipping request bursts.
MAX_RUNNING_REQUESTS=$((2 * CONC))
CUDA_GRAPH_MAX_BS=$((2 * CONC))
if [ "$DP_ATTENTION" = "true" ]; then
CUDA_GRAPH_MAX_BS=32
fi
CUDA_GRAPH_ARGS=(--cuda-graph-max-bs "$CUDA_GRAPH_MAX_BS")

export PYTHONNOUSERSITE=1
Expand Down Expand Up @@ -205,6 +226,7 @@ SGLANG_CMD=(
# across local ranks so post-load weight repacking reads from page cache
# instead of issuing redundant fragmented mmap faults from every rank.
--weight-loader-prefetch-checkpoints
--model-loader-extra-config '{"enable_multithread_load": true}'
"${METRICS_ARGS[@]}"
"${CACHE_ARGS[@]}"
)
Expand Down Expand Up @@ -241,7 +263,8 @@ if [ "$USE_SGLANG_ROUTER" = "true" ]; then
echo "Starting SGLang router on port $PORT for $TP DP ranks..."
"${SGLANG_ROUTER_CMD[@]}" \
--worker-urls "http://localhost:$SGLANG_BACKEND_PORT" \
--policy consistent_hashing \
--policy cache_aware \
"${ROUTER_POLICY_ARGS[@]}" \
--request-id-headers x-correlation-id \
--dp-aware \
--host 0.0.0.0 \
Expand Down
6 changes: 3 additions & 3 deletions configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -929,7 +929,7 @@ dsv4-fp4-b200-sglang:
- { tp: 8, ep: 8, dp-attn: true, conc-start: 256, conc-end: 1024 }

dsv4-fp4-b200-sglang-agentic-hicache-mtp:
image: lmsysorg/sglang:dev-nightly-0820
image: lmsysorg/sglang:nightly-dev-20260827-20621aa1
model: deepseek-ai/DeepSeek-V4-Pro
model-prefix: dsv4
runner: cluster:b200-nscale
Expand All @@ -941,8 +941,8 @@ dsv4-fp4-b200-sglang-agentic-hicache-mtp:
- dram-utilization: 0.80
search-space:
- { tp: 8, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 2, 3, 4, 5] }
- { tp: 8, kv-offloading: dram, kv-offload-backend: { name: hicache }, spec-decoding: mtp, conc-list: [8, 10, 16, 32] }
- { tp: 8, ep: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, spec-decoding: mtp, conc-list: [32, 64, 96], router: { name: sglang-router, version: "0.3.2" } }
- { tp: 8, kv-offloading: dram, kv-offload-backend: { name: hicache }, spec-decoding: mtp, conc-list: [8, 10, 16] }
- { tp: 8, ep: 8, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, spec-decoding: mtp, conc-list: [64, 96, 128, 160], router: { name: sglang-router, version: "0.3.2" } }

dsv4-fp4-b200-vllm:
image: vllm/vllm-openai:v0.25.0
Expand Down
12 changes: 12 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -6576,3 +6576,15 @@
- "Recipes sourced from srt-slurm (recipes/trtllm/qwen3.5-fp4/inferencex/gb300/{mtp,stp})."
- "Runner: launch_gb300-nv.sh bumped from NVIDIA/srt-slurm@v1.0.29 to v1.0.72 for the dynamo-trt+qwen3.5+fp4 path."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2730

- config-keys:
- dsv4-fp4-b200-sglang-agentic-hicache-mtp
scenario-type:
- agentic-coding
description:
- "Refresh the B200 SGLang AgentX curve with validated TP8/DP8/EP8 HiCache profiles at DP-attention concurrencies 64, 96, 128, and 160."
- "Use lmsysorg/sglang:nightly-dev-20260827-20621aa1 for the B200 SGLang AgentX sweep."
- "Scope cache-aware routing and --prefill-decode-interval to the DP-attention profiles: use interval 24 at concurrency 64, 96, and 128, and interval 20 at concurrency 160."
- "Use mem-fraction-static 0.88 for every DP-attention profile while retaining 0.90 for TP profiles. Full c96 and c128 sweeps at 0.90 exhausted HBM in the context-dependent FP4-indexer workspace; lowering the fraction adds about 3.57 GiB of headroom per GPU. Across every DP-attention profile, use a 49152-token global chunked-prefill budget, CUDA graph maximum batch size 32, total-requests DP balancing, one-second load snapshots, and correlation-ID session routing; set the cache-aware absolute balance threshold to 32 at concurrency 160."
- "Fit the HiCache tier to NScale host memory by using ratio 2.75 for replicated TP8 profiles and retaining ratio 8 for sharded DP8 profiles. Ratio 8 on TP8 requests about 351 GiB per rank and fails host-pool initialization on a 1.7 TiB NScale node; ratio 2.75 targets about 121 GiB per rank while leaving startup headroom."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2718
Loading