From f97a1202bbe758a6a6c095b0056bb761ef1cf530 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Thu, 27 Aug 2026 00:35:48 -0400 Subject: [PATCH 1/3] qwen3.8next-fp8-h100-sglang-agentic-mtp: day-zero Qwen3.8-Flash-Next AgentX on H100 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add the Qwen3.8-Flash-Next AgentX recipe on H100, served by SGLang with native NEXTN MTP. H100 is Hopper, so FP8: NVFP4 needs SM100 tensor cores. The SGLang cookbook does not list H100, so this is the H200 arm adjusted for the smaller part rather than a verified command. TP8 with EP8 instead of the cookbook's TP4/EP4, because 172.8 GiB at TP4 is ~43 GiB per rank of an 80 GB card and leaves too little for the 256k-capped agentic traces; TP8 halves that. Memory fraction 0.75 rather than 0.85, matching the Qwen3.5 H100 sibling. The SSM state is float32, as Hopper's flashinfer verify kernel requires and unlike the bfloat16 the Blackwell arms must use. No launcher change: the H100 model-path gate is inside the multinode branch, so the single-node path leaves MODEL_PATH unset and the bench script downloads into the mounted HuggingFace cache. 新增 H100 上的 Qwen3.8-Flash-Next AgentX 配方,由 SGLang 以原生 NEXTN MTP 提供 服务。H100 属 Hopper 架构,故使用 FP8(NVFP4 需要 SM100 张量核心)。 SGLang cookbook 未列出 H100,因此本配方是按较小硬件调整后的 H200 分支,而非官方 验证命令:采用 TP8/EP8 而非 cookbook 的 TP4/EP4——172.8 GiB 在 TP4 下每卡约 43 GiB,对 80 GB 显存而言留给 256k 智能体轨迹的空间过少,TP8 可将其减半;显存 占用取 0.75 而非 0.85,与 Qwen3.5 H100 同类配方一致。SSM 状态为 float32,这是 Hopper 上 flashinfer 验证内核的要求,与 Blackwell 分支必须使用的 bfloat16 相反。 无需改动 launcher:H100 的权重路径分支位于多节点条件内,单节点路径下 MODEL_PATH 保持未设置,基准脚本会下载到已挂载的 HuggingFace 缓存。 Co-Authored-By: Claude Opus 5 (1M context) --- .../agentic/qwen3.8next_fp8_h100_mtp.sh | 219 ++++++++++++++++++ configs/nvidia-master.yaml | 19 ++ perf-changelog.yaml | 10 + 3 files changed, 248 insertions(+) create mode 100755 benchmarks/single_node/agentic/qwen3.8next_fp8_h100_mtp.sh diff --git a/benchmarks/single_node/agentic/qwen3.8next_fp8_h100_mtp.sh b/benchmarks/single_node/agentic/qwen3.8next_fp8_h100_mtp.sh new file mode 100755 index 0000000000..99c4b35a78 --- /dev/null +++ b/benchmarks/single_node/agentic/qwen3.8next_fp8_h100_mtp.sh @@ -0,0 +1,219 @@ +#!/usr/bin/env bash +set -euo pipefail +set -x + +# Agentic trace replay benchmark for Qwen3.8-Flash-Next FP8 on H100 using +# SGLang with MTP speculative decoding. Day-zero recipe; SGLang is the +# plan-of-record engine for this model (MODELS.md), and it is spec-decode only, +# per the AgentX policy that new agentic arms ship with speculative decoding +# enabled rather than as an STP/MTP A/B. +# +# H100 is Hopper, so this arm is FP8 (Qwen/Qwen3.8-Flash-Next-FP8, 172.8 GiB) +# rather than the NVFP4 checkpoint the Blackwell arms use: NVFP4 needs SM100 +# tensor cores. The SGLang cookbook does not offer H100 at all, so this recipe +# is the H200 arm adjusted for the smaller part rather than a verified command: +# * TP8/EP8 instead of the cookbook's TP4/EP4. At TP4 the 172.8 GiB +# checkpoint is ~43 GiB per rank of an 80 GB card, which leaves too little +# for the 256k-capped agentic traces. TP8 halves that to ~22 GiB. +# * --mem-fraction-static 0.75 rather than 0.85, matching the Qwen3.5 H100 +# sibling: 80 GB HBM3 has far less slack than H200's 141 GB HBM3e. +# +# Structure follows the proven H100 MTP AgentX replay path (HiCache host-DRAM +# offload, the multi_tokenizer cached_tokens_details patch, aiperf-driven trace +# replay). Attention stays on the flashinfer linear-attention backends (sm_90); +# the trtllm_mha path is Blackwell-only. +# +# Speculative decoding is SGLANG_ENABLE_SPEC_V2=1 with NEXTN, 3 steps, +# eagle-topk 1 and 4 draft tokens, i.e. 3 speculative tokens per verification +# step, matching every other Qwen3.8-Flash-Next arm. +# +# Throughput runs pin acceptance to the committed golden AL through SGLang's +# simulated-acceptance path; the EVAL_ONLY accuracy run leaves it off and keeps +# real verification. See the SGLANG_SIMULATE_ACC_* block. +# +# Required env vars: +# MODEL, TP, CONC, KV_OFFLOADING, TOTAL_CPU_DRAM_GB, RESULT_DIR +# +# KV_OFFLOADING=dram requires KV_OFFLOAD_BACKEND=hicache. + +source "$(dirname "$0")/../../benchmark_lib.sh" + +check_env_vars MODEL TP CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION EP_SIZE + +SCHEDULER_RECV_INTERVAL=${SCHEDULER_RECV_INTERVAL:-10} + +if [[ -n "${SLURM_JOB_ID:-}" ]]; then + echo "JOB $SLURM_JOB_ID running on ${SLURMD_NODENAME:-unknown}" +fi + +# `hf download` creates the target dir if missing and is itself idempotent. +# When MODEL_PATH is unset (stand-alone runs), fall back to the HF_HUB_CACHE +# Either way, MODEL_PATH is what the server is launched with. +if [[ -n "${MODEL_PATH:-}" ]]; then + if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then + hf download "$MODEL" --local-dir "$MODEL_PATH" + fi +else + hf download "$MODEL" + export MODEL_PATH="$MODEL" +fi +nvidia-smi + +# ---- Resolve traces and install deps ---------------------------------------- +# Keep the 256k-capped with-subagents corpus the H100 Qwen3.5 AgentX recipe +# uses (470 traces, max in+out <= 256k). The unfiltered corpus has requests up +# to ~1M proxy tokens that the server would reject, and H100's 80 GB is the +# tightest part in this set, so the capped corpus matters most here. +export WEKA_LOADER_OVERRIDE=semianalysis_cc_traces_weka_with_subagents_256k + +resolve_trace_source +install_agentic_deps + +# ---- Server config ---------------------------------------------------------- +SERVER_LOG="$RESULT_DIR/server.log" +mkdir -p "$RESULT_DIR" + +CACHE_ARGS=() +if require_agentic_kv_offload_backend hicache; then + # HiCache extends RadixAttention, so do not pass --disable-radix-cache. + # Hybrid GDN/Mamba allocates one KV and one Mamba host pool per rank. + REQUESTED_HICACHE_TOTAL_GB="${HICACHE_TOTAL_CPU_DRAM_GB:-$TOTAL_CPU_DRAM_GB}" + if [ "$REQUESTED_HICACHE_TOTAL_GB" -gt "$TOTAL_CPU_DRAM_GB" ]; then + echo "Error: requested HiCache pool ${REQUESTED_HICACHE_TOTAL_GB} GB exceeds configured capacity ${TOTAL_CPU_DRAM_GB} GB" >&2 + exit 1 + fi + TOTAL_CPU_DRAM_GB="$REQUESTED_HICACHE_TOTAL_GB" + HICACHE_HOST_POOL_COUNT="${HICACHE_HOST_POOL_COUNT:-2}" + HICACHE_WRITE_POLICY="${HICACHE_WRITE_POLICY:-write_through_selective}" + MAX_HICACHE_SIZE_GB=$((TOTAL_CPU_DRAM_GB / TP / HICACHE_HOST_POOL_COUNT)) + HICACHE_SIZE_GB="${HICACHE_SIZE_GB:-$MAX_HICACHE_SIZE_GB}" + if [ "$HICACHE_SIZE_GB" -gt "$MAX_HICACHE_SIZE_GB" ]; then + echo "Error: HICACHE_SIZE_GB=$HICACHE_SIZE_GB exceeds configured per-pool limit $MAX_HICACHE_SIZE_GB" >&2 + exit 1 + fi + if [ "$HICACHE_SIZE_GB" -lt 1 ]; then + echo "Error: computed HICACHE_SIZE_GB=$HICACHE_SIZE_GB from TOTAL_CPU_DRAM_GB=$TOTAL_CPU_DRAM_GB, TP=$TP, HICACHE_HOST_POOL_COUNT=$HICACHE_HOST_POOL_COUNT" >&2 + exit 1 + fi + echo "HiCache CPU pool: ${HICACHE_SIZE_GB} GB per rank per host pool across TP=${TP}, host_pool_count=${HICACHE_HOST_POOL_COUNT}" + CACHE_ARGS=( + --page-size 64 + --enable-hierarchical-cache + --hicache-size "$HICACHE_SIZE_GB" + --hicache-io-backend kernel + --hicache-mem-layout page_first + --hicache-write-policy "$HICACHE_WRITE_POLICY" + ) +fi + +echo "Starting SGLang server..." +export PYTHONNOUSERSITE=1 +export SGLANG_ENABLE_SPEC_V2=1 + +# 3 speculative tokens per step (num-steps 3, eagle-topk 1, 4 draft tokens), +# the same MTP shape as the fixed-seq-len Qwen3.5 recipes. +SPEC_ARGS=( + --speculative-algorithm NEXTN + --speculative-num-steps 3 + --speculative-eagle-topk 1 + --speculative-num-draft-tokens 4 +) + +# AgentX pins acceptance to the committed golden AL so submissions are compared +# on system performance at a fixed acceptance target rather than on draft-head +# quality (golden_al_distribution/README.md). 3.39 is the Qwen3.5 MTP curve at +# num_speculative_tokens=3, thinking_on (golden_al_distribution/qwen3.5_mtp.yaml) +# -- the same value the GB300 Qwen3.5 AgentX srt-slurm recipes pin. +# SGLANG_SIMULATE_ACC_TOKEN_MODE landed in SGLang v0.5.16, which is why this +# recipe pins that image rather than the non-MTP agentic sibling's v0.5.12. +# +# EVAL_ONLY leaves simulated acceptance off: it commits drafted tokens +# regardless of the target logits, so generated text is wrong and the eval would +# score ~0. +if [ "${EVAL_ONLY:-false}" != "true" ]; then + # golden_al_distribution/qwen3.8next_mtp.yaml: + # qwen3.8-flash-next-fp8.thinking_on[3] = 2.32. + # --speculative-num-steps 3 with 4 draft tokens is 3 speculative tokens + # per verification step, i.e. the MTP=3 cell. AgentX replays run with + # thinking on, so the thinking_on row is the right one. + export SGLANG_SIMULATE_ACC_LEN=2.32 + export SGLANG_SIMULATE_ACC_METHOD=match-expected + export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token +fi + +SGLANG_MULTI_TOKENIZER=/sgl-workspace/sglang/python/sglang/srt/managers/multi_tokenizer_mixin.py +if ! sed -n '/elif isinstance(output, BatchStrOutput):/,/input_token_logprobs_val=_extract_field_by_index/p' "$SGLANG_MULTI_TOKENIZER" \ + | grep -q 'cached_tokens_details=_extract_field_by_index'; then + sed -i '/elif isinstance(output, BatchStrOutput):/,/input_token_logprobs_val=_extract_field_by_index/ { + /cached_tokens=_extract_field_by_index(output, "cached_tokens", i),/a\ + cached_tokens_details=_extract_field_by_index(\ + output, "cached_tokens_details", i\ + ), + }' "$SGLANG_MULTI_TOKENIZER" +fi + +{ set +x; } 2>/dev/null +# AgentX concurrency counts live session trees rather than individual HTTP +# requests. Leave room for subagent fan-out, and do not spend HBM capturing +# graphs above the batch sizes that stay useful for this long-context workload. +# The Qwen3.5 H200 template left both flags commented out, so neither variable +# existed; NEXTN silently caps --max-running-requests at 48 when it is unset. +MAX_RUNNING_REQUESTS=$((2 * CONC)) +CUDA_GRAPH_MAX_BS="$CONC" +if [ "$CUDA_GRAPH_MAX_BS" -gt 64 ]; then + CUDA_GRAPH_MAX_BS=64 +fi + +SGLANG_CMD=( + python3 -m sglang.launch_server + --model-path "$MODEL_PATH" + --served-model-name "$MODEL" + --host 0.0.0.0 + --port "$PORT" + --trust-remote-code + # Verified flags from the SGLang cookbook playground for this model on + # H200 / FP8 / low latency / single node, adjusted for H100's 80 GB. + # NVFP4 is greyed out for Hopper, so FP8 is the whole surface here. + --tp-size "$TP" + --ep-size "$EP_SIZE" + --dp-size 1 + --mem-fraction-static 0.75 + --chunked-prefill-size 8192 + --linear-attn-prefill-backend flashinfer + --linear-attn-decode-backend flashinfer + # float32, not the cookbook's bfloat16. With NEXTN enabled the GDN linear + # attention backend routes verification through flashinfer's + # gated_delta_rule_mtp, which asserts initial_state.dtype == torch.float32 + # and aborts CUDA graph capture on a bf16 SSM state: + # AssertionError: initial_state must be float32, got torch.bfloat16 + # flashinfer/gdn_decode.py:761, via gdn_backend.py target_verify + # The cookbook command pairs bfloat16 with NEXTN, but this flashinfer build + # rejects that combination, and the state dtype is the half that can move. + --mamba-ssm-dtype float32 + "${SPEC_ARGS[@]}" + --reasoning-parser auto + # NEXTN silently resets --max-running-requests to 48 when it is unset, so + # this must stay explicit and sized to the AgentX concurrency. + --max-running-requests "$MAX_RUNNING_REQUESTS" + --cuda-graph-max-bs "$CUDA_GRAPH_MAX_BS" + --stream-interval 50 + --scheduler-recv-interval "$SCHEDULER_RECV_INTERVAL" + --tokenizer-worker-num 6 + --tokenizer-path "$MODEL" + --enable-metrics + "${CACHE_ARGS[@]}" +) +printf '%q ' "${SGLANG_CMD[@]}" | tee "$RESULT_DIR/sglang_command.txt" +printf '\n' | tee -a "$RESULT_DIR/sglang_command.txt" +"${SGLANG_CMD[@]}" > "$SERVER_LOG" 2>&1 & +SERVER_PID=$! +echo "Server PID: $SERVER_PID" + +wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" + +if [ "${EVAL_ONLY}" = "true" ]; then + run_eval --port "$PORT" +else + build_replay_cmd "$RESULT_DIR" + run_agentic_replay_and_write_outputs "$RESULT_DIR" +fi diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 4c1036915d..c970a6eb44 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -7217,6 +7217,25 @@ qwen3.5-fp8-h100-sglang-agentic-mtp: - { tp: 8, ep: 8, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [4, 8, 12, 16] } + +# Qwen3.8-Flash-Next FP8 AgentX on H100 via SGLang with native NEXTN MTP. +# Day-zero recipe. H100 is Hopper, so FP8: NVFP4 needs SM100 tensor cores. The +# SGLang cookbook does not list H100, so this mirrors the H200 arm adjusted for +# the smaller part: TP8/EP8 rather than the cookbook's TP4/EP4, since 172.8 GiB +# at TP4 leaves too little of an 80 GB card for the 256k-capped traces. +qwen3.8next-fp8-h100-sglang-agentic-mtp: + image: lmsysorg/sglang:qwen38flashnext + model: Qwen/Qwen3.8-Flash-Next-FP8 + model-prefix: qwen3.8next + runner: cluster:h100-dgxc + precision: fp8 + framework: sglang + multinode: false + scenarios: + agentic-coding: + - dram-utilization: 0.8 + search-space: + - { tp: 8, ep: 8, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 4, 8, 12, 16] } qwen3.5-fp4-b200-trt: image: nvcr.io#nvidia/tensorrt-llm/release:1.3.0rc18 model: nvidia/Qwen3.5-397B-A17B-NVFP4 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index de9533314a..b4c33aead1 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -6490,3 +6490,13 @@ - "Required lanes promote exporter startup timeouts, launch failures, and endpoint resolution failures to blocking validation failures, so a recipe cannot publish a result whose power collection never started." - "Route only enabled recipes through the immutable producer fork and preserve non-power launcher revisions." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2688 +- config-keys: + - qwen3.8next-fp8-h100-sglang-agentic-mtp + scenario-type: + - agentic-coding + description: + - "Add the day-zero Qwen3.8-Flash-Next FP8 AgentX recipe on H100 with SGLang native NEXTN MTP at TP8 with EP8 and concurrency 1/4/8/12/16." + - "Mirror the H200 arm, adjusted for the smaller part: TP8 with EP8 rather than TP4 with EP4, and memory fraction 0.75 rather than 0.85." + - "Use a float32 Mamba SSM state, as Hopper's flashinfer verify kernel requires, unlike the bfloat16 the Blackwell arms must use." + - "Pin throughput runs to the committed golden thinking_on acceptance length of 2.32 at three speculative tokens; eval-only runs keep real target verification." + pr-link: TBD From 7bf9c3665f9c0b4dcc82ec962e6d61d85356a43e Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Thu, 27 Aug 2026 00:35:58 -0400 Subject: [PATCH 2/3] =?UTF-8?q?Fill=20perf-changelog=20pr-link=20for=20#27?= =?UTF-8?q?56=20/=20=E8=A1=A5=E5=85=A8=20#2756=20=E7=9A=84=20perf-changelo?= =?UTF-8?q?g=20pr-link?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 5 (1M context) --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index b4c33aead1..08d8a94f15 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -6499,4 +6499,4 @@ - "Mirror the H200 arm, adjusted for the smaller part: TP8 with EP8 rather than TP4 with EP4, and memory fraction 0.75 rather than 0.85." - "Use a float32 Mamba SSM state, as Hopper's flashinfer verify kernel requires, unlike the bfloat16 the Blackwell arms must use." - "Pin throughput runs to the committed golden thinking_on acceptance length of 2.32 at three speculative tokens; eval-only runs keep real target verification." - pr-link: TBD + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2756 From 54a903b833fa151a23e2965355fc1f0093244e0b Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Thu, 27 Aug 2026 02:28:01 -0400 Subject: [PATCH 3/3] Separate the appended changelog entry from history with a blank line MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Same fix as PRs #2753 and #2752. validate_perf_changelog.py requires the appended suffix to start with "\n- config-keys:" unless the base file already ends in a blank line, and merge_with_reuse.sh refuses the merge without it. Historical bytes were already exact; this only inserts the separator. 与 PR #2753、#2752 相同的修复:追加条目必须与历史之间空一行,否则 merge_with_reuse.sh 会拒绝合并。历史字节本就完全一致,此处仅插入该分隔空行。 Co-Authored-By: Claude Opus 5 (1M context) --- perf-changelog.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 08d8a94f15..ee0daa64e9 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -6490,6 +6490,7 @@ - "Required lanes promote exporter startup timeouts, launch failures, and endpoint resolution failures to blocking validation failures, so a recipe cannot publish a result whose power collection never started." - "Route only enabled recipes through the immutable producer fork and preserve non-power launcher revisions." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2688 + - config-keys: - qwen3.8next-fp8-h100-sglang-agentic-mtp scenario-type: