Skip to content
249 changes: 249 additions & 0 deletions benchmarks/single_node/agentic/minimaxm3_fp8_h200_mtp.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,249 @@
#!/usr/bin/env bash
set -euo pipefail
set -x

# MiniMax-M3 MXFP8 H200 AgentX (agentic-coding) recipe with EAGLE3 speculative
# decoding — the spec-decoding=mtp variant of agentic/minimaxm3_fp8_h200.sh.
# Everything outside the speculative block mirrors the non-MTP agentic sibling
# (Mooncake host-DRAM KV offload, --block-size 128, --language-model-only,
# --kv-cache-dtype fp8, TRITON_ATTN, minimax_m3 parsers, vllm-router for
# DP-attention), so the spec-decode delta is readable at equal concurrency.
#
# One deliberate exception: --gpu-memory-utilization is 0.90, not the non-MTP
# sibling's 0.92. That matches fixed_seq_len/minimaxm3_fp8_h200_mtp.sh, the
# proven MTP configuration on this SKU. The H100 twin of this recipe died
# mid-warmup at the sibling's higher value with torch.OutOfMemoryError on every
# rank once the EAGLE3 head and its KV were resident (run 30515793863).
#
# Speculative config: Inferact/MiniMax-M3-EAGLE3 draft head, 3 speculative
# tokens — the same draft/level as every merged MiniMax-M3 MTP recipe. The
# drafter is pinned to FLASH_ATTN as on the other CUDA M3 MTP recipes: the
# EAGLE3 head is MHA and FlashInfer only serves page size 128 through its
# trtllm-gen kernel, which requires GQA/MQA.
#
# Throughput runs pin synthetic acceptance to the committed golden AL; the
# EVAL_ONLY accuracy run keeps real target verification. See SYNTHETIC_ACCEPT_LEN.

source "$(dirname "$0")/../../benchmark_lib.sh"

check_env_vars MODEL TP CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION EP_SIZE DP_ATTENTION

DRAFT_MODEL="Inferact/MiniMax-M3-EAGLE3"

if [[ -n "${SLURM_JOB_ID:-}" ]]; then
echo "JOB $SLURM_JOB_ID running on ${SLURMD_NODENAME:-unknown}"
fi

resolve_complete_model_snapshot() {
python3 - "$1" <<'PY'
import json
import sys
from pathlib import Path

model_cache_dir = Path(sys.argv[1])
try:
revision = model_cache_dir.joinpath("refs/main").read_text().strip()
except OSError:
raise SystemExit

if not revision or Path(revision).name != revision:
raise SystemExit

snapshot = model_cache_dir / "snapshots" / revision
index_path = snapshot / "model.safetensors.index.json"
required_files = (
snapshot / "config.json",
snapshot / "tokenizer_config.json",
index_path,
)
if not all(path.is_file() for path in required_files):
raise SystemExit
try:
weight_map = json.loads(index_path.read_text())["weight_map"]
except (KeyError, json.JSONDecodeError, OSError):
raise SystemExit
shards = {snapshot / filename for filename in weight_map.values()}
if shards and all(path.is_file() for path in shards):
print(snapshot)
PY
}

if [[ -n "${MODEL_PATH:-}" ]]; then
if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then
hf download "$MODEL" --local-dir "$MODEL_PATH"
fi
else
MODEL_CACHE_ROOT="${HF_HUB_CACHE:-${HF_HOME:-$HOME/.cache/huggingface/hub}}"
MODEL_CACHE_DIR="$MODEL_CACHE_ROOT/models--${MODEL//\//--}"
mkdir -p "$MODEL_CACHE_ROOT"
MODEL_PATH=$(resolve_complete_model_snapshot "$MODEL_CACHE_DIR")
if [[ -z "$MODEL_PATH" ]]; then
exec 9>"$MODEL_CACHE_ROOT/.minimaxm3-download.lock"
flock -w 3600 9
MODEL_PATH=$(resolve_complete_model_snapshot "$MODEL_CACHE_DIR")
if [[ -z "$MODEL_PATH" ]]; then
DOWNLOADED_MODEL_PATH=$(hf download "$MODEL")
MODEL_PATH=$(resolve_complete_model_snapshot "$MODEL_CACHE_DIR")
if [[ -z "$MODEL_PATH" ]]; then
echo "Downloaded model snapshot is incomplete: $DOWNLOADED_MODEL_PATH" >&2
exit 1
fi
fi
flock -u 9
fi
echo "Using complete cached model snapshot: $MODEL_PATH"
export MODEL_PATH
fi

# The EAGLE3 draft is never pre-staged next to the target checkpoint; fetch it
# into the shared HF cache. That cache is a network FS where concurrent
# day-zero downloads hit huggingface_hub's WeakFileLock "[Errno 116] Stale file
# handle" race, so retry (the download resumes) as the fixed-seq-len MTP
# recipes do.
for attempt in 1 2 3 4 5; do
hf download "$DRAFT_MODEL" && break
if [ "$attempt" = 5 ]; then echo "hf download of $DRAFT_MODEL failed after $attempt attempts" >&2; exit 1; fi
echo "hf download attempt $attempt failed; retrying in 60s" >&2
sleep 60
done
nvidia-smi

export WEKA_LOADER_OVERRIDE=semianalysis_cc_traces_weka_062126
resolve_trace_source
install_agentic_deps

export VLLM_ENGINE_READY_TIMEOUT_S=3600
export PYTHONNOUSERSITE=1

SERVER_LOG="$RESULT_DIR/server.log"
ROUTER_LOG="$RESULT_DIR/router.log"
MOONCAKE_MASTER_LOG="$RESULT_DIR/mooncake_master.log"
mkdir -p "$RESULT_DIR"

OFFLOAD_ARGS=()
if require_agentic_kv_offload_backend mooncake; then
PER_RANK_GB=$((TOTAL_CPU_DRAM_GB / TP))
MOONCAKE_VERSION=0.3.11.post1
agentic_pip_install --quiet --no-cache-dir --no-deps \
--force-reinstall "mooncake-transfer-engine-cuda13==$MOONCAKE_VERSION"
python3 -c "from mooncake.store import MooncakeDistributedStore" >/dev/null
MOONCAKE_MASTER_PORT=$((PORT + 12000))
MOONCAKE_CONFIG_PATH="$RESULT_DIR/mooncake_config.json"
cat > "$MOONCAKE_CONFIG_PATH" <<EOF
{
"mode": "embedded",
"metadata_server": "P2PHANDSHAKE",
"master_server_address": "127.0.0.1:$MOONCAKE_MASTER_PORT",
"global_segment_size": "${PER_RANK_GB}GB",
"local_buffer_size": "4GB",
"protocol": "rdma",
"device_name": "",
"enable_offload": false
}
EOF
export MOONCAKE_CONFIG_PATH PYTHONHASHSEED=0 MC_SLICE_SIZE=1048576 MC_WORKERS_PER_CTX=4
export MC_ENABLE_DEST_DEVICE_AFFINITY=1
mooncake_master --port "$MOONCAKE_MASTER_PORT" \
--eviction_high_watermark_ratio=0.80 \
--eviction_ratio=0.10 > "$MOONCAKE_MASTER_LOG" 2>&1 &
MOONCAKE_MASTER_PID=$!
sleep 2
kill -0 "$MOONCAKE_MASTER_PID"
OFFLOAD_ARGS=(
--kv-transfer-config
'{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_connector_extra_config":{"load_async":true}}'
)
fi

PARALLEL_ARGS=(--tensor-parallel-size "$TP" --data-parallel-size 1)
if [[ "$DP_ATTENTION" == "true" ]]; then
PARALLEL_ARGS=(--tensor-parallel-size 1 --data-parallel-size "$TP")
fi

EP_ARGS=()
if (( EP_SIZE > 1 )); then
EP_ARGS=(--enable-expert-parallel)
fi

VLLM_BACKEND_PORT="$PORT"
if [[ "$DP_ATTENTION" == "true" ]]; then
VLLM_BACKEND_PORT=$((PORT + 1))
export AIPERF_HTTP_X_SESSION_ID_FROM_CORRELATION_ID=1
agentic_pip_install --quiet 'vllm-router==0.1.14'
fi

# use 3 speculative tokens for all configs, matching the MiniMax-M3 MTP recipes
NUM_SPEC_TOKENS=3
TOKENS_PER_SEQ=$((1 + NUM_SPEC_TOKENS))

# AgentX pins acceptance to the committed golden AL so submissions are compared
# on system performance at a fixed acceptance target rather than on draft-head
# quality (golden_al_distribution/README.md). 2.83 is the MiniMax-M3 EAGLE3
# curve at num_speculative_tokens=3, thinking_on
# (golden_al_distribution/minimaxm3_eagle3.yaml). The separate
# minimaxm3_eagle3_gqa.yaml curve belongs to the Inferact/MiniMax-M3-EAGLE3-GQA
# draft and is not mixed in here.
#
# EVAL_ONLY switches back to real verification: synthetic acceptance commits
# drafted tokens regardless of the target logits, so generated text is wrong and
# the eval would score ~0 (same split as dsv4_fp4_b*_vllm_mtp.sh).
SYNTHETIC_ACCEPT_LEN=2.83
if [ "${EVAL_ONLY:-false}" = "true" ]; then
SPEC_CONFIG="{\"method\": \"eagle3\", \"model\": \"$DRAFT_MODEL\", \"num_speculative_tokens\": $NUM_SPEC_TOKENS, \"attention_backend\": \"FLASH_ATTN\"}"
else
SPEC_CONFIG="{\"method\": \"eagle3\", \"model\": \"$DRAFT_MODEL\", \"num_speculative_tokens\": $NUM_SPEC_TOKENS, \"attention_backend\": \"FLASH_ATTN\", \"rejection_sample_method\": \"synthetic\", \"synthetic_acceptance_length\": $SYNTHETIC_ACCEPT_LEN}"
fi

# AgentX concurrency counts live session trees, not individual requests, so keep
# the non-MTP recipe's 2x scheduler headroom for subagent fan-out.
MAX_NUM_SEQS=$((2 * CONC))
# Cudagraph capture sizes are in TOKENS: a decode batch of S sequences verifies
# S*(1+NUM_SPEC_TOKENS) tokens, so cap capture at MAX_NUM_SEQS*(1+N) or the
# FULL_DECODE_ONLY ladder tops out at MAX_NUM_SEQS/(1+N) sequences and the
# largest decode batches fall back to eager.
MAX_CUDAGRAPH_CAPTURE_SIZE=$((MAX_NUM_SEQS * TOKENS_PER_SEQ))

vllm serve "$MODEL_PATH" --served-model-name "$MODEL" \
--host 0.0.0.0 \
--port "$VLLM_BACKEND_PORT" \
"${PARALLEL_ARGS[@]}" \
"${EP_ARGS[@]}" \
--gpu-memory-utilization 0.90 \
--kv-cache-dtype fp8 \
--attention-backend TRITON_ATTN \
--block-size 128 \
--language-model-only \
--enable-prefix-caching \
--max-num-seqs "$MAX_NUM_SEQS" \
--max-cudagraph-capture-size "$MAX_CUDAGRAPH_CAPTURE_SIZE" \
--speculative-config "$SPEC_CONFIG" \
--tool-call-parser minimax_m3 \
--reasoning-parser minimax_m3 \
--enable-auto-tool-choice \
--trust-remote-code \
"${OFFLOAD_ARGS[@]}" > "$SERVER_LOG" 2>&1 &
SERVER_PID=$!

wait_for_server_ready --port "$VLLM_BACKEND_PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID"

if [[ "$DP_ATTENTION" == "true" ]]; then
vllm-router \
--worker-urls "http://localhost:$VLLM_BACKEND_PORT" \
--policy consistent_hash \
--intra-node-data-parallel-size "$TP" \
--host 0.0.0.0 \
--port "$PORT" \
--prometheus-host 127.0.0.1 \
--prometheus-port "$((PORT + 10000))" \
--request-timeout-secs 14400 \
--disable-retries > "$ROUTER_LOG" 2>&1 &
ROUTER_PID=$!
wait_for_server_ready --port "$PORT" --server-log "$ROUTER_LOG" --server-pid "$ROUTER_PID"
fi

if [ "${EVAL_ONLY}" = "true" ]; then
run_eval --port "$PORT"
else
build_replay_cmd "$RESULT_DIR"
run_agentic_replay_and_write_outputs "$RESULT_DIR"
fi
25 changes: 25 additions & 0 deletions configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -7003,6 +7003,31 @@ minimaxm3-fp8-h200-vllm-agentic:
- { tp: 8, ep: 8, kv-offloading: none, conc-list: [2, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 18, 20] }
- { tp: 8, ep: 8, kv-offloading: dram, kv-offload-backend: { name: mooncake, version: "0.3.11.post1" }, conc-list: [5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 18, 20] }

# EAGLE3 speculative-decoding (spec-decoding: mtp) variant of
# minimaxm3-fp8-h200-vllm-agentic, pairing MiniMaxAI/MiniMax-M3-MXFP8 with the
# Inferact/MiniMax-M3-EAGLE3 draft head (3 speculative tokens, FLASH_ATTN
# drafter) and pinning synthetic acceptance to the golden AL 2.83
# (golden_al_distribution/minimaxm3_eagle3.yaml, thinking_on, K=3). Same TP8
# arms as the non-MTP entry so the spec-decode delta is readable at equal
# concurrency, trimmed at the extreme-conc end because the draft head and its KV
# pull from the same 141 GB budget that places the GPU-resident cliff near
# conc 10.
minimaxm3-fp8-h200-vllm-agentic-mtp:
image: vllm/vllm-openai:nightly-04c2a8deac44fdb1ca3e2b5ec3e6bf16f3f6a914
model: MiniMaxAI/MiniMax-M3-MXFP8
model-prefix: minimaxm3
runner: cluster:h200-dgxc
precision: fp8
framework: vllm
multinode: false
scenarios:
agentic-coding:
- dram-utilization: 0.80
search-space:
- { tp: 8, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 4, 8, 12, 16] }
- { tp: 8, ep: 8, spec-decoding: mtp, kv-offloading: none, conc-list: [8, 12, 16] }
- { tp: 8, ep: 8, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: mooncake, version: "0.3.11.post1" }, conc-list: [12, 16] }

qwen3.5-fp4-b200-sglang-agentic-mtp:
image: lmsysorg/sglang:v0.5.16-cu130
model: nvidia/Qwen3.5-397B-A17B-NVFP4
Expand Down
13 changes: 13 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -5633,3 +5633,16 @@
- "Staging fix after run 30729467646, in which all five cells failed identically. /lustre/fsw/gharunners/models/GLM-5.2-NVFP4 already existed on b200-dgxc holding config.json, generation_config.json, hf_quant_config.json, chat_template.jinja, README.md and .quant_summary.txt and no weights or tokenizer at all -- an aborted or metadata-only pull. The `ls -A` emptiness guard accepted it, every cell skipped the download, SGLang read config.json fine and then died in AutoTokenizer.from_pretrained with \"Couldn't instantiate the backend tokenizer\". The B300 sibling ran 5/5 green on the same image and the same MTP + simulated-acceptance config in run 30729399405, so this was staging only, not the recipe or the image."
- "Replaces the emptiness guard with a completeness check -- tokenizer_config.json, a tokenizer.json or tokenizer.model, model.safetensors.index.json, and every shard the index names -- and serializes the ~433 GB download behind an flock on <MODEL_PATH>.download.lock so one cell stages the checkpoint and the other four wait instead of five allocations racing the same Lustre path. hf download resumes into a partially-populated --local-dir, so it downloads on top of the stub. The launcher probe now also requires a candidate to hold at least one weight shard rather than merely exist, so a stub in one staged tree cannot be preferred over a real copy in another."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2448

- config-keys:
- minimaxm3-fp8-h200-vllm-agentic-mtp
description:
- "Add the EAGLE3 speculative-decoding (spec-decoding: mtp) variant of minimaxm3-fp8-h200-vllm-agentic: MiniMax-M3 MXFP8 on H200 with vLLM, agentic-coding scenario only, routed via spec-decoding=mtp to benchmarks/single_node/agentic/minimaxm3_fp8_h200_mtp.sh."
- "Speculative config: method eagle3 on the Inferact/MiniMax-M3-EAGLE3 draft head, num_speculative_tokens 3, drafter pinned to attention_backend FLASH_ATTN -- the same draft, level and pin as the merged MiniMax-M3 CUDA MTP recipes (the EAGLE3 head is MHA and FlashInfer only serves the mandatory page size 128 through its GQA/MQA-only trtllm-gen kernel)."
- "Throughput runs pin rejection_sample_method synthetic with synthetic_acceptance_length 2.83 -- the committed golden AL for MiniMax-M3 EAGLE3 at K=3, thinking_on (golden_al_distribution/minimaxm3_eagle3.yaml), per the AgentX policy in golden_al_distribution/README.md. The minimaxm3_eagle3_gqa.yaml curve was measured against the Inferact/MiniMax-M3-EAGLE3-GQA draft (run 29784780049) and is not mixed in. EVAL_ONLY runs drop synthetic acceptance and keep real target verification."
- "Serve shape is otherwise identical to the non-MTP agentic sibling (Mooncake host-DRAM KV offload, --block-size 128, --language-model-only, --kv-cache-dtype fp8, --attention-backend TRITON_ATTN, gpu-memory-utilization 0.90, minimax_m3 parsers, vllm-router for DP-attention), with max-num-seqs 2*CONC for AgentX subagent fan-out and cudagraph capture raised to MAX_NUM_SEQS*(1+3) TOKENS for vLLM's token-denominated FULL_DECODE_ONLY ladder."
- "Search space now runs on the AgentX MTP concurrency grid: GPU-resident rows [1, 4, 8, 12, 16] and the Mooncake row [4, 8, 12, 16]. Every step is at least 2 concurrency apart and every arm stops hard at conc 16 -- single-step sampling cannot separate configurations by more than run-to-run noise on the agentic corpus, and past conc 16 these SKUs are into the post-HBM-cliff thrashing regime that the non-MTP sweeps already characterized, which is not worth one GPU job per point."
- "gpu-memory-utilization is 0.90 rather than the non-MTP sibling's 0.92, matching fixed_seq_len/minimaxm3_fp8_h200_mtp.sh (the proven MTP setting on this SKU). The H100 twin of this recipe died mid-warmup at the sibling's higher value with torch.OutOfMemoryError on all 8 ranks once the EAGLE3 head and its KV were resident (run 30515793863)."
- "Image is the agentic sibling's vllm/vllm-openai:nightly-04c2a8deac44fdb1ca3e2b5ec3e6bf16f3f6a914 (upstream 2026-06-23) rather than the fixed-seq-len MTP entry's vllm/vllm-openai:minimax-m3 bring-up image, keeping the Mooncake connector, FP8 KV and parser flags on the build the agentic arm is validated on."
- "Trims the four TP8/EP8 points that AIPerf rejected on run 31207046020: GPU-resident conc 1 and 4, and Mooncake conc 4 and 8. None was an engine failure -- each served its requests with a 0% error rate and then failed the AIPerf profile-coverage gate (ProfileMetricCoverageError, TTFT coverage 95.0-97.6% against the required 98.0%), because at low concurrency on the long-context agentic corpus too few sessions start inside the one-hour profiling window for TTFT to be sampled to the end. The boundary is monotone and arm-specific: pure TP8 passes at every concurrency including 1, while the EP8 arms only clear the gate from conc 8 (GPU-resident) and conc 12 (Mooncake) upward, which is also where expert parallelism starts doing useful work rather than paying all-to-all overhead on a nearly empty batch. GPU-resident EP8 becomes [8, 12, 16] and Mooncake EP8 becomes [12, 16]; the TP8 arm is unchanged."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2424
Loading