From 6407c532353a06c16af5d10c35a26628bccafe9d Mon Sep 17 00:00:00 2001 From: Giovanni Guasti Date: Mon, 3 Aug 2026 08:36:05 +0200 Subject: [PATCH 1/8] [AMD] [WIP] [AGENTX] Adding GLM5.2 MI355X Support - Agentx Signed-off-by: Giovanni Guasti --- .../agentic/glm5.2_fp4_mi355x_sglang.sh | 273 ++++++++++++++++++ configs/amd-master.yaml | 18 ++ perf-changelog.yaml | 7 + 3 files changed, 298 insertions(+) create mode 100644 benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang.sh diff --git a/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang.sh b/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang.sh new file mode 100644 index 0000000000..28a2f847ef --- /dev/null +++ b/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang.sh @@ -0,0 +1,273 @@ +#!/usr/bin/env bash +set -eo pipefail +set -x + +source "$(dirname "$0")/../../benchmark_lib.sh" + +check_env_vars MODEL TP CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION EP_SIZE DP_ATTENTION + +if [[ -n "$SLURM_JOB_ID" ]]; then + echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" +fi + +# ROCR/HIP visibility under slurm cgroups. +if [ -n "$ROCR_VISIBLE_DEVICES" ]; then + export HIP_VISIBLE_DEVICES="$ROCR_VISIBLE_DEVICES" +fi + + +if [[ -n "$MODEL_PATH" ]]; then + if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then + hf download "$MODEL" --local-dir "$MODEL_PATH" + fi +else + hf download "$MODEL" + export MODEL_PATH="$MODEL" +fi +rocm-smi || true +amd-smi || true + +PORT=8765 + +# A server killed on this node minutes earlier (previous job, crashed run) +# can still be draining its ~1.4 TB of HBM: KFD reclaim takes minutes, and +# booting into a half-drained node fails RCCL init with HIP 'unhandled cuda +# error' / 'invalid argument' (observed as the mooncake-c64 CI failure). +# Wait for the GPUs to come back before launching. +# Per-GPU threshold: idle nodes hold a small driver/firmware VRAM baseline +# (observed up to ~4%/GPU, node-dependent), while a draining or occupied +# GPU sits at 50-90%. Require every GPU <= 10%. +GPU_CLEAN=false +for i in $(seq 1 90); do + VRAM_MAX=$(rocm-smi --showmemuse 2>/dev/null | grep -oE "GPU Memory Allocated \(VRAM%\): [0-9]+" | awk '{if ($NF > m) m = $NF} END {print m+0}') + if [ "${VRAM_MAX:-0}" -le 10 ]; then echo "GPUs clean (vram%max=$VRAM_MAX after $((i*10))s)"; GPU_CLEAN=true; break; fi + echo "waiting for prior-job GPU memory reclaim: vram%max=$VRAM_MAX"; sleep 10 +done +[ "$GPU_CLEAN" = "true" ] || { echo "Error: GPUs still draining prior job's memory after 15min" >&2; exit 1; } + +resolve_trace_source +install_agentic_deps + +SERVER_LOG="$RESULT_DIR/server.log" +ROUTER_LOG="$RESULT_DIR/router.log" +mkdir -p "$RESULT_DIR" + +export PYTHONNOUSERSITE=1 +# Agentic warmup dispatches hundreds of large prompts at once; allow up to +# 15 minutes of TCP progress before AIPerf declares a connection dead. +export AIPERF_HTTP_TCP_USER_TIMEOUT=900000 +# AIPerf pins one pooled keep-alive connection per session (client-side +# keep-alive 300s) while uvicorn's default SGLANG_TIMEOUT_KEEP_ALIVE is 5s; +# inter-turn idle gaps can reuse a socket exactly as the server closes it. +# Outlast the client pool so the race cannot occur. +export SGLANG_TIMEOUT_KEEP_ALIVE=900 +# The DSA indexer's top-k v2 kernel (default since v0.5.14) is JIT-compiled +# from CUDA-only source (cooperative_groups.h) and cannot build for gfx950; +# v1 dispatches to the precompiled HIP op in sgl-kernel (upstream MI355X CI +# runs DSA models the same way). +export SGLANG_OPT_USE_TOPK_V2=false + +# HiCache L2 (host DRAM), optionally extended with Mooncake L3. +# KV_OFFLOADING=dram requires KV_OFFLOAD_BACKEND=hicache or mooncake. +# +# Per-arm L2 ratio (sizing rationale below) applies to both backends unless +# overridden via HICACHE_RATIO. TP arm (182.7 GB/rank device pool): the +# working set oversubscribes the device pool ~3x at conc 32, so the host +# tier is what carries the radix hits - ratio 1.5 (~2.9 TB pinned incl. +# sidecars) validates through the conc-24 long-context storm for the +# mooncake arm. The DP-attention arm (159.4 GB/rank) only runs at conc >= +# 32, where each DP rank's ~8 sessions nearly fit in its own device pool +# (~1.5-1.6M of 1.7M tokens at conc 64) and the host tier just absorbs +# overflow - ratio 1.5 boots but the host OOM killer takes the server +# mid-storm at conc 48, so it runs ratio 0.5 (~1.2 TB pinned, ~1.8 TB of +# load headroom) at negligible hit-rate cost. The hicache-only arm has no +# L3 to fall back on, so these ratios are unvalidated there - override with +# HICACHE_RATIO if the host OOMs or hit-rate is poor. +CACHE_ARGS=() +if agentic_kv_offload_enabled; then + if [ "$DP_ATTENTION" = "true" ]; then + HICACHE_RATIO="${HICACHE_RATIO:-0.5}" + else + HICACHE_RATIO="${HICACHE_RATIO:-1.5}" + fi + HICACHE_WRITE_POLICY="${HICACHE_WRITE_POLICY:-write_through}" + HICACHE_IO_BACKEND="${HICACHE_IO_BACKEND:-direct}" + HICACHE_MEM_LAYOUT="${HICACHE_MEM_LAYOUT:-page_first_direct}" + case "$KV_OFFLOAD_BACKEND" in + hicache) + echo "HiCache (GPU+host DRAM only): ratio=$HICACHE_RATIO, write_policy=$HICACHE_WRITE_POLICY, io_backend=$HICACHE_IO_BACKEND, mem_layout=$HICACHE_MEM_LAYOUT" + CACHE_ARGS=( + --enable-hierarchical-cache + --hicache-ratio "$HICACHE_RATIO" + --hicache-write-policy "$HICACHE_WRITE_POLICY" + --hicache-io-backend "$HICACHE_IO_BACKEND" + --hicache-mem-layout "$HICACHE_MEM_LAYOUT" + ) + ;; + mooncake) + L3_PER_RANK_GB="${L3_PER_RANK_GB:-40}" + python3 -c "from mooncake.store import MooncakeDistributedStore" >/dev/null + MOONCAKE_MASTER_PORT=$((PORT + 12000)) + MOONCAKE_MASTER_LOG="$RESULT_DIR/mooncake_master.log" + MOONCAKE_CONFIG_PATH="$RESULT_DIR/mooncake_config.json" + cat > "$MOONCAKE_CONFIG_PATH" < "$MOONCAKE_MASTER_LOG" 2>&1 & + MOONCAKE_MASTER_PID=$! + sleep 2 + kill -0 "$MOONCAKE_MASTER_PID" + echo "HiCache+Mooncake: ratio=$HICACHE_RATIO, l3_per_rank=${L3_PER_RANK_GB} GB, dram_budget=${TOTAL_CPU_DRAM_GB} GB" + CACHE_ARGS=( + --enable-hierarchical-cache + --hicache-ratio "$HICACHE_RATIO" + --hicache-size 0 + --hicache-write-policy "$HICACHE_WRITE_POLICY" + --hicache-io-backend "$HICACHE_IO_BACKEND" + --hicache-mem-layout "$HICACHE_MEM_LAYOUT" + --hicache-storage-backend mooncake + --hicache-storage-prefetch-policy wait_complete + ) + ;; + *) + echo "Error: unsupported KV_OFFLOAD_BACKEND '$KV_OFFLOAD_BACKEND' (expected: hicache or mooncake)" >&2 + exit 1 + ;; + esac +fi + +# Arm selection. TP arm keeps the FP8 sibling's cookbook batch-shaping +# bands. +# +# NOTE: the DP-attention path below is currently DORMANT (no dp-attn arms +# in amd-master.yaml): DSA + dp-attention hangs a collective under +# long-context prefill on ROCm v0.5.14 (watchdog kills the scheduler with +# zero completions; reproduced with and without HiCache, with and without +# the DSv4 DP collective envs; short prompts are fine). Re-enable the +# config arm once upstream fixes the DSA DP prefill path. +# +# When active, the DP-attention (DEP) arm fronts the DP ranks with sglang-router +# using consistent hashing on the AIPerf correlation id so multi-turn +# sessions stay on the DP rank holding their radix/hicache prefix, and +# widens chunked-prefill (whole-engine, /dp ranks) like the B300 sibling. +USE_SGLANG_ROUTER=false +SGLANG_BACKEND_PORT="$PORT" +PARALLEL_ARGS=(--tp "$TP" --ep-size "$EP_SIZE") +MEM_FRACTION_STATIC=0.80 +if [ "$DP_ATTENTION" = "true" ]; then + USE_SGLANG_ROUTER=true + export AIPERF_HTTP_X_SMG_ROUTING_KEY_FROM_CORRELATION_ID=true + SGLANG_BACKEND_PORT=$((PORT + 1)) + SGLANG_ROUTER_METRICS_PORT=$((PORT + 10000)) + SGLANG_ROUTER_CMD=(python3 -m sglang_router.launch_router) + PARALLEL_ARGS+=(--dp "$TP" --enable-dp-attention) + CHUNKED_PREFILL_SIZE=32768 + export AGENTIC_WARMUP_GRACE_PERIOD=3600 + # Swap the DP gather collectives to gatherv/reduce-scatter on ROCm + # (dsv4_fp4_mi355x_sglang.sh precedent - the only green DP-attention + # config on this cluster/image): with the defaults the DSA DP path + # hangs a collective under long-context prefill load until the + # watchdog kills the scheduler (0/96 storm completions, twice). + export SGLANG_DP_USE_GATHERV=1 + export SGLANG_DP_USE_REDUCE_SCATTER=1 + export GPU_MAX_HW_QUEUES=5 +elif [ "$CONC" -le 16 ]; then + # A full 131072-token prefill chunk needs ~7 GiB/rank of activation + # headroom on top of the static pool; pair it with mem-fraction 0.80 + # like the FP8 sibling's low-conc band (0.85 OOMs the device mid-replay: + # "Tried to allocate 6.86 GiB ... 5.15 GiB is free", run 29751563205). + CHUNKED_PREFILL_SIZE=131072 + MEM_FRACTION_STATIC=0.80 +else + CHUNKED_PREFILL_SIZE=32768 + export AGENTIC_WARMUP_GRACE_PERIOD=3600 +fi +MAX_RUNNING_REQUESTS=$((1 * CONC)) +[ "$MAX_RUNNING_REQUESTS" -gt 256 ] && MAX_RUNNING_REQUESTS=256 +CUDA_GRAPH_MAX_BS=$MAX_RUNNING_REQUESTS + +SGLANG_CMD=( + python3 -m sglang.launch_server + --model-path "$MODEL_PATH" + --served-model-name "$MODEL" + --host 0.0.0.0 + --port "$SGLANG_BACKEND_PORT" + --trust-remote-code + "${PARALLEL_ARGS[@]}" + --kv-cache-dtype fp8_e4m3 + --dsa-prefill-backend tilelang + --dsa-decode-backend tilelang + # GLM-5.2 emits the GLM-4.7-style tool-call format; glm47 is required for + # structured message.tool_calls (SWE-bench agentic evals die without it). + # The glm45 reasoning parser keeps hybrid thinking in reasoning_content. + --tool-call-parser glm47 + --reasoning-parser glm45 + --chunked-prefill-size "$CHUNKED_PREFILL_SIZE" + --mem-fraction-static "$MEM_FRACTION_STATIC" + --max-running-requests "$MAX_RUNNING_REQUESTS" + --cuda-graph-max-bs "$CUDA_GRAPH_MAX_BS" + "${CACHE_ARGS[@]}" + --watchdog-timeout 1800 + --enable-metrics +) + +printf '%q ' "${SGLANG_CMD[@]}" | tee "$RESULT_DIR/sglang_command.txt" +printf '\n' | tee -a "$RESULT_DIR/sglang_command.txt" + +echo "Starting SGLang server for MI355X..." +"${SGLANG_CMD[@]}" > "$SERVER_LOG" 2>&1 & +SERVER_PID=$! +echo "Server PID: $SERVER_PID" + +wait_for_server_ready --port "$SGLANG_BACKEND_PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" + +if [ "$USE_SGLANG_ROUTER" = "true" ]; then + echo "Starting SGLang router on port $PORT for $TP DP ranks..." + "${SGLANG_ROUTER_CMD[@]}" \ + --worker-urls "http://localhost:$SGLANG_BACKEND_PORT" \ + --policy consistent_hashing \ + --request-id-headers x-correlation-id \ + --dp-aware \ + --host 0.0.0.0 \ + --port "$PORT" \ + --prometheus-host 127.0.0.1 \ + --prometheus-port "$SGLANG_ROUTER_METRICS_PORT" \ + --connect-timeout-secs 900 \ + --request-timeout-secs 14400 \ + --disable-health-check \ + --disable-retries > "$ROUTER_LOG" 2>&1 & + ROUTER_PID=$! + echo "Router PID: $ROUTER_PID" + wait_for_server_ready --port "$PORT" --server-log "$ROUTER_LOG" --server-pid "$ROUTER_PID" +fi + +if [ "${EVAL_ONLY}" = "true" ]; then + # GLM-5.2's chat template defaults to reasoning_effort=Max when the + # client passes no chat_template_kwargs (mini-swe-agent doesn't), and the + # heavy thinking burns the default 75-step budget before submission. + # Double the step budget for this recipe; others keep the shared default. + export SWEBENCH_AGENT_STEP_LIMIT=150 + # Pin eval agent parallelism to the proven-green level: workers default + # to CONC, and at 64 concurrent Modal sandboxes the cluster's egress + # collapses (18k "Cannot connect to *.modal.host" errors crippled the + # trajectories in run 29764760177) while 32 ran clean. The serving + # config is unchanged - only the agent's session fan-out is capped. + export SWEBENCH_AGENT_WORKERS="${SWEBENCH_AGENT_WORKERS:-32}" + run_eval --port "$PORT" +else + build_replay_cmd "$RESULT_DIR" + REPLAY_CMD+=" --server-metrics http://localhost:$SGLANG_BACKEND_PORT/metrics" + run_agentic_replay_and_write_outputs "$RESULT_DIR" +fi \ No newline at end of file diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index b8f6f4fc31..eae7f822dc 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -2360,3 +2360,21 @@ dsv4-fp8-mi325x-vllm-mtp: # is 55.8%. conc128 passed cleanly on the memory-tightest SKU (MI300X); # cap 8k1k MTP at 128 (normal holds 256, 1k1k holds 512). - { tp: 8, conc-start: 4, conc-end: 128, spec-decoding: mtp } + + +# GLM-5.2 MXFP4 single-node on MI355X via SGLang with HiCache KV offloading. +# Agentic-coding scenario with DRAM KV offload (hicache backend); TP8/EP8, +# concurrency sweep [1,4,8,12,16] to profile throughput at low-to-mid load. +glm5.2-fp4-mi355x-sglang-agentic-hicache: + image: lmsysorg/sglang-rocm:v0.5.16-rocm720-mi35x-20260728 + model: amd/GLM-5.2-MXFP4 + model-prefix: glm5.2 + runner: cluster:mi355x-amds + precision: fp4 + framework: sglang + multinode: false + scenarios: + agentic-coding: + - dram-utilization: 0.85 + search-space: + - { tp: 8, ep: 8, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [1, 4, 8, 12, 16] } \ No newline at end of file diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 6eebd97070..8e228b41de 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5355,3 +5355,10 @@ - "Apply the accuracy-gated Kimi-K2.5 MXFP4 settings: tuned AITER MXFP4 MoE, fused shared experts, FP8 KV cache, block size 16, 16384 batched tokens, 512 sequences, async scheduling, gpu-memory-utilization 0.85 (headroom for CUDA-graph capture on MI355X), and the AITER BF16 GEMM path" - "Extend the TP4 and TP8 8k1k concurrency sweep from 64 to 128 (1k1k deprecated per #2263)" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2213 + +- config-keys: + - glm5.2-fp4-mi355x-sglang-agentic-hicache + description: + - "Add GLM-5.2 FP4 SGLANG Single Node Agentic Support" + - "Image: lmsysorg/sglang-rocm:v0.5.16-rocm720-mi35x-20260728" + pr-link: TO-BE-ADDED \ No newline at end of file From 9b5be890ce7cd6ef7b30ef3c23a96cec135c12d5 Mon Sep 17 00:00:00 2001 From: Giovanni Guasti Date: Mon, 3 Aug 2026 09:02:05 +0200 Subject: [PATCH 2/8] [AMD] [AGENTX] Update perf-changelog.yaml with PR link Signed-off-by: Giovanni Guasti --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 8e228b41de..78fb822a85 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5361,4 +5361,4 @@ description: - "Add GLM-5.2 FP4 SGLANG Single Node Agentic Support" - "Image: lmsysorg/sglang-rocm:v0.5.16-rocm720-mi35x-20260728" - pr-link: TO-BE-ADDED \ No newline at end of file + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2463 \ No newline at end of file From 84c0609ef3f69b467794efab34862cbef5b8304a Mon Sep 17 00:00:00 2001 From: ajith-sirra-amd <122240613+ajith-sirra-amd@users.noreply.github.com> Date: Mon, 3 Aug 2026 13:49:49 +0530 Subject: [PATCH 3/8] [AMD] [WIP] [AGENTX] GML 5.2 - Add speculative algorithm options to script --- ...fp4_mi355x_sglang.sh => glm5.2_fp4_mi355x_sglang_mtp.sh} | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) rename benchmarks/single_node/agentic/{glm5.2_fp4_mi355x_sglang.sh => glm5.2_fp4_mi355x_sglang_mtp.sh} (98%) diff --git a/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang.sh b/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang_mtp.sh similarity index 98% rename from benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang.sh rename to benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang_mtp.sh index 28a2f847ef..892057a13c 100644 --- a/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang.sh +++ b/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang_mtp.sh @@ -218,6 +218,10 @@ SGLANG_CMD=( --mem-fraction-static "$MEM_FRACTION_STATIC" --max-running-requests "$MAX_RUNNING_REQUESTS" --cuda-graph-max-bs "$CUDA_GRAPH_MAX_BS" + --speculative-algorithm EAGLE + --speculative-num-steps 5 + --speculative-eagle-topk 1 + --speculative-num-draft-tokens 6 "${CACHE_ARGS[@]}" --watchdog-timeout 1800 --enable-metrics @@ -270,4 +274,4 @@ else build_replay_cmd "$RESULT_DIR" REPLAY_CMD+=" --server-metrics http://localhost:$SGLANG_BACKEND_PORT/metrics" run_agentic_replay_and_write_outputs "$RESULT_DIR" -fi \ No newline at end of file +fi From 818b7aaf9f60373512dc67fbff21f2e0fff22b85 Mon Sep 17 00:00:00 2001 From: ajith-sirra-amd <122240613+ajith-sirra-amd@users.noreply.github.com> Date: Mon, 3 Aug 2026 13:51:29 +0530 Subject: [PATCH 4/8] [AMD] [WIP] [AGENTX] GLM 5.2 - Rename GLM-5.2 configuration for clarity --- configs/amd-master.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index eae7f822dc..2f37ffa5d7 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -2365,7 +2365,7 @@ dsv4-fp8-mi325x-vllm-mtp: # GLM-5.2 MXFP4 single-node on MI355X via SGLang with HiCache KV offloading. # Agentic-coding scenario with DRAM KV offload (hicache backend); TP8/EP8, # concurrency sweep [1,4,8,12,16] to profile throughput at low-to-mid load. -glm5.2-fp4-mi355x-sglang-agentic-hicache: +glm5.2-fp4-mi355x-sglang-agentic-hicache-mtp: image: lmsysorg/sglang-rocm:v0.5.16-rocm720-mi35x-20260728 model: amd/GLM-5.2-MXFP4 model-prefix: glm5.2 @@ -2377,4 +2377,4 @@ glm5.2-fp4-mi355x-sglang-agentic-hicache: agentic-coding: - dram-utilization: 0.85 search-space: - - { tp: 8, ep: 8, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [1, 4, 8, 12, 16] } \ No newline at end of file + - { tp: 8, ep: 8, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [1, 4, 8, 12, 16] } From 8ff2f3c30303e7afc96faa5ab8dcc299f2e3bf60 Mon Sep 17 00:00:00 2001 From: ajith-sirra-amd <122240613+ajith-sirra-amd@users.noreply.github.com> Date: Mon, 3 Aug 2026 13:52:51 +0530 Subject: [PATCH 5/8] [AMD] [WIP] [AGENTX] GLM 5.2 - Update GLM-5.2 FP4 SGLANG support details --- perf-changelog.yaml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 78fb822a85..147598e1ee 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5357,8 +5357,8 @@ pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2213 - config-keys: - - glm5.2-fp4-mi355x-sglang-agentic-hicache + - glm5.2-fp4-mi355x-sglang-agentic-hicache-mtp description: - - "Add GLM-5.2 FP4 SGLANG Single Node Agentic Support" + - "Add GLM-5.2 FP4 SGLANG Single Node MTP Agentic Support" - "Image: lmsysorg/sglang-rocm:v0.5.16-rocm720-mi35x-20260728" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2463 \ No newline at end of file + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2463 From 0e31f0934d406455645175280812f17703b4df22 Mon Sep 17 00:00:00 2001 From: ajith-sirra-amd <122240613+ajith-sirra-amd@users.noreply.github.com> Date: Mon, 3 Aug 2026 15:41:42 +0530 Subject: [PATCH 6/8] [AMD] [WIP] [AGENTX] GLM 5.2 - Update memory fraction static value to 0.85 Increased memory fraction from 0.80 to 0.85 in multiple locations. --- .../single_node/agentic/glm5.2_fp4_mi355x_sglang_mtp.sh | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang_mtp.sh b/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang_mtp.sh index 892057a13c..d9b3f8998f 100644 --- a/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang_mtp.sh @@ -26,9 +26,7 @@ else fi rocm-smi || true amd-smi || true - -PORT=8765 - + # A server killed on this node minutes earlier (previous job, crashed run) # can still be draining its ~1.4 TB of HBM: KFD reclaim takes minutes, and # booting into a half-drained node fails RCCL init with HIP 'unhandled cuda @@ -165,7 +163,7 @@ fi USE_SGLANG_ROUTER=false SGLANG_BACKEND_PORT="$PORT" PARALLEL_ARGS=(--tp "$TP" --ep-size "$EP_SIZE") -MEM_FRACTION_STATIC=0.80 +MEM_FRACTION_STATIC=0.85 if [ "$DP_ATTENTION" = "true" ]; then USE_SGLANG_ROUTER=true export AIPERF_HTTP_X_SMG_ROUTING_KEY_FROM_CORRELATION_ID=true @@ -189,7 +187,7 @@ elif [ "$CONC" -le 16 ]; then # like the FP8 sibling's low-conc band (0.85 OOMs the device mid-replay: # "Tried to allocate 6.86 GiB ... 5.15 GiB is free", run 29751563205). CHUNKED_PREFILL_SIZE=131072 - MEM_FRACTION_STATIC=0.80 + MEM_FRACTION_STATIC=0.85 else CHUNKED_PREFILL_SIZE=32768 export AGENTIC_WARMUP_GRACE_PERIOD=3600 From dad30997043e7823f13381c030ebfad98081d91b Mon Sep 17 00:00:00 2001 From: ajith-sirra-amd <122240613+ajith-sirra-amd@users.noreply.github.com> Date: Mon, 3 Aug 2026 18:42:29 +0530 Subject: [PATCH 7/8] [AMD] [WIP] [AGENTX] GLM 5.2 - Rename glm5.2_fp4_mi355x_sglang_mtp.sh to glm5.2_fp4_mi355x_sglang.sh --- ...lm5.2_fp4_mi355x_sglang_mtp.sh => glm5.2_fp4_mi355x_sglang.sh} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename benchmarks/single_node/agentic/{glm5.2_fp4_mi355x_sglang_mtp.sh => glm5.2_fp4_mi355x_sglang.sh} (100%) diff --git a/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang_mtp.sh b/benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang.sh similarity index 100% rename from benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang_mtp.sh rename to benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang.sh From 1bd5703f0cb5c2de18aa5c73789cd6d439dacb7a Mon Sep 17 00:00:00 2001 From: ajith-sirra-amd <122240613+ajith-sirra-amd@users.noreply.github.com> Date: Tue, 4 Aug 2026 08:31:42 +0530 Subject: [PATCH 8/8] [AMD] [WIP] [AGENTX] Update perf-changelog.yaml with new entries --- perf-changelog.yaml | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index cf0221f6a7..4a137a84e2 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5417,3 +5417,11 @@ - "Bump vLLM nightly image to nightly-5e35a6f4f9bbc217c599692157ca985c894373f7 (fixed-len MiniMax-M3 fix)" - "Enable VLLM_MINIMAX_M3_MSA_DECODE_BACKEND=cutlass" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2468 + +- config-keys: + - glm5.2-fp4-mi355x-sglang-agentic-hicache-mtp + description: + - "Add GLM-5.2 FP4 SGLANG Single Node MTP Agentic Support" + - "Image: lmsysorg/sglang-rocm:v0.5.16-rocm720-mi35x-20260728" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2463 +