diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml index 488994e3bf..b97ad8cea5 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8-mtp/8k1k.yaml @@ -3,7 +3,7 @@ base: name: qwen3.5-fp8-mi355x-sglang-mtp-8k1k model: path: hf:Qwen/Qwen3.5-397B-A17B-FP8 - container: lmsysorg/sglang-rocm:v0.5.18-rocm720-mi35x-20260828 + container: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260924 precision: fp8 resources: gpu_type: mi355x @@ -29,7 +29,7 @@ base: tokenizer-worker-num: 6 enable-aiter-allreduce-fusion: true max-running-requests: 4 - cuda-graph-max-bs: 4 + cuda-graph-max-bs-decode: 4 disable-radix-cache: true chunked-prefill-size: 32768 scheduler-recv-interval: 30 @@ -60,7 +60,7 @@ zip_override_concurrency: roles: agg: args: - cuda-graph-max-bs: [4, 8, 16, 32, 64, 128, 256] + cuda-graph-max-bs-decode: [4, 8, 16, 32, 64, 128, 256] max-running-requests: [4, 8, 16, 32, 64, 128, 256] benchmark: env: diff --git a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml index c753c2b465..ddbe1db8c4 100644 --- a/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml +++ b/benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml @@ -3,7 +3,7 @@ base: name: qwen3.5-fp8-mi355x-sglang-8k1k model: path: hf:Qwen/Qwen3.5-397B-A17B-FP8 - container: lmsysorg/sglang-rocm:v0.5.18-rocm720-mi35x-20260828 + container: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260924 precision: fp8 resources: gpu_type: mi355x @@ -29,7 +29,7 @@ base: tokenizer-worker-num: 6 enable-aiter-allreduce-fusion: true max-running-requests: 4 - cuda-graph-max-bs: 4 + cuda-graph-max-bs-decode: 4 disable-radix-cache: true chunked-prefill-size: 32768 scheduler-recv-interval: 30 @@ -56,7 +56,7 @@ zip_override_concurrency: roles: agg: args: - cuda-graph-max-bs: [4, 8, 16, 32, 64, 128, 256] + cuda-graph-max-bs-decode: [4, 8, 16, 32, 64, 128, 256] max-running-requests: [4, 8, 16, 32, 64, 128, 256] benchmark: env: diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index 887822d322..74d5de7a34 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -182,7 +182,7 @@ qwen3.5-fp8-mi325x-sglang: srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi325x-fp8/8k1k.yaml qwen3.5-fp8-mi355x-sglang: - image: lmsysorg/sglang-rocm:v0.5.18-rocm720-mi35x-20260828 + image: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260924 model: Qwen/Qwen3.5-397B-A17B-FP8 model-prefix: qwen3.5 runner: mi355x @@ -201,7 +201,7 @@ qwen3.5-fp8-mi355x-sglang: srt-recipe: benchmarks/single_node/srt-slurm-recipes/qwen3.5/sglang/mi355x-fp8/8k1k.yaml qwen3.5-fp8-mi355x-sglang-mtp: - image: lmsysorg/sglang-rocm:v0.5.18-rocm720-mi35x-20260828 + image: lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260924 model: Qwen/Qwen3.5-397B-A17B-FP8 model-prefix: qwen3.5 runner: mi355x diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 95afb7e976..6da0e33ab5 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8902,3 +8902,10 @@ - "server_atom.sh exports PORT before sourcing benchmark_lib.sh on the eval path. run_lm_eval's check_env_vars guard runs before it parses --port and job.slurm's docker -e allowlist forwards ROUTER_PORT but never PORT, so every multi-node lm-eval cell aborted 8 s in; because check_env_vars exits rather than returning, node 0 also skipped its router teardown and the decode node hung in 'Waiting until router closes...' until the job was cancelled (run 35643358891)." - "Per ATOM DeepSeek-V4-Agentic-PD-Max recipe: prefix caching on, FP8 KV and index cache, block-size 256, TBO on prefill only, max-num-seqs = 2x concurrency." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3158 + +- config-keys: + - qwen3.5-fp8-mi355x-sglang + - qwen3.5-fp8-mi355x-sglang-mtp + description: + - "Update SGLang ROCm image from v0.5.18-rocm720-mi35x-20260828 to v0.5.20-rocm720-mi35x-20260924 (latest nightly)" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3423