From 193aff18d2575dbee266a16a6a2a65be27bb6f2d Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Fri, 18 Sep 2026 03:05:13 -0400 Subject: [PATCH 1/5] [Klaud Cold] Update qwen3.5-fp8-gb300-dynamo-sglang-mtp SGLang image to nightly-dev-cu13-20260918-20518d85 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bump the Qwen3.5 FP8 GB300 Dynamo-SGLang 8k1k MTP recipe from lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 to lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85. 将 qwen3.5-fp8-gb300-dynamo-sglang-mtp 的 SGLang 镜像更新至 nightly-dev-cu13-20260918-20518d85。 Co-Authored-By: Claude Fable 5.1 --- .../8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml | 8 ++++---- ...agg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml | 8 ++++---- ...1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml | 6 +++--- ...1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml | 6 +++--- ...d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml | 6 +++--- ...d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml | 6 +++--- ...p4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml | 6 +++--- configs/nvidia-master.yaml | 4 ++-- perf-changelog.yaml | 9 +++++++++ 9 files changed, 34 insertions(+), 25 deletions(-) diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml index 9050a22244..12dd3759c3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml @@ -10,7 +10,7 @@ dynamo: install: true source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + wheel: "1.5.0.dev20260917" frontend: type: dynamo enable_multiple_frontends: true @@ -19,7 +19,7 @@ frontend: model: path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" + container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" precision: "fp8" resources: @@ -70,7 +70,7 @@ roles: chunked-prefill-size: 16384 max-prefill-tokens: 16384 context-length: 16384 - cuda-graph-max-bs: 1024 + cuda-graph-max-bs-decode: 1024 decode-log-interval: 1 stream-interval: 50 disaggregation-mode: "prefill" @@ -130,7 +130,7 @@ roles: chunked-prefill-size: 16384 max-prefill-tokens: 16384 context-length: 16384 - cuda-graph-max-bs: 1024 + cuda-graph-max-bs-decode: 1024 decode-log-interval: 1 stream-interval: 50 disaggregation-mode: "decode" diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml index 8ce688a8ba..0f8daea78a 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml @@ -10,7 +10,7 @@ dynamo: install: true source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + wheel: "1.5.0.dev20260917" frontend: type: dynamo enable_multiple_frontends: true @@ -19,7 +19,7 @@ frontend: model: path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" + container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" precision: "fp8" resources: @@ -74,7 +74,7 @@ roles: chunked-prefill-size: 16384 max-prefill-tokens: 16384 context-length: 9236 - cuda-graph-max-bs: 320 + cuda-graph-max-bs-decode: 320 scheduler-recv-interval: 10 decode-log-interval: 50 stream-interval: 50 @@ -140,7 +140,7 @@ roles: chunked-prefill-size: 4096 max-prefill-tokens: 16384 context-length: 9236 - cuda-graph-max-bs: 320 + cuda-graph-max-bs-decode: 320 scheduler-recv-interval: 10 decode-log-interval: 50 stream-interval: 50 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml index e4588b3dda..803e3816ae 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml @@ -10,7 +10,7 @@ dynamo: install: true source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + wheel: "1.5.0.dev20260917" frontend: type: dynamo enable_multiple_frontends: true @@ -19,7 +19,7 @@ frontend: model: path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" + container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" precision: "fp8" resources: @@ -149,7 +149,7 @@ roles: mem-fraction-static: 0.7 max-mamba-cache-size: 1024 max-running-requests: 1024 - cuda-graph-max-bs: 128 + cuda-graph-max-bs-decode: 128 watchdog-timeout: 1000000 page-size: 64 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml index e0cc1ab0a3..ad4c0b035e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml @@ -10,7 +10,7 @@ dynamo: install: true source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + wheel: "1.5.0.dev20260917" frontend: type: dynamo enable_multiple_frontends: true @@ -19,7 +19,7 @@ frontend: model: path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" + container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" precision: "fp8" resources: @@ -149,7 +149,7 @@ roles: mem-fraction-static: 0.7 max-mamba-cache-size: 1024 max-running-requests: 1024 - cuda-graph-max-bs: 128 + cuda-graph-max-bs-decode: 128 watchdog-timeout: 1000000 page-size: 64 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml index 510e6146a5..9ac9917e00 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml @@ -10,7 +10,7 @@ dynamo: install: true source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + wheel: "1.5.0.dev20260917" frontend: type: dynamo enable_multiple_frontends: true @@ -19,7 +19,7 @@ frontend: model: path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" + container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" precision: "fp8" resources: @@ -149,7 +149,7 @@ roles: mem-fraction-static: 0.7 max-mamba-cache-size: 1024 max-running-requests: 1024 - cuda-graph-max-bs: 128 + cuda-graph-max-bs-decode: 128 watchdog-timeout: 1000000 page-size: 64 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml index 0759135190..45428a9d30 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml @@ -10,7 +10,7 @@ dynamo: install: true source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + wheel: "1.5.0.dev20260917" frontend: type: dynamo enable_multiple_frontends: true @@ -19,7 +19,7 @@ frontend: model: path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" + container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" precision: "fp8" resources: @@ -149,7 +149,7 @@ roles: mem-fraction-static: 0.7 max-mamba-cache-size: 2048 max-running-requests: 2048 - cuda-graph-max-bs: 128 + cuda-graph-max-bs-decode: 128 watchdog-timeout: 1000000 page-size: 64 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml index 505b8cef0e..f85092e9d7 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml @@ -10,7 +10,7 @@ dynamo: install: true source: - rev: 46520ca59afe992fb5ef61b3197b2316f8df9b2b + wheel: "1.5.0.dev20260917" frontend: type: dynamo enable_multiple_frontends: true @@ -19,7 +19,7 @@ frontend: model: path: "qwen3.5-fp8" - container: "lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3" + container: "lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85" precision: "fp8" resources: @@ -149,7 +149,7 @@ roles: mem-fraction-static: 0.7 max-mamba-cache-size: 2048 max-running-requests: 2048 - cuda-graph-max-bs: 128 + cuda-graph-max-bs-decode: 128 watchdog-timeout: 1000000 page-size: 64 diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index f28e74de34..84334115df 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -7821,13 +7821,13 @@ kimik3-fp4-b200-dynamo-vllm-agentic-dspark: additional-settings: - "CONFIG_FILE=recipes/kimik3/vllm/b200-fp4/agentx/agg-tp8pp2-mooncake-c96.yaml" qwen3.5-fp8-gb300-dynamo-sglang-mtp: - image: lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 + image: lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 model: Qwen/Qwen3.5-397B-A17B-FP8 model-prefix: qwen3.5 runner: gb300 precision: fp8 framework: dynamo-sglang - router: { name: dynamo-router, version: "46520ca59afe992fb5ef61b3197b2316f8df9b2b" } + router: { name: dynamo-router, version: "1.5.0.dev20260917" } kv-p2p-transfer: mooncake multinode: true disagg: true diff --git a/perf-changelog.yaml b/perf-changelog.yaml index a112ba56db..82cf5df99f 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8082,3 +8082,12 @@ - "Capture only full decode CUDA graphs (VLLM_USE_BREAKABLE_CUDAGRAPH=0 with cudagraph_mode FULL_DECODE_ONLY, as the MiniMax-M3 gfx942 arm does) and restore --moe-backend aiter: on gfx942 every worker segfaulted during piecewise graph capture with both the Triton W4A16 MoE kernel (run 35305045778) and the auto-selected unfused Triton kernel (run 35306398350), so the capture mode rather than the MoE kernel is the failing piece; prefill runs eagerly" - "仅捕获完整的 decode CUDA graph(VLLM_USE_BREAKABLE_CUDAGRAPH=0 并设置 cudagraph_mode FULL_DECODE_ONLY,与 MiniMax-M3 gfx942 配方一致)并恢复 --moe-backend aiter:在 gfx942 上,无论使用 Triton W4A16 MoE 内核(运行 35305045778)还是自动选择的未融合 Triton 内核(运行 35306398350),所有 worker 都在 piecewise graph 捕获期间段错误,说明问题在于捕获模式而非 MoE 内核;prefill 以 eager 方式运行" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3247 + +- config-keys: + - qwen3.5-fp8-gb300-dynamo-sglang-mtp + description: + - "Update the Qwen3.5 FP8 GB300 Dynamo-SGLang 8k1k MTP SGLang image from lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 to lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85 (2026-09-18 upstream cu13 dev nightly, sgl-project/sglang@20518d85, amd64 + arm64)" + - "In the 7 srt-slurm recipes of this key, set model.container to the same image, rename cuda-graph-max-bs to cuda-graph-max-bs-decode for the same reason, and move the Dynamo install from source commit 46520ca59afe992fb5ef61b3197b2316f8df9b2b to the public 1.5.0.dev20260917 wheel (the master router version follows), the newest Dynamo nightly, since the June source build predates the SGLang ServerArgs compatibility fixes that Dynamo 1.5.0.dev20260910 and later carry; recipe topologies, concurrencies and serving flags are otherwise unchanged" + - "将 Qwen3.5 FP8 GB300 Dynamo-SGLang 8k1k MTP 的 SGLang 镜像从 lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 更新为 lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85(2026-09-18 上游 cu13 dev nightly,sgl-project/sglang@20518d85,amd64 + arm64)" + - "在该 key 的 7 个 srt-slurm 配方中,将 model.container 设为同一镜像,出于同样原因将 cuda-graph-max-bs 改为 cuda-graph-max-bs-decode,并将 Dynamo 安装从源码提交 46520ca59afe992fb5ef61b3197b2316f8df9b2b 改为公开的 1.5.0.dev20260917 wheel(master 中的 router 版本随之更新),这是最新的 Dynamo nightly,因为 6 月的源码构建早于 Dynamo 1.5.0.dev20260910 及之后版本所包含的 SGLang ServerArgs 兼容性修复;配方拓扑、并发与服务参数保持不变" + pr-link: TBD From 55ed302ba5f745632a0fc71e25279603ae3911a8 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Fri, 18 Sep 2026 03:05:18 -0400 Subject: [PATCH 2/5] perf-changelog: fill pr-link for qwen3.5-fp8-gb300-dynamo-sglang-mtp MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 为 qwen3.5-fp8-gb300-dynamo-sglang-mtp 填写 pr-link。 Co-Authored-By: Claude Fable 5.1 --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 82cf5df99f..3253adf9db 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8090,4 +8090,4 @@ - "In the 7 srt-slurm recipes of this key, set model.container to the same image, rename cuda-graph-max-bs to cuda-graph-max-bs-decode for the same reason, and move the Dynamo install from source commit 46520ca59afe992fb5ef61b3197b2316f8df9b2b to the public 1.5.0.dev20260917 wheel (the master router version follows), the newest Dynamo nightly, since the June source build predates the SGLang ServerArgs compatibility fixes that Dynamo 1.5.0.dev20260910 and later carry; recipe topologies, concurrencies and serving flags are otherwise unchanged" - "将 Qwen3.5 FP8 GB300 Dynamo-SGLang 8k1k MTP 的 SGLang 镜像从 lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 更新为 lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85(2026-09-18 上游 cu13 dev nightly,sgl-project/sglang@20518d85,amd64 + arm64)" - "在该 key 的 7 个 srt-slurm 配方中,将 model.container 设为同一镜像,出于同样原因将 cuda-graph-max-bs 改为 cuda-graph-max-bs-decode,并将 Dynamo 安装从源码提交 46520ca59afe992fb5ef61b3197b2316f8df9b2b 改为公开的 1.5.0.dev20260917 wheel(master 中的 router 版本随之更新),这是最新的 Dynamo nightly,因为 6 月的源码构建早于 Dynamo 1.5.0.dev20260910 及之后版本所包含的 SGLang ServerArgs 兼容性修复;配方拓扑、并发与服务参数保持不变" - pr-link: TBD + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3255 From 05d8443408aa9da0f3d9ad4bf6023cdff314fbe0 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Fri, 18 Sep 2026 03:22:22 -0400 Subject: [PATCH 3/5] qwen3.5-fp8-gb300-dynamo-sglang-mtp: rename mamba-scheduler-strategy to mamba-radix-cache-strategy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 2026-09-18 SGLang nightly rejects the old flag (run 35317753866). 将 mamba-scheduler-strategy 改名为 mamba-radix-cache-strategy。 Co-Authored-By: Claude Fable 5.1 --- .../8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml | 4 ++-- ...isagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml | 4 ++-- ...3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml | 4 ++-- ...4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml | 4 ++-- ...p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml | 4 ++-- ...p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml | 4 ++-- ...-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml | 4 ++-- perf-changelog.yaml | 7 +++++++ 8 files changed, 21 insertions(+), 14 deletions(-) diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml index 12dd3759c3..a6167a385b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp4-d-tp4-b1024-c1x2x8-mtp.yaml @@ -60,7 +60,7 @@ roles: trust-remote-code: true attention-backend: "trtllm_mha" tensor-parallel-size: 4 - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 2048 mamba-ssm-dtype: "bfloat16" moe-runner-backend: "flashinfer_trtllm" @@ -113,7 +113,7 @@ roles: trust-remote-code: true attention-backend: "trtllm_mha" tensor-parallel-size: 4 - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 128 mamba-ssm-dtype: "bfloat16" max-mamba-cache-size: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml index 0f8daea78a..6eb926dde9 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-1p1d-p-tp8-ep8-d-tp8-ep8-b1024-c32x48x80-mtp.yaml @@ -63,7 +63,7 @@ roles: tensor-parallel-size: 8 data-parallel-size: 1 expert-parallel-size: 8 - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 2048 mamba-ssm-dtype: "bfloat16" moe-runner-backend: "flashinfer_trtllm" @@ -122,7 +122,7 @@ roles: tensor-parallel-size: 8 data-parallel-size: 1 expert-parallel-size: 8 - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 128 mamba-ssm-dtype: "bfloat16" max-mamba-cache-size: 1024 diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml index 803e3816ae..1a854ac60b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml @@ -67,7 +67,7 @@ roles: enable-dp-lm-head: true moe-dense-tp-size: 1 - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 2048 mamba-ssm-dtype: "bfloat16" disaggregation-mode: "prefill" @@ -131,7 +131,7 @@ roles: enable-dp-lm-head: true prefill-round-robin-balance: true - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 128 mamba-ssm-dtype: "bfloat16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml index ad4c0b035e..76c7ea7373 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml @@ -67,7 +67,7 @@ roles: enable-dp-lm-head: true moe-dense-tp-size: 1 - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 2048 mamba-ssm-dtype: "bfloat16" disaggregation-mode: "prefill" @@ -131,7 +131,7 @@ roles: enable-dp-lm-head: true prefill-round-robin-balance: true - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 128 mamba-ssm-dtype: "bfloat16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml index 9ac9917e00..fce3130c84 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml @@ -67,7 +67,7 @@ roles: enable-dp-lm-head: true moe-dense-tp-size: 1 - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 2048 mamba-ssm-dtype: "bfloat16" disaggregation-mode: "prefill" @@ -131,7 +131,7 @@ roles: enable-dp-lm-head: true prefill-round-robin-balance: true - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 128 mamba-ssm-dtype: "bfloat16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml index 45428a9d30..ab91779e05 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml @@ -67,7 +67,7 @@ roles: enable-dp-lm-head: true moe-dense-tp-size: 1 - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 2048 mamba-ssm-dtype: "bfloat16" disaggregation-mode: "prefill" @@ -131,7 +131,7 @@ roles: enable-dp-lm-head: true prefill-round-robin-balance: true - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 128 mamba-ssm-dtype: "bfloat16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml index f85092e9d7..ea19689d1e 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml @@ -67,7 +67,7 @@ roles: enable-dp-lm-head: true moe-dense-tp-size: 1 - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 2048 mamba-ssm-dtype: "bfloat16" disaggregation-mode: "prefill" @@ -131,7 +131,7 @@ roles: enable-dp-lm-head: true prefill-round-robin-balance: true - mamba-scheduler-strategy: "no_buffer" + mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 128 mamba-ssm-dtype: "bfloat16" diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 3253adf9db..3eb9796bfe 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8091,3 +8091,10 @@ - "将 Qwen3.5 FP8 GB300 Dynamo-SGLang 8k1k MTP 的 SGLang 镜像从 lmsysorg/sglang:v0.5.14-cu130@sha256:5027e95bf6ec536856b1b52a91d1f35ff5c564ab83e8a94758a169ff09bb8df3 更新为 lmsysorg/sglang:nightly-dev-cu13-20260918-20518d85(2026-09-18 上游 cu13 dev nightly,sgl-project/sglang@20518d85,amd64 + arm64)" - "在该 key 的 7 个 srt-slurm 配方中,将 model.container 设为同一镜像,出于同样原因将 cuda-graph-max-bs 改为 cuda-graph-max-bs-decode,并将 Dynamo 安装从源码提交 46520ca59afe992fb5ef61b3197b2316f8df9b2b 改为公开的 1.5.0.dev20260917 wheel(master 中的 router 版本随之更新),这是最新的 Dynamo nightly,因为 6 月的源码构建早于 Dynamo 1.5.0.dev20260910 及之后版本所包含的 SGLang ServerArgs 兼容性修复;配方拓扑、并发与服务参数保持不变" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3255 + +- config-keys: + - qwen3.5-fp8-gb300-dynamo-sglang-mtp + description: + - "Rename mamba-scheduler-strategy to mamba-radix-cache-strategy (value no_buffer unchanged) in the 7 srt-slurm recipes of this key that set it: SGLang nightly-dev-cu13-20260918-20518d85 rejects the old spelling as an unrecognized argument (run 35317753866, every prefill and decode worker exited at argument parsing), and its help lists --mamba-radix-cache-strategy {auto,no_buffer,extra_buffer,extra_buffer_lazy} as the replacement" + - "在该 key 中设置了该参数的 7 个 srt-slurm 配方里,将 mamba-scheduler-strategy 改名为 mamba-radix-cache-strategy(取值 no_buffer 不变):SGLang nightly-dev-cu13-20260918-20518d85 将旧写法视为未识别参数(运行 35317753866 中所有 prefill 与 decode worker 在参数解析阶段退出),其帮助信息列出 --mamba-radix-cache-strategy {auto,no_buffer,extra_buffer,extra_buffer_lazy} 为替代" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3255 From dddbfc5bdbabe225acb5d3c8174ebf7e3f0b3c32 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Fri, 18 Sep 2026 03:33:09 -0400 Subject: [PATCH 4/5] qwen3.5-fp8-gb300-dynamo-sglang-mtp: replace decode prefill-round-robin-balance with load-balance-method follow_bootstrap_room MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 2026-09-18 SGLang nightly no longer has --prefill-round-robin-balance (run 35319100186). 将 decode 角色的 prefill-round-robin-balance 替换为 load-balance-method follow_bootstrap_room。 Co-Authored-By: Claude Fable 5.1 --- ...3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml | 3 +-- ...4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml | 3 +-- ...p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml | 3 +-- ...p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml | 3 +-- ...-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml | 3 +-- perf-changelog.yaml | 7 +++++++ 6 files changed, 12 insertions(+), 10 deletions(-) diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml index 1a854ac60b..70627208a1 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-3p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c480-mtp.yaml @@ -129,8 +129,7 @@ roles: moe-dense-tp-size: 1 enable-dp-attention: true enable-dp-lm-head: true - prefill-round-robin-balance: true - + load-balance-method: "follow_bootstrap_room" mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 128 mamba-ssm-dtype: "bfloat16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml index 76c7ea7373..f1a7b98476 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-4p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c768-mtp.yaml @@ -129,8 +129,7 @@ roles: moe-dense-tp-size: 1 enable-dp-attention: true enable-dp-lm-head: true - prefill-round-robin-balance: true - + load-balance-method: "follow_bootstrap_room" mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 128 mamba-ssm-dtype: "bfloat16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml index fce3130c84..66b5848a9d 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-6p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b1024-c1280-mtp.yaml @@ -129,8 +129,7 @@ roles: moe-dense-tp-size: 1 enable-dp-attention: true enable-dp-lm-head: true - prefill-round-robin-balance: true - + load-balance-method: "follow_bootstrap_room" mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 128 mamba-ssm-dtype: "bfloat16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml index ab91779e05..889912e380 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-7p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1344-mtp.yaml @@ -129,8 +129,7 @@ roles: moe-dense-tp-size: 1 enable-dp-attention: true enable-dp-lm-head: true - prefill-round-robin-balance: true - + load-balance-method: "follow_bootstrap_room" mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 128 mamba-ssm-dtype: "bfloat16" diff --git a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml index ea19689d1e..746465737c 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/qwen3.5/sglang/gb300-fp8/8k1k/disagg-8p1d-p-tp4-ep4-dp4-d-tp16-ep16-dp16-b2048-c1920x2304-mtp.yaml @@ -129,8 +129,7 @@ roles: moe-dense-tp-size: 1 enable-dp-attention: true enable-dp-lm-head: true - prefill-round-robin-balance: true - + load-balance-method: "follow_bootstrap_room" mamba-radix-cache-strategy: "no_buffer" mamba-track-interval: 128 mamba-ssm-dtype: "bfloat16" diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 3eb9796bfe..1ddffc447d 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8098,3 +8098,10 @@ - "Rename mamba-scheduler-strategy to mamba-radix-cache-strategy (value no_buffer unchanged) in the 7 srt-slurm recipes of this key that set it: SGLang nightly-dev-cu13-20260918-20518d85 rejects the old spelling as an unrecognized argument (run 35317753866, every prefill and decode worker exited at argument parsing), and its help lists --mamba-radix-cache-strategy {auto,no_buffer,extra_buffer,extra_buffer_lazy} as the replacement" - "在该 key 中设置了该参数的 7 个 srt-slurm 配方里,将 mamba-scheduler-strategy 改名为 mamba-radix-cache-strategy(取值 no_buffer 不变):SGLang nightly-dev-cu13-20260918-20518d85 将旧写法视为未识别参数(运行 35317753866 中所有 prefill 与 decode worker 在参数解析阶段退出),其帮助信息列出 --mamba-radix-cache-strategy {auto,no_buffer,extra_buffer,extra_buffer_lazy} 为替代" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3255 + +- config-keys: + - qwen3.5-fp8-gb300-dynamo-sglang-mtp + description: + - "Replace the decode-role prefill-round-robin-balance: true with load-balance-method: follow_bootstrap_room in the 5 srt-slurm recipes of this key that set it: SGLang nightly-dev-cu13-20260918-20518d85 no longer has --prefill-round-robin-balance (run 35319100186, every decode worker of the DEP16 arm exited at argument parsing) and its --load-balance-method choices are auto, round_robin, follow_bootstrap_room, total_requests and total_tokens; the prefill roles keep load-balance-method: round_robin" + - "在该 key 中设置了该参数的 5 个 srt-slurm 配方里,将 decode 角色的 prefill-round-robin-balance: true 替换为 load-balance-method: follow_bootstrap_room:SGLang nightly-dev-cu13-20260918-20518d85 已没有 --prefill-round-robin-balance(运行 35319100186 中 DEP16 分支的所有 decode worker 在参数解析阶段退出),其 --load-balance-method 可选值为 auto、round_robin、follow_bootstrap_room、total_requests 与 total_tokens;prefill 角色保持 load-balance-method: round_robin" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3255 From d1c8011105716ac264fb18eee3549326bbff468b Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Fri, 18 Sep 2026 19:35:17 -0400 Subject: [PATCH 5/5] chore: refresh PR #3255 for sweep reuse [skip-sweep] MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Sync with origin/main after the green sweep run 35319974816; the reuse gate authorizes that run on this head. 在绿色 sweep 运行 35319974816 之后与 origin/main 同步;reuse gate 在此 head 上授权该运行。 Co-Authored-By: Claude Fable 5.1