From 32dd112f70469b655d012c205bcee748e553cd80 Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:30:44 +0800 Subject: [PATCH 1/2] [TileRT] Bump GLM-5.3 FP8 MI355X AgentX to tilert 0.1.6.post2 --- .../multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh | 2 +- configs/amd-master.yaml | 2 +- perf-changelog.yaml | 9 +++++++++ 3 files changed, 11 insertions(+), 2 deletions(-) diff --git a/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh b/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh index a8f6c3ac9f..82e81aee32 100644 --- a/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh +++ b/benchmarks/multi_node/agentic/glm5.3_fp8_mi355x_tilert.sh @@ -61,7 +61,7 @@ fi # TileRT configuration. Every value is explicit here: server_tilert.sh # validates each one with check_env_vars and supplies no defaults of its own. -export TILERT_VERSION=0.1.6.post1 +export TILERT_VERSION=0.1.6.post2 export TILERT_PROFILE=glm5_2 # decode_server --model (TileRT model profile) export TILERT_MODEL_TYPE=glm-5 # weight_converter --model_type (fallback converter) export TILERT_MODEL_PKG=glm_5_2_rocm # per-model converter package, preferred when importable diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index b26f8afda6..8594c144be 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -1530,7 +1530,7 @@ glm5.3-fp8-mi355x-tilert-agentic: runner: cluster:mi355x-amds precision: fp8 framework: tilert - router: { name: tilert-pd-router, version: "0.1.6.post1" } + router: { name: tilert-pd-router, version: "0.1.6.post2" } multinode: true disagg: true kv-p2p-transfer: mooncake diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 3f05812325..4e08e029c3 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8770,3 +8770,12 @@ - "Prioritize DSpark across the full AgentX concurrency curve and test 16 decode steps between prefill chunks with unchanged shipped precision and canonical acceptance; remove the STP sweep arm because measured DSpark points dominate its completed diagnostics." - "Reserve 64 SWA prefix tails per concurrency at C2 and above, capped at 4096, within the existing static memory budget. The 1024-tail TP4 C16 test improved throughput 4.32x over the matched default-tail baseline while preserving checkpoint math and precision." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3344 + +- config-keys: + - glm5.3-fp8-mi355x-tilert-agentic + scenario-type: + - agentic-coding + description: + - "Bump tilert 0.1.6.post1 -> 0.1.6.post2 (PyPI, 2026-09-23) on both TileRT ranks; router metadata follows. post2 changes six pd_vllm files: the ROCm decode inject (glm5_rocm_engine.py) now broadcasts each layer's KV to all eight rank caches with grouped RCCL ncclBroadcast (TILERT_INJECT_MODE, default rccl, falling back to per-source-device rotating streams, then the old serial copy) instead of serial default-stream copy_ calls, and the prefill extract (mla_nsa.py) gathers each layer with one index_select per device on side streams (TILERT_EXTRACT_MODE, default fast). The ROCm fp8 KV layout, multi-sender staging (TILERT_PD_SENDERS) and host-resident PD buffers (TILERT_PD_BUFFER_DEVICE) are new but opt-in and left off, so the 1M-context bf16 KV, layer-sharded on-GPU buffer configuration is otherwise unchanged. TileRT reports AgentX TTFT p50 5.7 s -> 2.2 s at concurrency 1. No patches, images unchanged." + - "两侧 TileRT rank 上 tilert 由 0.1.6.post1 升级到 0.1.6.post2(PyPI,2026-09-23),router 元数据随之更新。post2 修改了 pd_vllm 的六个文件:ROCm decode 注入(glm5_rocm_engine.py)改为用分组 RCCL ncclBroadcast 把每层 KV 广播到全部 8 个 rank 的缓存(TILERT_INJECT_MODE,默认 rccl,失败时回退到按源设备轮转的多流拷贝,再回退到旧的串行拷贝),不再在默认流上串行 copy_;prefill 提取(mla_nsa.py)在各设备的旁路流上每层一次 index_select 完成收集(TILERT_EXTRACT_MODE,默认 fast)。ROCm fp8 KV 布局、多发送端暂存(TILERT_PD_SENDERS)与主机内存 PD 缓冲(TILERT_PD_BUFFER_DEVICE)为新增的可选项,默认关闭,因此 1M 上下文、bf16 KV、按层分片的 GPU 缓冲配置其余不变。TileRT 报告并发 1 下 AgentX TTFT p50 由 5.7 秒降至 2.2 秒。无补丁,镜像不变。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXXX From 954b9b76194191c03619b39ba53f28416e5863a8 Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:31:07 +0800 Subject: [PATCH 2/2] Set changelog pr-link --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 4e08e029c3..594b168c42 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8778,4 +8778,4 @@ description: - "Bump tilert 0.1.6.post1 -> 0.1.6.post2 (PyPI, 2026-09-23) on both TileRT ranks; router metadata follows. post2 changes six pd_vllm files: the ROCm decode inject (glm5_rocm_engine.py) now broadcasts each layer's KV to all eight rank caches with grouped RCCL ncclBroadcast (TILERT_INJECT_MODE, default rccl, falling back to per-source-device rotating streams, then the old serial copy) instead of serial default-stream copy_ calls, and the prefill extract (mla_nsa.py) gathers each layer with one index_select per device on side streams (TILERT_EXTRACT_MODE, default fast). The ROCm fp8 KV layout, multi-sender staging (TILERT_PD_SENDERS) and host-resident PD buffers (TILERT_PD_BUFFER_DEVICE) are new but opt-in and left off, so the 1M-context bf16 KV, layer-sharded on-GPU buffer configuration is otherwise unchanged. TileRT reports AgentX TTFT p50 5.7 s -> 2.2 s at concurrency 1. No patches, images unchanged." - "两侧 TileRT rank 上 tilert 由 0.1.6.post1 升级到 0.1.6.post2(PyPI,2026-09-23),router 元数据随之更新。post2 修改了 pd_vllm 的六个文件:ROCm decode 注入(glm5_rocm_engine.py)改为用分组 RCCL ncclBroadcast 把每层 KV 广播到全部 8 个 rank 的缓存(TILERT_INJECT_MODE,默认 rccl,失败时回退到按源设备轮转的多流拷贝,再回退到旧的串行拷贝),不再在默认流上串行 copy_;prefill 提取(mla_nsa.py)在各设备的旁路流上每层一次 index_select 完成收集(TILERT_EXTRACT_MODE,默认 fast)。ROCm fp8 KV 布局、多发送端暂存(TILERT_PD_SENDERS)与主机内存 PD 缓冲(TILERT_PD_BUFFER_DEVICE)为新增的可选项,默认关闭,因此 1M 上下文、bf16 KV、按层分片的 GPU 缓冲配置其余不变。TileRT 报告并发 1 下 AgentX TTFT p50 由 5.7 秒降至 2.2 秒。无补丁,镜像不变。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3389