From edada598c82190fe4070ec0bfb24c22611d0fde6 Mon Sep 17 00:00:00 2001 From: wangjunwei Date: Sun, 28 Jun 2026 11:29:37 +0800 Subject: [PATCH] fix(weixin): stream LLM calls + buffer reply delivery to dodge non-stream relay bug MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit WeixinConfig lacked a streaming field, so channels.weixin.streaming was silently dropped by pydantic and supports_streaming stayed False, forcing the non-streaming Messages API. Some upstream Anthropic relays drop tool_use id/name/input on the non-stream path (but handle SSE fine), breaking WeChat tool calls. Two parts: 1. Add a streaming field (default True) so WeChat routes LLM calls through the streaming API. WeChat iLink has no native incremental delivery, so this is user-invisible — it only changes how the LLM is called. 2. WeChat send_delta previously dropped content, and the manager bypasses send for the _streamed final answer, so a streamed reply never reached the user. send_delta now buffers content deltas and flushes the full reply in one shot at _stream_end (also stopping the typing indicator via send). Co-Authored-By: Claude Opus 4.8 --- nanobot/channels/weixin.py | 37 +++++++++++++++++++++++++++++++------ 1 file changed, 31 insertions(+), 6 deletions(-) diff --git a/nanobot/channels/weixin.py b/nanobot/channels/weixin.py index 0452ea9b..75e6ddcc 100644 --- a/nanobot/channels/weixin.py +++ b/nanobot/channels/weixin.py @@ -129,6 +129,13 @@ class WeixinConfig(Base): token: str = "" # Manually set token, or obtained via QR login state_dir: str = "" # Default: ~/.nanobot/weixin/ poll_timeout: int = DEFAULT_LONG_POLL_TIMEOUT_S # seconds for long-poll + # Default on: WeChat iLink has no native incremental delivery (send_delta is + # buffered and the final answer is still sent in one shot), so streaming has + # zero user-facing effect here — it only switches the LLM call to the + # streaming API. That avoids upstream Anthropic relays that drop tool_use + # id/name/input on the non-streaming Messages path (a common third-party + # relay bug). Set to false only if a relay's streaming/SSE path is broken. + streaming: bool = True class WeixinChannel(BaseChannel): @@ -167,6 +174,10 @@ class WeixinChannel(BaseChannel): self._typing_tickets: dict[str, dict[str, Any]] = {} self._context_token_at: dict[str, float] = {} self._pending_tool_hints: dict[str, list[str]] = {} + # Buffers streamed content deltas per chat. WeChat iLink has no native + # incremental delivery, so when streaming is enabled we accumulate the + # deltas and flush the full reply in one shot at _stream_end. + self._stream_buffers: dict[str, list[str]] = {} # ------------------------------------------------------------------ # State persistence @@ -1223,14 +1234,28 @@ class WeixinChannel(BaseChannel): async def send_delta( self, chat_id: str, delta: str, metadata: dict[str, Any] | None = None ) -> None: - """Weixin iLink does not support native streaming deltas. + """Deliver a streamed reply to WeChat. - We only hook ``_stream_end`` so buffered tool hints are flushed even - when the final answer carries the ``_streamed`` flag and bypasses - :meth:`send`. + WeChat iLink has no native incremental delivery, and the manager + bypasses :meth:`send` for the ``_streamed`` final answer. So we + accumulate the content deltas here and flush the full reply as a + single message at ``_stream_end`` — otherwise a streamed reply would + never reach the user. Reasoning deltas are invisible in WeChat and are + dropped. """ - if metadata and metadata.get("_stream_end"): - await self._flush_tool_hints(chat_id) + meta = metadata or {} + if meta.get("_reasoning_delta") or meta.get("_reasoning"): + return + if delta: + self._stream_buffers.setdefault(chat_id, []).append(delta) + if not meta.get("_stream_end"): + return + full = "".join(self._stream_buffers.pop(chat_id, [])).strip() + await self._flush_tool_hints(chat_id) + if full: + await self.send( + OutboundMessage(channel=self.name, chat_id=chat_id, content=full) + ) async def _start_typing(self, chat_id: str, context_token: str = "") -> None: """Start typing indicator immediately when a message is received."""