diff --git a/nanobot/channels/telegram.py b/nanobot/channels/telegram.py index 2dbfe2b3..6496463a 100644 --- a/nanobot/channels/telegram.py +++ b/nanobot/channels/telegram.py @@ -286,15 +286,17 @@ def _markdown_to_telegram_html(text: str) -> str: return text -def _split_telegram_markdown_html(content: str, max_html_len: int) -> list[str]: - """Split raw Telegram Markdown and return HTML chunks within Telegram's limit.""" - chunks: list[str] = [] +def _split_telegram_markdown_html_chunks( + content: str, max_html_len: int, +) -> list[tuple[str, str]]: + """Return raw Markdown and rendered HTML chunk pairs within Telegram's limit.""" + chunks: list[tuple[str, str]] = [] pending = _split_telegram_markdown(content, TELEGRAM_MAX_MESSAGE_LEN) while pending: chunk = pending.pop(0) html = _markdown_to_telegram_html(chunk) if len(html) <= max_html_len: - chunks.append(html) + chunks.append((chunk, html)) continue # Markdown can expand when rendered as HTML (tags/entities). Re-split @@ -302,16 +304,19 @@ def _split_telegram_markdown_html(content: str, max_html_len: int) -> list[str]: next_limit = max(1, int(len(chunk) * max_html_len / len(html)) - 8) next_limit = min(next_limit, len(chunk) - 1) if next_limit <= 0: - chunks.extend(split_message(html, max_html_len)) - continue + raise ValueError("A rendered Telegram HTML token exceeds the message limit") parts = _split_telegram_markdown(chunk, next_limit) if len(parts) == 1 and parts[0] == chunk: - chunks.extend(split_message(html, max_html_len)) - continue + raise ValueError("Unable to split Telegram Markdown within the HTML limit") pending = parts + pending return chunks +def _split_telegram_markdown_html(content: str, max_html_len: int) -> list[str]: + """Split raw Telegram Markdown and return HTML chunks within Telegram's limit.""" + return [html for _, html in _split_telegram_markdown_html_chunks(content, max_html_len)] + + _SEND_MAX_RETRIES = 3 _SEND_RETRY_BASE_DELAY = 0.5 # seconds, doubled each retry _STREAM_EDIT_INTERVAL_DEFAULT = 0.6 # min seconds between edit_message_text calls @@ -1057,28 +1062,27 @@ class TelegramChannel(BaseChannel): intermediate chunks as standalone messages, then opens a new message for the tail so subsequent deltas continue streaming into it. """ - chunks = _split_telegram_markdown(buf.text, TELEGRAM_MAX_MESSAGE_LEN) + chunks = _split_telegram_markdown_html_chunks(buf.text, TELEGRAM_HTML_MAX_LEN) if len(chunks) <= 1: return - html_chunks = [_markdown_to_telegram_html(chunk) for chunk in chunks] + _, first_html = chunks[0] try: await self._call_with_retry( self._app.bot.edit_message_text, chat_id=chat_id, message_id=buf.message_id, - text=html_chunks[0], + text=first_html, parse_mode="HTML", ) except Exception as e: if not self._is_not_modified_error(e): self.logger.warning("Stream overflow edit failed: {}", e) raise - for chunk in html_chunks[1:-1]: + for _, html in chunks[1:-1]: await self._call_with_retry( self._app.bot.send_message, - chat_id=chat_id, text=chunk, parse_mode="HTML", **thread_kwargs, + chat_id=chat_id, text=html, parse_mode="HTML", **thread_kwargs, ) - markdown_tail = chunks[-1] - tail_html = html_chunks[-1] + markdown_tail, tail_html = chunks[-1] sent = await self._call_with_retry( self._app.bot.send_message, chat_id=chat_id, text=tail_html, parse_mode="HTML", **thread_kwargs, diff --git a/tests/channels/test_telegram_channel.py b/tests/channels/test_telegram_channel.py index a6ce5464..fbbeb944 100644 --- a/tests/channels/test_telegram_channel.py +++ b/tests/channels/test_telegram_channel.py @@ -851,7 +851,42 @@ async def test_send_delta_incremental_edit_splits_oversized_buffer() -> None: """Mid-stream overflow: once buf.text exceeds Telegram's limit, split into chunks, edit the current message with the first chunk, and re-anchor the buffer to a new message for the tail so further deltas keep streaming.""" - from nanobot.channels.telegram import TELEGRAM_MAX_MESSAGE_LEN + channel = TelegramChannel( + TelegramConfig(enabled=True, token="123:abc", allow_from=["*"]), + MessageBus(), + ) + channel._app = _FakeApp(lambda: None) + channel._app.bot.edit_message_text = AsyncMock() + channel._app.bot.send_message = AsyncMock(return_value=SimpleNamespace(message_id=99)) + + first_chunk = f"**{'x' * 3900}**\n" + tail = "**tail** " * 50 + oversized = first_chunk + tail + channel._stream_bufs["123"] = _StreamBuf( + text=oversized, message_id=7, last_edit=0.0, stream_id="s:0" + ) + + await channel.send_delta("123", "y", stream_id="s:0") + + channel._app.bot.edit_message_text.assert_called_once() + edit_kwargs = channel._app.bot.edit_message_text.call_args.kwargs + assert edit_kwargs["text"] == _markdown_to_telegram_html(first_chunk.rstrip()) + assert edit_kwargs["parse_mode"] == "HTML" + + channel._app.bot.send_message.assert_called_once() + send_kwargs = channel._app.bot.send_message.call_args.kwargs + assert send_kwargs["parse_mode"] == "HTML" + buf = channel._stream_bufs["123"] + assert buf.message_id == 99 + assert buf.text == tail + "y" + assert send_kwargs["text"] == _markdown_to_telegram_html(buf.text) + assert buf.last_edit > 0.0 + + +@pytest.mark.asyncio +async def test_send_delta_incremental_html_expansion_does_not_overflow() -> None: + """Mid-stream HTML chunks stay within Telegram's rendered payload limit.""" + from nanobot.channels.telegram import TELEGRAM_HTML_MAX_LEN channel = TelegramChannel( TelegramConfig(enabled=True, token="123:abc", allow_from=["*"]), @@ -861,22 +896,25 @@ async def test_send_delta_incremental_edit_splits_oversized_buffer() -> None: channel._app.bot.edit_message_text = AsyncMock() channel._app.bot.send_message = AsyncMock(return_value=SimpleNamespace(message_id=99)) - oversized = "x" * (TELEGRAM_MAX_MESSAGE_LEN + 500) + oversized = "**bold** " * 501 + assert len(_markdown_to_telegram_html(oversized)) > TELEGRAM_HTML_MAX_LEN channel._stream_bufs["123"] = _StreamBuf( text=oversized, message_id=7, last_edit=0.0, stream_id="s:0" ) await channel.send_delta("123", "y", stream_id="s:0") - channel._app.bot.edit_message_text.assert_called_once() - edit_text = channel._app.bot.edit_message_text.call_args.kwargs.get("text", "") - assert len(edit_text) <= TELEGRAM_MAX_MESSAGE_LEN + payloads = [ + channel._app.bot.edit_message_text.call_args.kwargs, + *[call.kwargs for call in channel._app.bot.send_message.call_args_list], + ] + assert len(payloads) > 1 + assert all(payload["parse_mode"] == "HTML" for payload in payloads) + assert all(len(payload["text"]) <= TELEGRAM_HTML_MAX_LEN for payload in payloads) - channel._app.bot.send_message.assert_called_once() buf = channel._stream_bufs["123"] - assert buf.message_id == 99 - assert len(buf.text) <= TELEGRAM_MAX_MESSAGE_LEN - assert buf.last_edit > 0.0 + assert payloads[-1]["text"] == _markdown_to_telegram_html(buf.text) + assert "" not in buf.text @pytest.mark.asyncio