From bb42672178a9bfe41f5afc606f63af775e8a22e0 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Thu, 17 Sep 2026 20:26:30 +0530 Subject: [PATCH] refactor(telegram): a confirmed stall logs one hand-off line, not a retry promise _schedule_polling_recovery promised the gateway 'stays alive and will retry' for every error, but a _PollingStallError goes straight to _go_fatal_network (supervisor rebuild). Branch the wording on the error type, drop the watchdog's own pre-log so a stall yields exactly one error-level line (from _go_fatal_network), and carry stalled_for/generation in the stall error text instead. Test module docstring and test name updated to match the hand-off semantics. --- plugins/platforms/telegram/adapter.py | 23 ++++++++++++------- .../test_telegram_polling_stall_watchdog.py | 12 +++++----- 2 files changed, 21 insertions(+), 14 deletions(-) diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index acdbcf35ec..937a477030 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -1816,9 +1816,16 @@ class TelegramAdapter(BasePlatformAdapter): # connected for as long as the recovery ladder runs (#101391: 11 h). if getattr(self, "_running", False): self._mark_degraded() - logger.warning( - "[%s] Telegram polling degraded (%s); gateway stays alive and will retry. Error: %s", self.name, reason, - _redact_telegram_error_text(error)) + if isinstance(error, _PollingStallError): + # Not a retry promise: the recovery path hands a confirmed stall straight to the supervisor + # (``_go_fatal_network`` logs the single error-level line for it). + logger.warning( + "[%s] Telegram polling stall confirmed (%s); handing off to the supervisor for an adapter rebuild. " + "Error: %s", self.name, reason, _redact_telegram_error_text(error)) + else: + logger.warning( + "[%s] Telegram polling degraded (%s); gateway stays alive and will retry. Error: %s", self.name, reason, + _redact_telegram_error_text(error)) self._spawn_polling_recovery(asyncio.get_running_loop(), self._handle_polling_network_error(error)) async def _delete_webhook_best_effort(self, *, require_success: bool = False) -> bool: @@ -2244,12 +2251,12 @@ class TelegramAdapter(BasePlatformAdapter): return if stalled_for <= _POLLING_STALL_TIMEOUT: return - logger.error( - "[%s] Telegram polling stalled: no getUpdates progress for %.0fs " - "(generation %d). Handing the adapter to the supervisor for a rebuild instead of staying silently deaf.", - self.name, stalled_for, getattr(self, "_polling_generation", 0)) + # No pre-log here: the recovery path logs the hand-off and ``_go_fatal_network`` the one + # error-level line, so a stall does not announce itself twice. self._schedule_polling_recovery( - _PollingStallError("getUpdates made no progress for %.0fs (polling stall watchdog)" % stalled_for), + _PollingStallError( + "getUpdates made no progress for %.0fs (generation %d; polling stall watchdog)" + % (stalled_for, getattr(self, "_polling_generation", 0))), reason="polling stall watchdog") def _verifier_stale(self, generation: int, progress: asyncio.Event) -> bool: diff --git a/tests/gateway/test_telegram_polling_stall_watchdog.py b/tests/gateway/test_telegram_polling_stall_watchdog.py index 8fbfd0fa1c..cf0b231d4f 100644 --- a/tests/gateway/test_telegram_polling_stall_watchdog.py +++ b/tests/gateway/test_telegram_polling_stall_watchdog.py @@ -10,10 +10,10 @@ only a full restart recovers it. ``_check_polling_stall`` closes that hole: Telegram answers a long-poll within ~50s, so a poller with no successful getUpdates round-trip for ``_POLLING_STALL_TIMEOUT`` seconds is unambiguously wedged, and the check -escalates loudly through the existing reconnect ladder -(``_handle_polling_network_error``). ``_polling_heartbeat_loop`` runs the -check every probe, so steady-state wedges are caught without any Bot API -call. +raises a ``_PollingStallError`` through the recovery path, which skips the +reconnect ladder and hands the adapter to the supervisor for a rebuild +(#113618). ``_polling_heartbeat_loop`` runs the check every probe, so +steady-state wedges are caught without any Bot API call. """ import asyncio import time as _time @@ -58,9 +58,9 @@ async def test_recent_progress_does_not_escalate(): @pytest.mark.asyncio -async def test_stalled_long_poll_escalates_to_reconnect_ladder(): +async def test_stalled_long_poll_hands_off_to_supervisor(): """#92991: with an empty queue and healthy get_me(), only the stall - timestamp can detect the wedged consumer — and it must.""" + timestamp can detect the wedged consumer — and it must hand off.""" adapter = _make_adapter(stalled_seconds=400) recovery = AsyncMock() with patch.object(adapter, "_handle_polling_network_error", new=recovery):