refactor(telegram): a confirmed stall logs one hand-off line, not a retry promise

_schedule_polling_recovery promised the gateway 'stays alive and will retry' for every error, but a _PollingStallError goes straight to _go_fatal_network (supervisor rebuild). Branch the wording on the error type, drop the watchdog's own pre-log so a stall yields exactly one error-level line (from _go_fatal_network), and carry stalled_for/generation in the stall error text instead. Test module docstring and test name updated to match the hand-off semantics.
This commit is contained in:
kshitijk4poor
2026-09-17 20:26:30 +05:30
committed by kshitij
parent 5ca3e5546c
commit bb42672178
2 changed files with 21 additions and 14 deletions

View File

@@ -1816,9 +1816,16 @@ class TelegramAdapter(BasePlatformAdapter):
# connected for as long as the recovery ladder runs (#101391: 11 h).
if getattr(self, "_running", False):
self._mark_degraded()
logger.warning(
"[%s] Telegram polling degraded (%s); gateway stays alive and will retry. Error: %s", self.name, reason,
_redact_telegram_error_text(error))
if isinstance(error, _PollingStallError):
# Not a retry promise: the recovery path hands a confirmed stall straight to the supervisor
# (``_go_fatal_network`` logs the single error-level line for it).
logger.warning(
"[%s] Telegram polling stall confirmed (%s); handing off to the supervisor for an adapter rebuild. "
"Error: %s", self.name, reason, _redact_telegram_error_text(error))
else:
logger.warning(
"[%s] Telegram polling degraded (%s); gateway stays alive and will retry. Error: %s", self.name, reason,
_redact_telegram_error_text(error))
self._spawn_polling_recovery(asyncio.get_running_loop(), self._handle_polling_network_error(error))
async def _delete_webhook_best_effort(self, *, require_success: bool = False) -> bool:
@@ -2244,12 +2251,12 @@ class TelegramAdapter(BasePlatformAdapter):
return
if stalled_for <= _POLLING_STALL_TIMEOUT:
return
logger.error(
"[%s] Telegram polling stalled: no getUpdates progress for %.0fs "
"(generation %d). Handing the adapter to the supervisor for a rebuild instead of staying silently deaf.",
self.name, stalled_for, getattr(self, "_polling_generation", 0))
# No pre-log here: the recovery path logs the hand-off and ``_go_fatal_network`` the one
# error-level line, so a stall does not announce itself twice.
self._schedule_polling_recovery(
_PollingStallError("getUpdates made no progress for %.0fs (polling stall watchdog)" % stalled_for),
_PollingStallError(
"getUpdates made no progress for %.0fs (generation %d; polling stall watchdog)"
% (stalled_for, getattr(self, "_polling_generation", 0))),
reason="polling stall watchdog")
def _verifier_stale(self, generation: int, progress: asyncio.Event) -> bool:

View File

@@ -10,10 +10,10 @@ only a full restart recovers it.
``_check_polling_stall`` closes that hole: Telegram answers a long-poll
within ~50s, so a poller with no successful getUpdates round-trip for
``_POLLING_STALL_TIMEOUT`` seconds is unambiguously wedged, and the check
escalates loudly through the existing reconnect ladder
(``_handle_polling_network_error``). ``_polling_heartbeat_loop`` runs the
check every probe, so steady-state wedges are caught without any Bot API
call.
raises a ``_PollingStallError`` through the recovery path, which skips the
reconnect ladder and hands the adapter to the supervisor for a rebuild
(#113618). ``_polling_heartbeat_loop`` runs the check every probe, so
steady-state wedges are caught without any Bot API call.
"""
import asyncio
import time as _time
@@ -58,9 +58,9 @@ async def test_recent_progress_does_not_escalate():
@pytest.mark.asyncio
async def test_stalled_long_poll_escalates_to_reconnect_ladder():
async def test_stalled_long_poll_hands_off_to_supervisor():
"""#92991: with an empty queue and healthy get_me(), only the stall
timestamp can detect the wedged consumer — and it must."""
timestamp can detect the wedged consumer — and it must hand off."""
adapter = _make_adapter(stalled_seconds=400)
recovery = AsyncMock()
with patch.object(adapter, "_handle_polling_network_error", new=recovery):