Files
hermes-agent/tests/hermes_cli/test_quiet_single_query.py
teknium1 4ce832eeb9 fix(cron): bot-chat delivery cap bounds the bot's turn, not its exit linger
A cron job delivering to Bot Chat booked "timed out after 600s" for turns
that finished in seconds: cron/scheduler_delivery.py::_deliver_to_bot_chat
waited for the `hermes chat -Q` child to EXIT, but when the bot's turn had
messaged a teammate (message_agent -> notify_on_complete runner) the child
then runs the one-shot exit linger, bounded by
terminal.oneshot_completion_wait_seconds (default 600) — the same default as
cron.bot_chat_delivery_timeout_seconds. The cap counts from the claim, the
linger only starts when the turn ends, so the cap expired first on every
such delivery, booked a completed turn as a timeout, held the job's fire
fence for the full cap, and killed the child mid-linger — tearing down the
reply the linger exists to protect (#90879).

Ordering the two bounds cannot fix it (the turn has no duration bound; the
linger has its own contract), so the cap stops competing with the linger:

- hermes_cli/quiet_single_query.py: the -Q child accepts a per-process
  report path (HERMES_QUIET_TURN_REPORT_FILE), popped before the turn like
  HERMES_TURN_AUTHOR so nothing the turn spawns inherits it; cli.py writes
  {pid, exit_code, error} there the moment the turn ends, BEFORE the linger.
- cron: _run_bot_chat_turn polls the child and that report under the cap.
  Report present -> the delivery is booked from it (real exit code and
  stream tails when the child exits within a short grace) and the still-
  lingering child is left running, drained and reaped by a daemon thread.
  No report by the cap -> the turn never ended: killed and booked as a
  timeout, exactly as before. The linger itself is untouched.

Live repro (real _deliver_to_bot_chat, real `hermes chat -Q` against a
loopback provider whose turn spawns `sleep 90` with notify_on_complete,
cap 45s): base books the timeout at 45.7s for a turn that ended at +5s and
kills the child; fixed head books success at 11.2s, the child lingers, the
teammate follow-up turn runs at +97s and the child exits on its own.

Supersedes #113649's cap = delivery + linger (the thread shows a headroom
only moves the race and lengthens the fence hold); analysis credit to the
reporter and the thread's independent verification.

Fixes #113608

Co-authored-by: KoNit-K <konit.block@protonmail.com>
2026-09-18 10:29:13 -07:00

112 lines
5.6 KiB
Python

"""``hermes chat -Q`` as a dispatcher's re-run of a failed bot delivery (#111721).
A failed delivery turn persists its user row before the provider call; the policy-gated re-run
replays the same session and payload in a fresh process. Told so via HERMES_RESUME_UNANSWERED_TURN,
the re-run resumes that row instead of appending a second copy of the DM.
"""
from __future__ import annotations
import os
from types import SimpleNamespace
import cli
from agent.context_compressor import _DB_PERSISTED_MARKER
from tools.bot_relay import RESUME_UNANSWERED_TURN_ENV
def _quiet_turn(monkeypatch, history, marker):
monkeypatch.delenv("HERMES_KANBAN_GOAL_MODE", raising=False)
monkeypatch.delenv("HERMES_KANBAN_TASK", raising=False)
if marker is None:
monkeypatch.delenv(RESUME_UNANSWERED_TURN_ENV, raising=False)
else:
monkeypatch.setenv(RESUME_UNANSWERED_TURN_ENV, marker)
seen = {}
def run_conversation(**kwargs):
seen["history"] = list(kwargs["conversation_history"])
seen["env"] = os.environ.get(RESUME_UNANSWERED_TURN_ENV)
return {"final_response": "ok"}
agent = SimpleNamespace(run_conversation=run_conversation, session_id="s-1")
the_cli = SimpleNamespace(agent=agent, conversation_history=history, session_id="s-1")
try:
cli._run_quiet_single_query(the_cli, "hello")
except SystemExit as exc:
assert exc.code == 0
return agent, seen
def test_rerun_resumes_the_unanswered_user_row_it_already_persisted(monkeypatch):
tail = {"role": "user", "content": "hello"}
agent, seen = _quiet_turn(monkeypatch, [{"role": "assistant", "content": "earlier"}, tail], "1")
# The persisted row leaves the history and becomes this turn's staged user dict, already
# durable — so the agent reuses it as the turn's user message and the flush writes no new row.
assert seen["history"] == [{"role": "assistant", "content": "earlier"}]
assert agent._pending_cli_user_message is tail
assert tail[_DB_PERSISTED_MARKER] is True
assert seen["env"] is None, "the marker is consumed before the turn so tool subprocesses never inherit it"
def test_without_the_marker_or_an_unanswered_tail_nothing_is_adopted(monkeypatch):
"""A person re-sending the same word after a failed turn has sent a second real message; an
answered tail is not a re-run either. Only the dispatcher's explicit marker plus an identical
unanswered tail adopts a row."""
tail = {"role": "user", "content": "hello"}
agent, seen = _quiet_turn(monkeypatch, [tail], None)
assert seen["history"] == [tail] and not hasattr(agent, "_pending_cli_user_message")
assert _DB_PERSISTED_MARKER not in tail
answered = [{"role": "user", "content": "hello"}, {"role": "assistant", "content": "done"}]
agent, seen = _quiet_turn(monkeypatch, list(answered), "1")
assert seen["history"] == answered and not hasattr(agent, "_pending_cli_user_message")
def test_rerun_adopts_the_dm_behind_the_failed_attempts_tool_scaffolding(monkeypatch):
"""A 503 after a tool round persisted user + assistant(tool_calls) + tool before the failure text,
and the dispatcher retries it. The DM is still unanswered: the re-run adopts it and drops the
failed attempt's scaffolding from the in-memory turn instead of appending a second copy."""
tail = {"role": "user", "content": "hello"}
scaffolding = [
{"role": "assistant", "content": None, "tool_calls": [{"id": "c1", "type": "function",
"function": {"name": "t", "arguments": "{}"}}]},
{"role": "tool", "tool_call_id": "c1", "content": "result"},
]
agent, seen = _quiet_turn(monkeypatch, [{"role": "assistant", "content": "earlier"}, tail, *scaffolding], "1")
assert seen["history"] == [{"role": "assistant", "content": "earlier"}]
assert agent._pending_cli_user_message is tail and tail[_DB_PERSISTED_MARKER] is True
def test_turn_report_is_written_before_the_exit_linger_and_the_path_is_not_inherited(monkeypatch, tmp_path):
"""A spawner that bounds only the turn (cron Bot Chat lane, #113608) reads the outcome from
HERMES_QUIET_TURN_REPORT_FILE: written the moment the turn ends — before the one-shot exit
linger — stamped with this pid, and the variable is popped before the turn spawns anything."""
from hermes_cli import quiet_single_query as qsq
monkeypatch.delenv("HERMES_KANBAN_GOAL_MODE", raising=False)
monkeypatch.delenv("HERMES_KANBAN_TASK", raising=False)
report = tmp_path / "turn.json"
monkeypatch.setenv(qsq.TURN_REPORT_FILE_ENV, str(report))
seen = {}
def run_conversation(**kwargs):
seen["env_during_turn"] = os.environ.get(qsq.TURN_REPORT_FILE_ENV)
seen["report_during_turn"] = report.exists()
return {"final_response": "ok"}
def linger(*args, **kwargs):
seen["report_at_linger"] = qsq.read_turn_report(str(report), os.getpid())
return {"waited": [], "completed": [], "timed_out": []}
monkeypatch.setattr("tools.process_registry.process_registry.wait_for_pending_completions", linger)
agent = SimpleNamespace(run_conversation=run_conversation, session_id="s-1")
try:
cli._run_quiet_single_query(SimpleNamespace(agent=agent, conversation_history=[], session_id="s-1"), "hello")
except SystemExit as exc:
assert exc.code == 0
assert seen["env_during_turn"] is None and seen["report_during_turn"] is False
assert seen["report_at_linger"] == {"pid": os.getpid(), "exit_code": 0, "error": ""}
# Another process's record is not this child's report.
assert qsq.read_turn_report(str(report), os.getpid() + 1) is None