From 3a37a24efda838bdd0e19a3003d972d869ca1d8f Mon Sep 17 00:00:00 2001 From: Moep90 <3042152+Moep90@users.noreply.github.com> Date: Sun, 20 Sep 2026 07:40:05 +0200 Subject: [PATCH] fix(moa): mark reused advisor guidance as predating the tool results With `fanout: user_turn` (and off-cadence `every_n` iterations) the advisors run once per user turn and their guidance is replayed verbatim into every later iteration of that turn. The block reads as fresh instruction, so an advisor that proposes a tool call keeps proposing it after the acting model already ran it and has the result in the transcript. Observed with the clarify tool: the card was answered, the next iteration replayed the same guidance, the model issued a second identical card, the user dismissed it, and the turn then held two contradicting results for one question (an answer and an empty skip). The following aggregation degenerated into a repetition loop until it hit the output cap. - `_STALE_GUIDANCE_NOTE` is appended when cached guidance is reused on an iteration that already has tool activity since the last real user turn. The advice text is handed over unchanged; only the framing says it may be out of date. - The advisor system prompt now rules out emitting a tool call or a JSON tool-call object. Advisors hold no tools, and a tool-call object in advisory text is what the aggregator replays. No change to fanout cadence, caching or accounting. The cadence test that pinned byte-identical reuse now asserts the advice text is reused and carries the marker. Signed-off-by: Moep90 <3042152+Moep90@users.noreply.github.com> (cherry picked from commit 343787f287ad9915345090fba351df5ffa758c9e) --- agent/moa_loop.py | 40 +++++++- tests/agent/test_moa_fanout_cadence.py | 12 ++- tests/agent/test_moa_stale_guidance_note.py | 101 ++++++++++++++++++++ 3 files changed, 146 insertions(+), 7 deletions(-) create mode 100644 tests/agent/test_moa_stale_guidance_note.py diff --git a/agent/moa_loop.py b/agent/moa_loop.py index 9bebe57895..031c530d6b 100644 --- a/agent/moa_loop.py +++ b/agent/moa_loop.py @@ -206,7 +206,9 @@ _REFERENCE_SYSTEM_PROMPT = ( "systems exist and reason about them from the context given rather than " "asking for access.\n\n" "Respond with your advice directly — no preamble, no disclaimers about " - "tools or access. Your response is private guidance handed to the " + "tools or access. Advise in prose: never emit a tool call or a JSON " + "tool-call object, because the aggregator replays what looks like one. " + "Your response is private guidance handed to the " "aggregator, not an answer shown to the user. NEVER claim to have executed " "anything." ) @@ -635,6 +637,17 @@ def _render_tool_calls(tool_calls: Any) -> str: return "\n".join(lines) +# Cached guidance (user_turn / off-cadence every_n fanout) is reused on later iterations of the +# same turn, where it predates the tool results the acting model now sees. Without this line the +# block reads as fresh instruction and an advisor's suggested tool call gets replayed after it +# already ran. +_STALE_GUIDANCE_NOTE = ( + "This guidance was produced earlier in this turn, before the tool results below it. " + "Check the transcript before acting on it: a step it suggests may already have run, and " + "repeating a completed tool call is never the next step.\n" +) + + _ADVISORY_INSTRUCTION = ( "[The conversation above is the current state of the task. Give your " "most intelligent judgement: what is going on, what should happen next, " @@ -643,6 +656,19 @@ _ADVISORY_INSTRUCTION = ( ) +def _tool_activity_since_last_user(messages: list[dict[str, Any]]) -> bool: + """Whether the acting model has already called tools since the last real user turn. + Guidance cached from the start of the turn predates those results, so replaying a + tool call it suggests can repeat work the transcript already shows as done.""" + for msg in reversed(messages): + role = msg.get("role") + if role == "user": + return False + if role == "tool" or (role == "assistant" and msg.get("tool_calls")): + return True + return False + + def _reference_messages(messages: list[dict[str, Any]]) -> list[dict[str, Any]]: """Build the advisory (reference-model) view of the conversation. @@ -1266,6 +1292,7 @@ class MoAChatCompletions: def _build_guidance( self, reference_outputs: list[tuple[str, str, Any]], aggregator: dict[str, Any], degraded_reference_policy: str, + stale: bool = False, ) -> str | None: """Render the reference block attached to the aggregator prompt (None = nothing).""" agg_refs, degraded, all_failed = _guidance_inputs( @@ -1296,7 +1323,8 @@ class MoAChatCompletions: f"{header}" f"References: {', '.join(label for label, _, _ in agg_refs)}\n\n" "Use the reference responses below as private context. You are the aggregator and acting model: " - "answer the user directly or call tools as needed.\n\n" + "answer the user directly or call tools as needed.\n" + f"{_STALE_GUIDANCE_NOTE if stale else ''}\n" f"{_join_reference_outputs(agg_refs, degraded)}" ) return None @@ -1327,7 +1355,8 @@ class MoAChatCompletions: ref_messages = _reference_messages(messages) cache_key = self._fanout_cache_key(preset, ref_messages, reference_models) - if cache_key == self._ref_cache_key and self._ref_cache_outputs: + cache_hit = bool(cache_key == self._ref_cache_key and self._ref_cache_outputs) + if cache_hit: # HIT: already ran and accounted. Do NOT zero pending totals (a late # interrupted reference may have deposited) and no trace (not a new turn). reference_outputs = list(self._ref_cache_outputs) @@ -1336,7 +1365,10 @@ class MoAChatCompletions: reference_outputs = self._run_fanout(preset, ref_messages, reference_models, aggregator, aggregator_temperature, cache_key) agg_messages = [dict(m) for m in messages] - guidance = self._build_guidance(reference_outputs, aggregator, str(preset.get("degraded_reference_policy") or "loud")) + guidance = self._build_guidance( + reference_outputs, aggregator, str(preset.get("degraded_reference_policy") or "loud"), + stale=cache_hit and _tool_activity_since_last_user(messages), + ) if guidance: _attach_reference_guidance(agg_messages, guidance) diff --git a/tests/agent/test_moa_fanout_cadence.py b/tests/agent/test_moa_fanout_cadence.py index 120ad8cb39..35a0b2d85f 100644 --- a/tests/agent/test_moa_fanout_cadence.py +++ b/tests/agent/test_moa_fanout_cadence.py @@ -114,10 +114,16 @@ def test_every_n_off_cadence_iterations_reuse_cached_guidance(monkeypatch, tmp_p assert len(ref_runs) == 1 # Every iteration's aggregator request carries reference guidance... assert all(p["guidance"] for p in prepared) - # ...and the off-cadence ones reuse iteration 1's exact advice text. + # ...and the off-cadence ones reuse iteration 1's exact advice text, marked as + # predating the tool results they are now shown beside. + from agent.moa_loop import _STALE_GUIDANCE_NOTE + assert "advice #1" in prepared[0]["guidance"] - assert prepared[1]["guidance"] == prepared[0]["guidance"] - assert prepared[2]["guidance"] == prepared[0]["guidance"] + assert _STALE_GUIDANCE_NOTE not in prepared[0]["guidance"] + for reused in prepared[1:]: + assert "advice #1" in reused["guidance"] + assert _STALE_GUIDANCE_NOTE in reused["guidance"] + assert reused["guidance"].replace(_STALE_GUIDANCE_NOTE, "") == prepared[0]["guidance"] diff --git a/tests/agent/test_moa_stale_guidance_note.py b/tests/agent/test_moa_stale_guidance_note.py new file mode 100644 index 0000000000..9a36961797 --- /dev/null +++ b/tests/agent/test_moa_stale_guidance_note.py @@ -0,0 +1,101 @@ +"""Cached advisor guidance must announce that it predates the tool results below it. + +With `user_turn` (and off-cadence `every_n`) fanout the advisors run once per user turn and +their guidance is replayed verbatim into every later iteration of that turn. A reference that +proposes a tool call therefore keeps proposing it after the acting model already ran it and +got an answer, which is how a clarify card gets issued twice in one turn: once answered, once +a duplicate the user has to dismiss. +""" + +from types import SimpleNamespace + + +def _response(content="done", *, tool_calls=None): + message = SimpleNamespace(content=content, tool_calls=tool_calls or []) + return SimpleNamespace(choices=[SimpleNamespace(message=message, finish_reason="stop")], + usage=None, model="fake-model") + + +def _config(home, fanout="user_turn"): + home.mkdir() + (home / "config.yaml").write_text( + f""" +moa: + default_preset: review + presets: + review: + fanout: "{fanout}" + reference_models: + - provider: openai-codex + model: gpt-5.5 + aggregator: + provider: openrouter + model: anthropic/claude-opus-4.8 +""".strip(), + encoding="utf-8", + ) + + +def _install_fake_llm(monkeypatch, ref_runs): + def fake_call_llm(**kwargs): + if kwargs["task"] == "moa_reference": + ref_runs.append(kwargs["model"]) + return _response("call the clarify tool with three questions") + return _response("acted") + + monkeypatch.setattr("agent.moa_loop.call_llm", fake_call_llm) + + +def _after_tool_call(base): + return base + [ + {"role": "assistant", "content": "", "tool_calls": [ + {"id": "c1", "function": {"name": "clarify", "arguments": "{}"}}]}, + {"role": "tool", "tool_call_id": "c1", "content": '{"responses": [{"user_response": "Development"}]}'}, + ] + + +def _prepared(monkeypatch, tmp_path, fanout="user_turn"): + home = tmp_path / ".hermes" + _config(home, fanout) + monkeypatch.setenv("HERMES_HOME", str(home)) + ref_runs = [] + _install_fake_llm(monkeypatch, ref_runs) + + from agent.moa_loop import MoAChatCompletions + + facade = MoAChatCompletions("review") + base = [{"role": "user", "content": "use the clarify tool now"}] + first = facade.create(messages=base, tools=[], _moa_prepare_only=True) + second = facade.create(messages=_after_tool_call(base), tools=[], _moa_prepare_only=True) + return first, second, ref_runs + + +def test_reused_guidance_is_marked_as_predating_the_tool_results(monkeypatch, tmp_path): + from agent.moa_loop import _STALE_GUIDANCE_NOTE + + first, second, ref_runs = _prepared(monkeypatch, tmp_path) + + assert len(ref_runs) == 1, "user_turn fanout reuses the first run's advisors" + assert _STALE_GUIDANCE_NOTE not in first["guidance"] + assert _STALE_GUIDANCE_NOTE in second["guidance"] + # The advice itself is still handed over unchanged. + assert "call the clarify tool with three questions" in second["guidance"] + + +def test_fresh_guidance_carries_no_note(monkeypatch, tmp_path): + """per_iteration advisors see the tool result themselves, so nothing is stale.""" + from agent.moa_loop import _STALE_GUIDANCE_NOTE + + first, second, ref_runs = _prepared(monkeypatch, tmp_path, fanout="per_iteration") + + assert len(ref_runs) == 2 + assert _STALE_GUIDANCE_NOTE not in first["guidance"] + assert _STALE_GUIDANCE_NOTE not in second["guidance"] + + +def test_reference_prompt_forbids_emitting_tool_calls(monkeypatch, tmp_path): + """The advisor holds no tools; a tool-call object in its text gets replayed by the + aggregator, so the prompt has to rule it out explicitly.""" + from agent.moa_loop import _REFERENCE_SYSTEM_PROMPT + + assert "never emit a tool call" in _REFERENCE_SYSTEM_PROMPT.lower()