diff --git a/agent/moa_loop.py b/agent/moa_loop.py index 9bebe57895..031c530d6b 100644 --- a/agent/moa_loop.py +++ b/agent/moa_loop.py @@ -206,7 +206,9 @@ _REFERENCE_SYSTEM_PROMPT = ( "systems exist and reason about them from the context given rather than " "asking for access.\n\n" "Respond with your advice directly — no preamble, no disclaimers about " - "tools or access. Your response is private guidance handed to the " + "tools or access. Advise in prose: never emit a tool call or a JSON " + "tool-call object, because the aggregator replays what looks like one. " + "Your response is private guidance handed to the " "aggregator, not an answer shown to the user. NEVER claim to have executed " "anything." ) @@ -635,6 +637,17 @@ def _render_tool_calls(tool_calls: Any) -> str: return "\n".join(lines) +# Cached guidance (user_turn / off-cadence every_n fanout) is reused on later iterations of the +# same turn, where it predates the tool results the acting model now sees. Without this line the +# block reads as fresh instruction and an advisor's suggested tool call gets replayed after it +# already ran. +_STALE_GUIDANCE_NOTE = ( + "This guidance was produced earlier in this turn, before the tool results below it. " + "Check the transcript before acting on it: a step it suggests may already have run, and " + "repeating a completed tool call is never the next step.\n" +) + + _ADVISORY_INSTRUCTION = ( "[The conversation above is the current state of the task. Give your " "most intelligent judgement: what is going on, what should happen next, " @@ -643,6 +656,19 @@ _ADVISORY_INSTRUCTION = ( ) +def _tool_activity_since_last_user(messages: list[dict[str, Any]]) -> bool: + """Whether the acting model has already called tools since the last real user turn. + Guidance cached from the start of the turn predates those results, so replaying a + tool call it suggests can repeat work the transcript already shows as done.""" + for msg in reversed(messages): + role = msg.get("role") + if role == "user": + return False + if role == "tool" or (role == "assistant" and msg.get("tool_calls")): + return True + return False + + def _reference_messages(messages: list[dict[str, Any]]) -> list[dict[str, Any]]: """Build the advisory (reference-model) view of the conversation. @@ -1266,6 +1292,7 @@ class MoAChatCompletions: def _build_guidance( self, reference_outputs: list[tuple[str, str, Any]], aggregator: dict[str, Any], degraded_reference_policy: str, + stale: bool = False, ) -> str | None: """Render the reference block attached to the aggregator prompt (None = nothing).""" agg_refs, degraded, all_failed = _guidance_inputs( @@ -1296,7 +1323,8 @@ class MoAChatCompletions: f"{header}" f"References: {', '.join(label for label, _, _ in agg_refs)}\n\n" "Use the reference responses below as private context. You are the aggregator and acting model: " - "answer the user directly or call tools as needed.\n\n" + "answer the user directly or call tools as needed.\n" + f"{_STALE_GUIDANCE_NOTE if stale else ''}\n" f"{_join_reference_outputs(agg_refs, degraded)}" ) return None @@ -1327,7 +1355,8 @@ class MoAChatCompletions: ref_messages = _reference_messages(messages) cache_key = self._fanout_cache_key(preset, ref_messages, reference_models) - if cache_key == self._ref_cache_key and self._ref_cache_outputs: + cache_hit = bool(cache_key == self._ref_cache_key and self._ref_cache_outputs) + if cache_hit: # HIT: already ran and accounted. Do NOT zero pending totals (a late # interrupted reference may have deposited) and no trace (not a new turn). reference_outputs = list(self._ref_cache_outputs) @@ -1336,7 +1365,10 @@ class MoAChatCompletions: reference_outputs = self._run_fanout(preset, ref_messages, reference_models, aggregator, aggregator_temperature, cache_key) agg_messages = [dict(m) for m in messages] - guidance = self._build_guidance(reference_outputs, aggregator, str(preset.get("degraded_reference_policy") or "loud")) + guidance = self._build_guidance( + reference_outputs, aggregator, str(preset.get("degraded_reference_policy") or "loud"), + stale=cache_hit and _tool_activity_since_last_user(messages), + ) if guidance: _attach_reference_guidance(agg_messages, guidance) diff --git a/tests/agent/test_moa_fanout_cadence.py b/tests/agent/test_moa_fanout_cadence.py index 120ad8cb39..35a0b2d85f 100644 --- a/tests/agent/test_moa_fanout_cadence.py +++ b/tests/agent/test_moa_fanout_cadence.py @@ -114,10 +114,16 @@ def test_every_n_off_cadence_iterations_reuse_cached_guidance(monkeypatch, tmp_p assert len(ref_runs) == 1 # Every iteration's aggregator request carries reference guidance... assert all(p["guidance"] for p in prepared) - # ...and the off-cadence ones reuse iteration 1's exact advice text. + # ...and the off-cadence ones reuse iteration 1's exact advice text, marked as + # predating the tool results they are now shown beside. + from agent.moa_loop import _STALE_GUIDANCE_NOTE + assert "advice #1" in prepared[0]["guidance"] - assert prepared[1]["guidance"] == prepared[0]["guidance"] - assert prepared[2]["guidance"] == prepared[0]["guidance"] + assert _STALE_GUIDANCE_NOTE not in prepared[0]["guidance"] + for reused in prepared[1:]: + assert "advice #1" in reused["guidance"] + assert _STALE_GUIDANCE_NOTE in reused["guidance"] + assert reused["guidance"].replace(_STALE_GUIDANCE_NOTE, "") == prepared[0]["guidance"] diff --git a/tests/agent/test_moa_stale_guidance_note.py b/tests/agent/test_moa_stale_guidance_note.py new file mode 100644 index 0000000000..9a36961797 --- /dev/null +++ b/tests/agent/test_moa_stale_guidance_note.py @@ -0,0 +1,101 @@ +"""Cached advisor guidance must announce that it predates the tool results below it. + +With `user_turn` (and off-cadence `every_n`) fanout the advisors run once per user turn and +their guidance is replayed verbatim into every later iteration of that turn. A reference that +proposes a tool call therefore keeps proposing it after the acting model already ran it and +got an answer, which is how a clarify card gets issued twice in one turn: once answered, once +a duplicate the user has to dismiss. +""" + +from types import SimpleNamespace + + +def _response(content="done", *, tool_calls=None): + message = SimpleNamespace(content=content, tool_calls=tool_calls or []) + return SimpleNamespace(choices=[SimpleNamespace(message=message, finish_reason="stop")], + usage=None, model="fake-model") + + +def _config(home, fanout="user_turn"): + home.mkdir() + (home / "config.yaml").write_text( + f""" +moa: + default_preset: review + presets: + review: + fanout: "{fanout}" + reference_models: + - provider: openai-codex + model: gpt-5.5 + aggregator: + provider: openrouter + model: anthropic/claude-opus-4.8 +""".strip(), + encoding="utf-8", + ) + + +def _install_fake_llm(monkeypatch, ref_runs): + def fake_call_llm(**kwargs): + if kwargs["task"] == "moa_reference": + ref_runs.append(kwargs["model"]) + return _response("call the clarify tool with three questions") + return _response("acted") + + monkeypatch.setattr("agent.moa_loop.call_llm", fake_call_llm) + + +def _after_tool_call(base): + return base + [ + {"role": "assistant", "content": "", "tool_calls": [ + {"id": "c1", "function": {"name": "clarify", "arguments": "{}"}}]}, + {"role": "tool", "tool_call_id": "c1", "content": '{"responses": [{"user_response": "Development"}]}'}, + ] + + +def _prepared(monkeypatch, tmp_path, fanout="user_turn"): + home = tmp_path / ".hermes" + _config(home, fanout) + monkeypatch.setenv("HERMES_HOME", str(home)) + ref_runs = [] + _install_fake_llm(monkeypatch, ref_runs) + + from agent.moa_loop import MoAChatCompletions + + facade = MoAChatCompletions("review") + base = [{"role": "user", "content": "use the clarify tool now"}] + first = facade.create(messages=base, tools=[], _moa_prepare_only=True) + second = facade.create(messages=_after_tool_call(base), tools=[], _moa_prepare_only=True) + return first, second, ref_runs + + +def test_reused_guidance_is_marked_as_predating_the_tool_results(monkeypatch, tmp_path): + from agent.moa_loop import _STALE_GUIDANCE_NOTE + + first, second, ref_runs = _prepared(monkeypatch, tmp_path) + + assert len(ref_runs) == 1, "user_turn fanout reuses the first run's advisors" + assert _STALE_GUIDANCE_NOTE not in first["guidance"] + assert _STALE_GUIDANCE_NOTE in second["guidance"] + # The advice itself is still handed over unchanged. + assert "call the clarify tool with three questions" in second["guidance"] + + +def test_fresh_guidance_carries_no_note(monkeypatch, tmp_path): + """per_iteration advisors see the tool result themselves, so nothing is stale.""" + from agent.moa_loop import _STALE_GUIDANCE_NOTE + + first, second, ref_runs = _prepared(monkeypatch, tmp_path, fanout="per_iteration") + + assert len(ref_runs) == 2 + assert _STALE_GUIDANCE_NOTE not in first["guidance"] + assert _STALE_GUIDANCE_NOTE not in second["guidance"] + + +def test_reference_prompt_forbids_emitting_tool_calls(monkeypatch, tmp_path): + """The advisor holds no tools; a tool-call object in its text gets replayed by the + aggregator, so the prompt has to rule it out explicitly.""" + from agent.moa_loop import _REFERENCE_SYSTEM_PROMPT + + assert "never emit a tool call" in _REFERENCE_SYSTEM_PROMPT.lower()