fix(moa): mark reused advisor guidance as predating the tool results

With `fanout: user_turn` (and off-cadence `every_n` iterations) the advisors run
once per user turn and their guidance is replayed verbatim into every later
iteration of that turn. The block reads as fresh instruction, so an advisor that
proposes a tool call keeps proposing it after the acting model already ran it and
has the result in the transcript.

Observed with the clarify tool: the card was answered, the next iteration replayed
the same guidance, the model issued a second identical card, the user dismissed it,
and the turn then held two contradicting results for one question (an answer and an
empty skip). The following aggregation degenerated into a repetition loop until it
hit the output cap.

- `_STALE_GUIDANCE_NOTE` is appended when cached guidance is reused on an iteration
  that already has tool activity since the last real user turn. The advice text is
  handed over unchanged; only the framing says it may be out of date.
- The advisor system prompt now rules out emitting a tool call or a JSON tool-call
  object. Advisors hold no tools, and a tool-call object in advisory text is what
  the aggregator replays.

No change to fanout cadence, caching or accounting. The cadence test that pinned
byte-identical reuse now asserts the advice text is reused and carries the marker.

Signed-off-by: Moep90 <3042152+Moep90@users.noreply.github.com>
(cherry picked from commit 343787f287ad9915345090fba351df5ffa758c9e)
This commit is contained in:
Moep90
2026-09-20 07:40:05 +02:00
committed by Teknium
parent 0765099ff4
commit 3a37a24efd
3 changed files with 146 additions and 7 deletions

View File

@@ -206,7 +206,9 @@ _REFERENCE_SYSTEM_PROMPT = (
"systems exist and reason about them from the context given rather than "
"asking for access.\n\n"
"Respond with your advice directly — no preamble, no disclaimers about "
"tools or access. Your response is private guidance handed to the "
"tools or access. Advise in prose: never emit a tool call or a JSON "
"tool-call object, because the aggregator replays what looks like one. "
"Your response is private guidance handed to the "
"aggregator, not an answer shown to the user. NEVER claim to have executed "
"anything."
)
@@ -635,6 +637,17 @@ def _render_tool_calls(tool_calls: Any) -> str:
return "\n".join(lines)
# Cached guidance (user_turn / off-cadence every_n fanout) is reused on later iterations of the
# same turn, where it predates the tool results the acting model now sees. Without this line the
# block reads as fresh instruction and an advisor's suggested tool call gets replayed after it
# already ran.
_STALE_GUIDANCE_NOTE = (
"This guidance was produced earlier in this turn, before the tool results below it. "
"Check the transcript before acting on it: a step it suggests may already have run, and "
"repeating a completed tool call is never the next step.\n"
)
_ADVISORY_INSTRUCTION = (
"[The conversation above is the current state of the task. Give your "
"most intelligent judgement: what is going on, what should happen next, "
@@ -643,6 +656,19 @@ _ADVISORY_INSTRUCTION = (
)
def _tool_activity_since_last_user(messages: list[dict[str, Any]]) -> bool:
"""Whether the acting model has already called tools since the last real user turn.
Guidance cached from the start of the turn predates those results, so replaying a
tool call it suggests can repeat work the transcript already shows as done."""
for msg in reversed(messages):
role = msg.get("role")
if role == "user":
return False
if role == "tool" or (role == "assistant" and msg.get("tool_calls")):
return True
return False
def _reference_messages(messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
"""Build the advisory (reference-model) view of the conversation.
@@ -1266,6 +1292,7 @@ class MoAChatCompletions:
def _build_guidance(
self, reference_outputs: list[tuple[str, str, Any]], aggregator: dict[str, Any], degraded_reference_policy: str,
stale: bool = False,
) -> str | None:
"""Render the reference block attached to the aggregator prompt (None = nothing)."""
agg_refs, degraded, all_failed = _guidance_inputs(
@@ -1296,7 +1323,8 @@ class MoAChatCompletions:
f"{header}"
f"References: {', '.join(label for label, _, _ in agg_refs)}\n\n"
"Use the reference responses below as private context. You are the aggregator and acting model: "
"answer the user directly or call tools as needed.\n\n"
"answer the user directly or call tools as needed.\n"
f"{_STALE_GUIDANCE_NOTE if stale else ''}\n"
f"{_join_reference_outputs(agg_refs, degraded)}"
)
return None
@@ -1327,7 +1355,8 @@ class MoAChatCompletions:
ref_messages = _reference_messages(messages)
cache_key = self._fanout_cache_key(preset, ref_messages, reference_models)
if cache_key == self._ref_cache_key and self._ref_cache_outputs:
cache_hit = bool(cache_key == self._ref_cache_key and self._ref_cache_outputs)
if cache_hit:
# HIT: already ran and accounted. Do NOT zero pending totals (a late
# interrupted reference may have deposited) and no trace (not a new turn).
reference_outputs = list(self._ref_cache_outputs)
@@ -1336,7 +1365,10 @@ class MoAChatCompletions:
reference_outputs = self._run_fanout(preset, ref_messages, reference_models, aggregator, aggregator_temperature, cache_key)
agg_messages = [dict(m) for m in messages]
guidance = self._build_guidance(reference_outputs, aggregator, str(preset.get("degraded_reference_policy") or "loud"))
guidance = self._build_guidance(
reference_outputs, aggregator, str(preset.get("degraded_reference_policy") or "loud"),
stale=cache_hit and _tool_activity_since_last_user(messages),
)
if guidance:
_attach_reference_guidance(agg_messages, guidance)

View File

@@ -114,10 +114,16 @@ def test_every_n_off_cadence_iterations_reuse_cached_guidance(monkeypatch, tmp_p
assert len(ref_runs) == 1
# Every iteration's aggregator request carries reference guidance...
assert all(p["guidance"] for p in prepared)
# ...and the off-cadence ones reuse iteration 1's exact advice text.
# ...and the off-cadence ones reuse iteration 1's exact advice text, marked as
# predating the tool results they are now shown beside.
from agent.moa_loop import _STALE_GUIDANCE_NOTE
assert "advice #1" in prepared[0]["guidance"]
assert prepared[1]["guidance"] == prepared[0]["guidance"]
assert prepared[2]["guidance"] == prepared[0]["guidance"]
assert _STALE_GUIDANCE_NOTE not in prepared[0]["guidance"]
for reused in prepared[1:]:
assert "advice #1" in reused["guidance"]
assert _STALE_GUIDANCE_NOTE in reused["guidance"]
assert reused["guidance"].replace(_STALE_GUIDANCE_NOTE, "") == prepared[0]["guidance"]

View File

@@ -0,0 +1,101 @@
"""Cached advisor guidance must announce that it predates the tool results below it.
With `user_turn` (and off-cadence `every_n`) fanout the advisors run once per user turn and
their guidance is replayed verbatim into every later iteration of that turn. A reference that
proposes a tool call therefore keeps proposing it after the acting model already ran it and
got an answer, which is how a clarify card gets issued twice in one turn: once answered, once
a duplicate the user has to dismiss.
"""
from types import SimpleNamespace
def _response(content="done", *, tool_calls=None):
message = SimpleNamespace(content=content, tool_calls=tool_calls or [])
return SimpleNamespace(choices=[SimpleNamespace(message=message, finish_reason="stop")],
usage=None, model="fake-model")
def _config(home, fanout="user_turn"):
home.mkdir()
(home / "config.yaml").write_text(
f"""
moa:
default_preset: review
presets:
review:
fanout: "{fanout}"
reference_models:
- provider: openai-codex
model: gpt-5.5
aggregator:
provider: openrouter
model: anthropic/claude-opus-4.8
""".strip(),
encoding="utf-8",
)
def _install_fake_llm(monkeypatch, ref_runs):
def fake_call_llm(**kwargs):
if kwargs["task"] == "moa_reference":
ref_runs.append(kwargs["model"])
return _response("call the clarify tool with three questions")
return _response("acted")
monkeypatch.setattr("agent.moa_loop.call_llm", fake_call_llm)
def _after_tool_call(base):
return base + [
{"role": "assistant", "content": "", "tool_calls": [
{"id": "c1", "function": {"name": "clarify", "arguments": "{}"}}]},
{"role": "tool", "tool_call_id": "c1", "content": '{"responses": [{"user_response": "Development"}]}'},
]
def _prepared(monkeypatch, tmp_path, fanout="user_turn"):
home = tmp_path / ".hermes"
_config(home, fanout)
monkeypatch.setenv("HERMES_HOME", str(home))
ref_runs = []
_install_fake_llm(monkeypatch, ref_runs)
from agent.moa_loop import MoAChatCompletions
facade = MoAChatCompletions("review")
base = [{"role": "user", "content": "use the clarify tool now"}]
first = facade.create(messages=base, tools=[], _moa_prepare_only=True)
second = facade.create(messages=_after_tool_call(base), tools=[], _moa_prepare_only=True)
return first, second, ref_runs
def test_reused_guidance_is_marked_as_predating_the_tool_results(monkeypatch, tmp_path):
from agent.moa_loop import _STALE_GUIDANCE_NOTE
first, second, ref_runs = _prepared(monkeypatch, tmp_path)
assert len(ref_runs) == 1, "user_turn fanout reuses the first run's advisors"
assert _STALE_GUIDANCE_NOTE not in first["guidance"]
assert _STALE_GUIDANCE_NOTE in second["guidance"]
# The advice itself is still handed over unchanged.
assert "call the clarify tool with three questions" in second["guidance"]
def test_fresh_guidance_carries_no_note(monkeypatch, tmp_path):
"""per_iteration advisors see the tool result themselves, so nothing is stale."""
from agent.moa_loop import _STALE_GUIDANCE_NOTE
first, second, ref_runs = _prepared(monkeypatch, tmp_path, fanout="per_iteration")
assert len(ref_runs) == 2
assert _STALE_GUIDANCE_NOTE not in first["guidance"]
assert _STALE_GUIDANCE_NOTE not in second["guidance"]
def test_reference_prompt_forbids_emitting_tool_calls(monkeypatch, tmp_path):
"""The advisor holds no tools; a tool-call object in its text gets replayed by the
aggregator, so the prompt has to rule it out explicitly."""
from agent.moa_loop import _REFERENCE_SYSTEM_PROMPT
assert "never emit a tool call" in _REFERENCE_SYSTEM_PROMPT.lower()