feat(agent): fold an agent-as-provider's own tool work back into the turn

Most providers are models: they ask Hermes to run a tool and Hermes runs it,
so the transcript and the loop's counters see every tool iteration. Some
providers are agents — an ACP CLI behind a client shim, or the codex
app-server, which already takes an analogous path in `agent/codex_runtime.py`.
They execute their own read/edit/execute tools inside their own session, and
by the time Hermes sees the response that work is done.

Those calls must never come back as pending `tool_calls` — Hermes would
re-run finished work. But summarising them into `reasoning` blinds two
subsystems:

- the self-improvement loop, which distils memories and skills by replaying
  `messages`; a one-line activity feed teaches it nothing;
- the skill-review nudge, whose `_iters_since_skill` counter only moves on
  Hermes tool iterations, of which there are none.

So a client may hand both back on the completion object —
`hermes_projected_messages` (completed assistant(tool_calls) + tool(result)
rows) and `hermes_provider_tool_iterations` — and
`splice_provider_projection` applies them. Rows go through `append_message`
like every other live-transcript append, so they carry a timestamp and
persist the same way the codex projection path's rows do.

The splice is append-only, sits before this turn's assistant message so the
order reads call -> result -> answer, and is a no-op for every client that
sets neither attribute, i.e. every ordinary OpenAI-compatible provider.
Garbage attribute values are tolerated rather than allowed to break the turn.
This commit is contained in:
Alexander Prendota
2026-08-17 16:24:05 +02:00
committed by Teknium
parent 083c5920e5
commit 07200e9cd6
3 changed files with 293 additions and 0 deletions

View File

@@ -84,6 +84,7 @@ from agent.prompt_caching import (
strip_anthropic_cache_control,
strip_anthropic_tool_cache_control,
)
from agent.provider_projection import splice_provider_projection
from agent.retry_utils import (
adaptive_rate_limit_backoff,
is_zai_coding_overload_error,
@@ -6781,6 +6782,15 @@ def run_conversation(
else:
assistant_message.content = str(raw)
# ── Agent-as-provider projection ──────────────────────────────
# A provider that IS an agent ran its own tools inside its own
# session before we got here: splice that work into the transcript
# as completed call/result rows and tick the skill-review nudge
# with the iterations Hermes never saw. Appended before this turn's
# assistant message, so the order reads call → result → answer.
# No-op for ordinary providers; see agent/provider_projection.py.
splice_provider_projection(agent, response, messages)
try:
from hermes_cli.lifecycle import (
has_hook,

View File

@@ -0,0 +1,70 @@
"""Fold an agent-as-provider's own activity back into Hermes' turn state.
Most providers are models: they ask Hermes to run a tool and Hermes runs it, so
the transcript and the loop's counters see every tool iteration. Some providers
are *agents* — an ACP CLI reached through a client shim, or the codex
app-server, which takes an analogous path in ``agent/codex_runtime.py``. They
execute their own read/edit/execute tools inside their own session, and by the
time Hermes sees the response that work is already done.
Those calls must never come back as pending ``tool_calls`` — Hermes would re-run
finished work. But two subsystems go blind if they are merely summarised into
the ``reasoning`` field:
* the **self-improvement loop**, which distils memories and skills by replaying
``messages`` — a one-line activity feed teaches it nothing;
* the **skill-review nudge**, whose counter (``_iters_since_skill``) only moves
on Hermes tool iterations, of which there are none.
So the provider client hands both back on the completion object and this helper
applies them: ``hermes_projected_messages`` (already-completed
``assistant(tool_calls=[…])`` + ``tool(result)`` history rows) and
``hermes_provider_tool_iterations`` (how many tool iterations happened inside
the provider). Clients that set neither are unaffected, which is every ordinary
OpenAI-compatible provider.
The splice is append-only and rows go through ``append_message`` like every
other live-transcript append, so they carry a timestamp and persist the same way
the codex projection path's rows do.
"""
from __future__ import annotations
import logging
from typing import Any
from agent.message_metadata import append_message
logger = logging.getLogger(__name__)
__all__ = ["splice_provider_projection"]
def splice_provider_projection(
agent: Any, response: Any, messages: list[dict[str, Any]]
) -> int:
"""Append the provider's projected history rows and tick the nudge counter.
Returns the number of rows spliced. Tolerates absent/garbage attributes so a
third-party OpenAI-compatible client can't break the turn.
"""
projected = getattr(response, "hermes_projected_messages", None)
rows = [m for m in projected if isinstance(m, dict)] if isinstance(projected, list) else []
for row in rows:
append_message(messages, row)
if rows:
logger.debug(
"spliced %d provider-projected transcript row(s) from %s",
len(rows),
getattr(agent, "provider", "?"),
)
raw_iters = getattr(response, "hermes_provider_tool_iterations", 0)
try:
iterations = int(raw_iters or 0)
except (TypeError, ValueError):
iterations = 0
if iterations > 0:
agent._iters_since_skill = getattr(agent, "_iters_since_skill", 0) + iterations
return len(rows)

View File

@@ -0,0 +1,213 @@
"""Agent-as-provider transcript projection + skill-nudge tick.
A provider that IS an agent executes its own tools inside its own session. Those
calls never come back as pending ``tool_calls`` (Hermes would re-run finished
work), so two subsystems would otherwise be blind to them:
* the self-improvement loop, which distils skills/memories from ``messages``;
* the skill-review nudge, whose counter only moves on Hermes tool iterations.
``splice_provider_projection`` closes both gaps. The helper is unit tested here,
and the wiring is exercised for real: the last tests drive a whole
``AIAgent.run_conversation`` turn against an in-process fake client and assert on
the resulting transcript and counters, so they fail if the loop ever stops
applying the projection.
"""
from __future__ import annotations
import os
import sys
from types import SimpleNamespace
_REPO_ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
if _REPO_ROOT not in sys.path:
sys.path.insert(0, _REPO_ROOT)
from agent.provider_projection import splice_provider_projection # noqa: E402
_PROJECTED = [
{
"role": "assistant",
"content": None,
"tool_calls": [
{
"id": "acp_s1_t1",
"type": "function",
"function": {"name": "acpagent_edit", "arguments": '{"path": "main.py"}'},
}
],
},
{
"role": "tool",
"tool_call_id": "acp_s1_t1",
"name": "acpagent_edit",
"content": "1 file changed",
},
]
def _projected_rows():
"""Fresh copies — the splice stamps a timestamp onto the dicts it appends."""
return [dict(row) for row in _PROJECTED]
def _agent(iters: int = 0) -> SimpleNamespace:
return SimpleNamespace(provider="acp-agent", _iters_since_skill=iters)
def _response(**attrs):
return SimpleNamespace(**attrs)
# ── unit ─────────────────────────────────────────────────────────────────────
def test_projected_rows_are_appended_and_the_nudge_ticks():
agent = _agent()
messages = [{"role": "user", "content": "edit main.py"}]
spliced = splice_provider_projection(
agent,
_response(hermes_projected_messages=_projected_rows(), hermes_provider_tool_iterations=1),
messages,
)
assert spliced == 2
assert messages[1]["tool_calls"][0]["function"]["name"] == "acpagent_edit"
assert messages[2]["content"] == "1 file changed"
assert agent._iters_since_skill == 1
def test_rows_are_stamped_like_every_other_live_transcript_append():
"""They go through ``append_message``; an unstamped row persists differently
from the ones the loop appends itself."""
messages: list = []
splice_provider_projection(
_agent(), _response(hermes_projected_messages=_projected_rows()), messages
)
assert all(isinstance(m.get("timestamp"), float) for m in messages)
def test_iterations_accumulate_across_calls():
agent = _agent(iters=2)
splice_provider_projection(agent, _response(hermes_provider_tool_iterations=3), [])
assert agent._iters_since_skill == 5
def test_ordinary_provider_response_is_a_no_op():
agent = _agent(iters=1)
messages = [{"role": "user", "content": "hi"}]
# A normal OpenAI completion carries neither attribute.
assert splice_provider_projection(agent, SimpleNamespace(choices=[]), messages) == 0
assert messages == [{"role": "user", "content": "hi"}]
assert agent._iters_since_skill == 1
def test_garbage_attributes_cannot_break_the_turn():
agent = _agent()
messages: list = []
assert splice_provider_projection(
agent,
_response(
hermes_projected_messages="not-a-list",
hermes_provider_tool_iterations="lots",
),
messages,
) == 0
assert messages == []
assert agent._iters_since_skill == 0
# A list with non-dict entries keeps only the usable rows.
assert splice_provider_projection(
agent,
_response(hermes_projected_messages=[{"role": "tool", "content": "ok"}, "junk", None]),
messages,
) == 1
assert [m["role"] for m in messages] == ["tool"]
# ── wired into the real conversation loop ────────────────────────────────────
class _FakeAgentProviderCompletions:
"""One canned completion, shaped like what an ACP client returns."""
def __init__(self, projected, iterations):
self._projected = projected
self._iterations = iterations
def create(self, **_kwargs):
message = SimpleNamespace(content="Edited main.py.", tool_calls=[], reasoning=None)
return SimpleNamespace(
choices=[SimpleNamespace(message=message, finish_reason="stop")],
usage=None,
hermes_projected_messages=self._projected,
hermes_provider_tool_iterations=self._iterations,
)
class _FakeAgentProviderClient:
def __init__(self, projected, iterations):
self.chat = SimpleNamespace(
completions=_FakeAgentProviderCompletions(projected, iterations)
)
def _run_turn(monkeypatch, *, projected, iterations):
"""Drive one real ``run_conversation`` turn against the fake client."""
from run_agent import AIAgent
monkeypatch.setattr(
"run_agent.OpenAI",
lambda **_kw: _FakeAgentProviderClient(projected, iterations),
)
monkeypatch.setattr("run_agent.get_tool_definitions", lambda *a, **k: [])
agent = AIAgent(
model="test-model",
api_key="test-key",
base_url="http://localhost:8080/v1",
platform="cli",
max_iterations=3,
quiet_mode=True,
skip_memory=True,
)
agent._disable_streaming = True
result = agent.run_conversation("edit main.py")
return agent, result
def test_provider_work_lands_in_the_transcript_through_the_real_loop(monkeypatch):
_agent_, result = _run_turn(monkeypatch, projected=_projected_rows(), iterations=1)
messages = result["messages"]
tool_rows = [
m for m in messages
if isinstance(m, dict) and m.get("role") == "tool" and m.get("name") == "acpagent_edit"
]
assert tool_rows, messages
assert "1 file changed" in tool_rows[0]["content"]
# The projected call precedes its result, and both precede the final answer.
idx_call = next(
i for i, m in enumerate(messages)
if isinstance(m, dict) and m.get("role") == "assistant" and m.get("tool_calls")
)
idx_result = messages.index(tool_rows[0])
assert idx_call < idx_result < len(messages) - 1
def test_provider_iterations_tick_the_skill_nudge_through_the_real_loop(monkeypatch):
"""Isolated from the loop's own per-iteration bump by running the same turn
with and without provider iterations."""
agent_with, _ = _run_turn(monkeypatch, projected=_projected_rows(), iterations=1)
agent_without, _ = _run_turn(monkeypatch, projected=[], iterations=0)
assert agent_with._iters_since_skill - agent_without._iters_since_skill == 1
def test_ordinary_provider_turn_is_unchanged(monkeypatch):
"""A completion without the attributes must not gain rows or counter ticks."""
_agent_, result = _run_turn(monkeypatch, projected=[], iterations=0)
assert not [
m for m in result["messages"]
if isinstance(m, dict) and str(m.get("name") or "").startswith("acpagent_")
]