Two parallel "real usage" mechanisms fought each other: the usage anchor (real + delta) and the compressor's rough/real projection (should_defer_preflight_to_real_usage with last_rough_tokens_when_real_prompt_fit / _pending_request_rough_tokens / note_request_rough_estimate baselines). The projection stored an anchored, real-scale figure as its "rough" baseline, so a rewind that invalidated the anchor produced phantom growth and a spurious compaction (#103391). Now there is one authority: - Post-tool gate (turn_preflight.compress_after_tool_results): anchored figure first (the raw last_prompt_tokens ignored the tool results just appended), then real, then rough. - Gateway hygiene (run_turn._hmwa_hygiene_plan): real session count, else the anchor persisted on the session row, else rough. - Preflight / pre-API gates: an anchored figure is never deferred. A whole-context rough estimate over threshold waits ONE request for the provider's real count instead of compressing on a guess (first request, rewind/edit-resend, reloaded history without a persisted anchor). - The wait is one request, never a disable: a provider that omits usage (note_usage_less_response, #2153 class), a real reading already over threshold, a rough figure past the whole window, and provider-proven overflow all compress immediately; the post-compaction latch (#36718 / #104192) is unchanged. - Projection baselines and their bookkeeping deleted (-101 LOC in context_compressor); the fixtures that scripted whole-history estimates now state the fact they relied on (provider omits usage). Fixes #103391 (closes #103397 by construction — the baseline it repaired no longer exists).
103 lines
5.3 KiB
Python
103 lines
5.3 KiB
Python
"""Pre-API pressure gate for the conversation turn loop: the Ollama runtime-context floor,
|
|
the provider-overflow re-check arming, the insufficient-progress blocker (compares fully
|
|
assembled requests, not raw ``messages``) and the call into
|
|
``turn_preflight.run_preflight_compression``. Nothing here imports
|
|
``agent.conversation_loop`` at module level (cycle)."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from contextlib import suppress
|
|
from typing import Any
|
|
|
|
from agent.message_metadata import append_message
|
|
from agent.turn_context import _compression_warrants_another_preflight_pass
|
|
from agent.turn_preflight import PreflightGateVerdict, run_preflight_compression
|
|
|
|
logger = logging.getLogger("agent.conversation_loop")
|
|
|
|
|
|
def run_preflight_gate(
|
|
agent: Any, *, request_pressure_tokens: Any, _moa_prepared_request: Any,
|
|
pending_moa_prepared_request: Any, messages: Any, system_message: Any, user_message: Any,
|
|
active_system_prompt: Any, conversation_history: Any, api_call_count: Any,
|
|
compression_attempts: Any, max_compression_attempts: Any, effective_task_id: Any,
|
|
final_response: Any, failed: Any, _turn_exit_reason: Any, _compression_timeout_exhausted: Any,
|
|
_preflight_compression_blocked: Any, _provider_overflow_recovery_pending: Any,
|
|
_last_preflight_pressure: Any,
|
|
) -> PreflightGateVerdict:
|
|
"""Run the pre-API guard chain in the original order. ``_last_preflight_pressure`` is
|
|
consumed here (set to None) and re-armed only by a compression pass, so a blocked
|
|
preflight never compares against a stale figure."""
|
|
from agent.conversation_loop import _ollama_context_limit_error
|
|
|
|
v = PreflightGateVerdict(
|
|
action="fallthrough", pending_moa_prepared_request=pending_moa_prepared_request,
|
|
messages=messages, active_system_prompt=active_system_prompt,
|
|
conversation_history=conversation_history, api_call_count=api_call_count,
|
|
compression_attempts=compression_attempts, final_response=final_response, failed=failed,
|
|
_turn_exit_reason=_turn_exit_reason,
|
|
_compression_timeout_exhausted=_compression_timeout_exhausted,
|
|
_preflight_compression_blocked=_preflight_compression_blocked,
|
|
_provider_overflow_recovery_pending=_provider_overflow_recovery_pending,
|
|
_last_preflight_pressure=None,
|
|
)
|
|
|
|
_runtime_context_error = _ollama_context_limit_error(agent, request_pressure_tokens)
|
|
if _runtime_context_error:
|
|
v.final_response = _runtime_context_error
|
|
v.failed = True
|
|
v._turn_exit_reason = "ollama_runtime_context_too_small"
|
|
append_message(messages, {"role": "assistant", "content": v.final_response})
|
|
agent._emit_status("❌ Ollama runtime context is too small for Hermes tool use")
|
|
v.api_call_count -= 1
|
|
agent._api_call_count = v.api_call_count
|
|
with suppress(Exception):
|
|
agent.iteration_budget.refund()
|
|
v.action = "break"
|
|
return v
|
|
|
|
# Pre-API pressure check: tool results grow a turn and last_prompt_tokens lags
|
|
# them. Mirror the turn-prologue guard chain: defer on noisy estimate, skip in
|
|
# failure cooldown, then should_compress().
|
|
_compressor = agent.context_compressor
|
|
_preflight_threshold = int(getattr(_compressor, "threshold_tokens", 0) or 0)
|
|
_provider_overflow_preflight = _provider_overflow_recovery_pending and (
|
|
_preflight_threshold <= 0 or request_pressure_tokens >= _preflight_threshold
|
|
)
|
|
if _provider_overflow_recovery_pending and not _provider_overflow_preflight:
|
|
# The outer-loop rebuild includes system prompt, request-only injections and
|
|
# tool schemas; only that full request with output runway may be sent.
|
|
v._provider_overflow_recovery_pending = False
|
|
# Compare fully assembled requests, not raw ``messages`` (which omit
|
|
# api_content, plugin injections, prefills, MoA context, ephemeral system text).
|
|
if (
|
|
_last_preflight_pressure is not None
|
|
and request_pressure_tokens >= _preflight_threshold
|
|
and not _compression_warrants_another_preflight_pass(
|
|
_last_preflight_pressure, request_pressure_tokens, _preflight_threshold
|
|
)
|
|
):
|
|
# Stop proactive retries this turn without consuming the shared overflow-
|
|
# recovery budget; the provider's error handler may still compact.
|
|
v._preflight_compression_blocked = True
|
|
logger.warning(
|
|
"Pre-API compression made insufficient progress: ~%s -> "
|
|
"~%s request tokens; skipping additional preflight passes",
|
|
f"{_last_preflight_pressure:,}",
|
|
f"{request_pressure_tokens:,}",
|
|
)
|
|
return run_preflight_compression(
|
|
agent, v, compressor=_compressor, request_pressure_tokens=request_pressure_tokens,
|
|
provider_overflow_preflight=_provider_overflow_preflight,
|
|
# An anchored figure is real usage + delta: never deferred. Only a whole-context rough
|
|
# estimate waits for the provider's count.
|
|
defer_preflight=(
|
|
(lambda _t: False) if getattr(agent, "_request_pressure_anchored", False)
|
|
else getattr(_compressor, "should_defer_preflight_to_real_usage", lambda _t: False)
|
|
),
|
|
moa_prepared_request=_moa_prepared_request, system_message=system_message,
|
|
user_message=user_message, max_compression_attempts=max_compression_attempts,
|
|
effective_task_id=effective_task_id,
|
|
)
|