refactor(agent/turn): AST-neutral packing of multi-line string literals

This commit is contained in:
Teknium
2026-09-02 18:45:24 -07:00
parent 863fc9afa8
commit 498c7a1c08
4 changed files with 31 additions and 65 deletions

View File

@@ -243,8 +243,7 @@ def nous_rate_limit_guard(
return _verdict("return", {
"final_response": (
f"⏳ {_nous_msg}\n\n"
"No fallback provider available. "
"Try again after the reset, or add a "
"No fallback provider available. Try again after the reset, or add a "
"fallback provider in config.yaml."
),
"messages": messages,

View File

@@ -64,8 +64,7 @@ def _retry_empty(
n = agent._empty_content_retries
wait_time = jittered_backoff(n, base_delay=5.0, max_delay=60.0)
logger.warning(
"Empty response (no content or reasoning) — "
"retry %d/%d in %.1fs (model=%s)",
"Empty response (no content or reasoning) — retry %d/%d in %.1fs (model=%s)",
n, budget, wait_time, agent.model,
)
_budget_note = (
@@ -115,8 +114,7 @@ def _terminal_empty(agent: Any, assistant_message: Any, finish_reason: str, mess
if not reasoning_text:
logger.warning(
"Empty response (no content or reasoning) "
"after %d retries. No fallback available. "
"Empty response (no content or reasoning) after %d retries. No fallback available. "
"model=%s provider=%s",
agent._empty_content_retries, agent.model,
agent.provider,
@@ -130,20 +128,15 @@ def _terminal_empty(agent: Any, assistant_message: Any, finish_reason: str, mess
reasoning_preview = reasoning_text[:500] + "..." if len(reasoning_text) > 500 else reasoning_text
logger.warning(
"Reasoning-only response (no visible content) "
"after exhausting retries and fallback. "
"Reasoning: %s", reasoning_preview,
"Reasoning-only response (no visible content) after exhausting retries and fallback. Reasoning: %s", reasoning_preview,
)
agent._emit_status(
"⚠️ Model produced reasoning but no visible "
"response after all retries. Returning empty."
"⚠️ Model produced reasoning but no visible response after all retries. Returning empty."
)
return (
"⚠️ The model produced only internal reasoning and "
"no final answer, despite retries"
"⚠️ The model produced only internal reasoning and no final answer, despite retries"
+ (" and fallback" if agent._fallback_chain else "")
+ ". Its last reasoning, which may contain the "
"answer:\n\n" + reasoning_preview
+ ". Its last reasoning, which may contain the answer:\n\n" + reasoning_preview
)
@@ -175,8 +168,7 @@ def recover_empty_response(
_turn_exit_reason = "partial_stream_recovery"
_recovered = agent._strip_think_blocks(_partial_streamed).strip()
logger.info(
"Partial stream content delivered (%d chars) "
"— using as final response",
"Partial stream content delivered (%d chars) — using as final response",
len(_recovered),
)
agent._emit_status("↻ Stream interrupted — using delivered content " "as final response")
@@ -239,8 +231,7 @@ def recover_empty_response(
if _has_structured and agent._thinking_prefill_retries < 2:
agent._thinking_prefill_retries += 1
logger.info(
"Thinking-only response (no visible content) — "
"prefilling to continue (%d/2)",
"Thinking-only response (no visible content) — prefilling to continue (%d/2)",
agent._thinking_prefill_retries,
)
agent._buffer_status(
@@ -266,23 +257,19 @@ def recover_empty_response(
if _truly_empty and _deterministic_empty:
logger.warning(
"Deterministic empty response detected "
"(consecutive zero-output completions, "
"model=%s provider=%s finish_reason=%s) — "
"skipping remaining retries",
"Deterministic empty response detected (consecutive zero-output completions, "
"model=%s provider=%s finish_reason=%s) — skipping remaining retries",
agent.model, agent.provider, finish_reason,
)
agent._buffer_status(
"⚠️ Model is deterministically returning empty "
"(zero output tokens) — skipping further retries "
"⚠️ Model is deterministically returning empty (zero output tokens) — skipping further retries "
"to avoid repeat charges"
)
# Exhausted retries — try the next provider in the chain before "(empty)".
if _truly_empty and agent._fallback_chain:
logger.warning(
"Empty response after %d retries — "
"attempting fallback (model=%s, provider=%s)",
"Empty response after %d retries — attempting fallback (model=%s, provider=%s)",
agent._empty_content_retries, agent.model,
agent.provider,
)
@@ -292,8 +279,7 @@ def recover_empty_response(
agent._empty_content_retries = 0
agent._buffer_status(f"↻ Switched to fallback: {agent.model} " f"({agent.provider})")
logger.info(
"Fallback activated after empty responses: "
"now using %s on %s",
"Fallback activated after empty responses: now using %s on %s",
agent.model, agent.provider,
)
# OUTER loop: `continue` re-runs preflight against the fallback's window;

View File

@@ -367,12 +367,9 @@ def _recover_context_length(st: _Recovery, _retry: TurnRetryState, error_msg: st
"max_tokens exceeds the provider's output cap for this model. "
"Lower model.max_tokens in config.yaml.",
notices=(
"❌ The provider rejected the request because "
"max_tokens exceeds its output cap for this model.",
" 💡 Lower model.max_tokens in your config.yaml to "
"at or below the model's max-output limit. "
"(This is an output-cap error, not a context overflow — "
"compression cannot fix it.)",
"❌ The provider rejected the request because max_tokens exceeds its output cap for this model.",
" 💡 Lower model.max_tokens in your config.yaml to at or below the model's max-output limit. "
"(This is an output-cap error, not a context overflow — compression cannot fix it.)",
),
log=(
f"{agent.log_prefix}Output-cap error not routed into compression "

View File

@@ -31,40 +31,26 @@ _FIRST_TRUNCATED_FINAL = "First response truncated due to output length limit"
_THINKING_EXHAUSTED = (
"💭 Reasoning exhausted the output token budget — no visible response was produced.",
"⚠️ **Thinking Budget Exhausted**\n\n"
"The model used all its output tokens on reasoning "
"and had none left for the actual response.\n\n"
"To fix this:\n"
"⚠️ **Thinking Budget Exhausted**\n\nThe model used all its output tokens on reasoning "
"and had none left for the actual response.\n\nTo fix this:\n"
"→ Lower reasoning effort: `/reasoning low` or `/reasoning minimal`\n"
"→ Or switch to a larger/non-reasoning model with `/model`",
"Model used all output tokens on reasoning with none left "
"for the response. Try lowering reasoning effort or "
"increasing max_tokens.",
"for the response. Try lowering reasoning effort or increasing max_tokens.",
)
_REPETITION_DOMINATED = (
"🔁 Response dominated by repeated text — stopping instead of "
"continuing a degenerate response.",
"⚠️ **Response Stopped — Repetition Detected**\n\n"
"The model fell into a repetition loop while "
"writing this response, so continuing would only "
"produce more repeated text. The partial response "
"was discarded.\n\n"
"→ Switch to a different model with `/model`\n"
"→ Or resend your message (your conversation "
"history is preserved)",
"Model output entered a repetition loop and was "
"truncated mid-loop; refusing to continue a "
"🔁 Response dominated by repeated text — stopping instead of continuing a degenerate response.",
"⚠️ **Response Stopped — Repetition Detected**\n\nThe model fell into a repetition loop while "
"writing this response, so continuing would only produce more repeated text. The partial response "
"was discarded.\n\n→ Switch to a different model with `/model`\n"
"→ Or resend your message (your conversation history is preserved)",
"Model output entered a repetition loop and was truncated mid-loop; refusing to continue a "
"degenerate response.",
)
_CEILING_NO_TEXT = (
"⚠️ **No visible answer was produced.** The "
"model hit its output-token limit on every "
"continuation attempt — its reasoning "
"consumed the entire budget each time.\n\n"
"To fix this:\n"
"→ Lower reasoning effort: `/reasoning low` "
"or `/reasoning none`\n"
"→ Or raise max_tokens for this model"
"⚠️ **No visible answer was produced.** The model hit its output-token limit on every "
"continuation attempt — its reasoning consumed the entire budget each time.\n\nTo fix this:\n"
"→ Lower reasoning effort: `/reasoning low` or `/reasoning none`\n→ Or raise max_tokens for this model"
)
@@ -526,8 +512,7 @@ def handle_content_policy_refusal(
agent._flush_status_buffer()
_refusal_log = _refusal_text[:500] + "..." if len(_refusal_text) > 500 else _refusal_text
logger.warning(
"%sModel declined to respond (finish_reason=content_filter). "
"model=%s provider=%s refusal=%s",
"%sModel declined to respond (finish_reason=content_filter). model=%s provider=%s refusal=%s",
agent.log_prefix, agent.model, agent.provider,
_refusal_log or "(no text)",
)
@@ -536,8 +521,7 @@ def handle_content_policy_refusal(
f"Model's explanation: {_refusal_text}" if _refusal_text else "The model returned no explanation."
)
_refusal_response = (
"⚠️ The model declined to respond to this request "
"(safety refusal — not a Hermes/gateway failure).\n\n"
"⚠️ The model declined to respond to this request (safety refusal — not a Hermes/gateway failure).\n\n"
f"{_refusal_detail}\n\n"
f"{_CONTENT_POLICY_RECOVERY_HINT}"
)