refactor(agent/prompt): dispatch tables and helper extraction in display, context refs, breakdown, compaction

build_tool_preview -> _PREVIEW_BUILDERS per-tool table; git @refs -> _GIT_REFERENCE_ARGS; context_breakdown
_skills_block/_append_overflow dedupe; prune_pre_checkpoint_items summary retention folded into one closure;
build_skill_invocation_message reuses _render_skill_block; ruff SIM collapses; restored two compacted
cache-policy invariant comments.
This commit is contained in:
Teknium
2026-09-02 13:28:23 -07:00
parent 593c8e9146
commit 44982309b8
9 changed files with 238 additions and 254 deletions

View File

@@ -29,15 +29,15 @@ _CATEGORY_COLORS = {
def _chars_to_tokens(text: str) -> int:
if not text:
return 0
return (len(text) + 3) // 4
return (len(text) + 3) // 4 if text else 0
def _json_tokens(value: Any) -> int:
if not value:
return 0
return _chars_to_tokens(json.dumps(value, ensure_ascii=False))
return _chars_to_tokens(json.dumps(value, ensure_ascii=False)) if value else 0
def _bytes_to_tokens(size: Optional[int]) -> Optional[int]:
return None if size is None else (int(size) + 3) // 4
def _tool_name(tool: dict) -> str:
@@ -47,6 +47,12 @@ def _tool_name(tool: dict) -> str:
return str(tool.get("name") or "")
def _skills_block(stable: str) -> str:
"""The live ``<available_skills>`` block inside the stable tier, or ''."""
m = _SKILLS_BLOCK_RE.search(stable)
return m.group(0) if m else ""
def _split_tools(tools: Sequence[dict]) -> Tuple[List[dict], List[dict], List[dict]]:
builtin: List[dict] = []
mcp: List[dict] = []
@@ -63,8 +69,7 @@ def _split_tools(tools: Sequence[dict]) -> Tuple[List[dict], List[dict], List[di
def _memory_blocks(agent: Any) -> Tuple[str, str]:
memory_block = ""
user_block = ""
memory_block = user_block = ""
store = getattr(agent, "_memory_store", None)
if store is None:
return memory_block, user_block
@@ -98,9 +103,7 @@ def compute_session_context_breakdown(
stable = parts.get("stable", "") or ""
context = parts.get("context", "") or ""
volatile = parts.get("volatile", "") or ""
skills_match = _SKILLS_BLOCK_RE.search(stable)
skills_index = skills_match.group(0) if skills_match else ""
skills_index = _skills_block(stable)
memory_block, user_block = _memory_blocks(agent)
memory_text = "\n\n".join(part for part in (memory_block, user_block) if part).strip()
@@ -109,11 +112,7 @@ def compute_session_context_breakdown(
system_tail = _strip_blocks(volatile, memory_block, user_block)
system_prompt_text = "\n\n".join(part for part in (system_core, system_tail) if part).strip()
tools = list(getattr(agent, "tools", None) or [])
builtin_tools, mcp_tools, subagent_tools = _split_tools(tools)
conversation_tokens = estimate_messages_tokens_rough(messages or [])
builtin_tools, mcp_tools, subagent_tools = _split_tools(list(getattr(agent, "tools", None) or []))
categories = [
("system_prompt", "System prompt", _chars_to_tokens(system_prompt_text)),
("tool_definitions", "Tool definitions", _json_tokens(builtin_tools)),
@@ -122,43 +121,33 @@ def compute_session_context_breakdown(
("mcp", "MCP", _json_tokens(mcp_tools)),
("subagent_definitions", "Subagent definitions", _json_tokens(subagent_tools)),
("memory", "Memory", _chars_to_tokens(memory_text)),
("conversation", "Conversation", conversation_tokens),
("conversation", "Conversation", estimate_messages_tokens_rough(messages or [])),
]
estimated_total = sum(tokens for _, _, tokens in categories)
comp = getattr(agent, "context_compressor", None)
context_max = int(getattr(comp, "context_length", 0) or 0) if comp else 0
# Prefer the usage-anchored figure: provider-exact prompt+completion of
# the last response plus a delta estimate of anything appended since —
# fresher than the raw last_prompt_tokens (which lags messages appended
# after the response) and far more accurate than the heuristic total.
# Usage-anchored figure (provider-exact tokens of a response + delta of what
# was appended since) beats last_prompt_tokens (lags) and the heuristic.
# Prefer the turn-base anchor: on reasoning models later same-turn
# responses inflate prompt_tokens with replayed thinking that evaporates at
# the turn boundary, so anchoring on the LAST response makes the meter
# sawtooth. Fall back to last-response anchor, then measured/estimated.
from agent.model_metadata import anchored_context_tokens
# Prefer the turn-base anchor (first response of the current turn): on
# reasoning models, later same-turn responses inflate prompt_tokens with
# replayed thinking that evaporates at the turn boundary, so anchoring on
# the LAST response makes the meter sawtooth. Fall back to the last-
# response anchor, then to measured/estimated figures.
anchored_used = anchored_context_tokens(
messages or [],
getattr(agent, "_turn_base_usage_anchor", None),
charge_stale_thinking=False,
)
if anchored_used is None:
anchored_used = anchored_context_tokens(
messages or [], getattr(agent, "_usage_anchor", None)
)
anchored_used = anchored_context_tokens(messages or [], getattr(agent, "_usage_anchor", None))
measured_used = int(getattr(comp, "last_prompt_tokens", 0) or 0) if comp else 0
if anchored_used is not None:
context_used = anchored_used
else:
context_used = measured_used if measured_used > 0 else estimated_total
context_percent = (
max(0, min(100, round(context_used / context_max * 100)))
if context_max
else 0
)
context_percent = max(0, min(100, round(context_used / context_max * 100))) if context_max else 0
return {
"categories": [
@@ -180,10 +169,8 @@ def compute_session_context_breakdown(
# ── /context rendering (CLI + gateway) ──────────────────────────────────────
#
# Pure text renderers over the payload above. The CLI shows a glyph block-grid
# plus a category table; the gateway uses the same table without the grid
# (proportional monospace is not guaranteed on messaging platforms).
# Pure text renderers over the payload above. The gateway skips the glyph grid
# (monospace is not guaranteed on messaging platforms).
_CATEGORY_GLYPHS = {
"system_prompt": "■",
@@ -198,66 +185,43 @@ _CATEGORY_GLYPHS = {
_FREE_GLYPH = "·"
_GRID_COLUMNS = 20
_GRID_ROWS = 5 # 100 cells → 1 cell per percent of the context window
# Human-readable tables cap the expanded listings; nothing is dropped from
# the underlying data.
_DETAILS_TABLE_LIMIT = 15
def _bytes_to_tokens(size: Optional[int]) -> Optional[int]:
if size is None:
return None
return (int(size) + 3) // 4
_DETAILS_TABLE_LIMIT = 15 # display cap only; the underlying data keeps everything
def compute_context_details(agent: Any) -> Dict[str, Any]:
"""Expanded per-skill / per-toolset cost listing for ``/context all``.
Reuses the ``hermes prompt-size`` attribution mechanism (PR #66656):
per-skill index-line bytes parsed from the live ``<available_skills>``
block, and per-toolset schema bytes attributed via the tool registry's
canonical tool→toolset map. Byte figures are converted to the same
chars/4 token heuristic the categories above use.
"""
"""Expanded per-skill / per-toolset cost listing for ``/context all``,
reusing the ``hermes prompt-size`` attribution (index-line bytes from the
live skills block; schema bytes via the registry's tool→toolset map)."""
from hermes_cli.prompt_size import (
_compute_skills_breakdown,
_compute_toolsets_breakdown,
)
from agent.system_prompt import build_system_prompt_parts
parts = build_system_prompt_parts(agent)
stable = parts.get("stable", "") or ""
skills_match = _SKILLS_BLOCK_RE.search(stable)
skills_block = skills_match.group(0) if skills_match else ""
skills: List[Dict[str, Any]] = []
if skills_block:
for entry in _compute_skills_breakdown(skills_block):
skills.append({
"name": entry.get("name", ""),
"index_tokens": _bytes_to_tokens(entry.get("index_line_bytes")) or 0,
"skill_md_tokens": _bytes_to_tokens(entry.get("skill_md_bytes")),
})
toolsets: List[Dict[str, Any]] = []
skills_block = _skills_block(build_system_prompt_parts(agent).get("stable", "") or "")
skills = [
{
"name": entry.get("name", ""),
"index_tokens": _bytes_to_tokens(entry.get("index_line_bytes")) or 0,
"skill_md_tokens": _bytes_to_tokens(entry.get("skill_md_bytes")),
}
for entry in (_compute_skills_breakdown(skills_block) if skills_block else [])
]
tools = list(getattr(agent, "tools", None) or [])
if tools:
for group in _compute_toolsets_breakdown(tools):
toolsets.append({
"toolset": group.get("toolset", ""),
"tool_count": int(group.get("tool_count", 0) or 0),
"schema_tokens": _bytes_to_tokens(group.get("json_bytes")) or 0,
})
toolsets = [
{
"toolset": group.get("toolset", ""),
"tool_count": int(group.get("tool_count", 0) or 0),
"schema_tokens": _bytes_to_tokens(group.get("json_bytes")) or 0,
}
for group in (_compute_toolsets_breakdown(tools) if tools else [])
]
return {"skills": skills, "toolsets": toolsets}
def render_context_grid(payload: Dict[str, Any]) -> List[str]:
"""Render the payload as a Claude Code-style glyph block grid.
100 cells (5×20), each one percent of the model context window. Categories
fill in declaration order; the remainder renders as free space.
"""
"""Glyph block grid: 100 cells, one per percent of the context window;
categories fill in declaration order, the remainder is free space."""
context_max = int(payload.get("context_max") or 0)
categories = payload.get("categories") or []
total_cells = _GRID_COLUMNS * _GRID_ROWS
@@ -307,6 +271,12 @@ def render_context_category_lines(payload: Dict[str, Any]) -> List[str]:
return lines
def _append_overflow(lines: List[str], count: int) -> None:
remaining = count - _DETAILS_TABLE_LIMIT
if remaining > 0:
lines.append(f" … and {remaining} more")
def render_context_details_lines(details: Dict[str, Any]) -> List[str]:
"""Render the expanded ``/context all`` per-skill / per-toolset tables."""
lines: List[str] = []
@@ -319,9 +289,7 @@ def render_context_details_lines(details: Dict[str, Any]) -> List[str]:
f" {group['toolset']:<24} {group['tool_count']:>3} tools"
f" {group['schema_tokens']:>8,} tokens"
)
remaining = len(toolsets) - _DETAILS_TABLE_LIMIT
if remaining > 0:
lines.append(f" … and {remaining} more")
_append_overflow(lines, len(toolsets))
skills = details.get("skills") or []
if skills:
@@ -338,9 +306,7 @@ def render_context_details_lines(details: Dict[str, Any]) -> List[str]:
f" {name:<28} index {entry['index_tokens']:>6,}"
f" SKILL.md {md_str} tokens"
)
remaining = len(skills) - _DETAILS_TABLE_LIMIT
if remaining > 0:
lines.append(f" … and {remaining} more")
_append_overflow(lines, len(skills))
return lines
@@ -351,12 +317,8 @@ def render_context_breakdown_lines(
details: Optional[Dict[str, Any]] = None,
grid: bool = True,
) -> List[str]:
"""Render the full /context view as plain-text lines.
``grid=True`` (CLI) prepends the glyph block grid; the gateway passes
``grid=False`` and keeps its own gauge. ``details`` (from
:func:`compute_context_details`) appends the expanded listings.
"""
"""Full /context view. ``grid`` prepends the glyph grid (CLI; the gateway
keeps its own gauge); ``details`` appends the expanded listings."""
lines: List[str] = []
if grid:
lines.extend(render_context_grid(payload))
@@ -368,9 +330,7 @@ def render_context_breakdown_lines(
if context_max > 0:
pct = int(payload.get("context_percent") or 0)
lines.append("")
lines.append(
f"Context window: {context_used:,} / {context_max:,} tokens ({pct}%)"
)
lines.append(f"Context window: {context_used:,} / {context_max:,} tokens ({pct}%)")
if details is not None:
detail_lines = render_context_details_lines(details)

View File

@@ -294,6 +294,19 @@ async def preprocess_context_references_async(
)
def _git_log_args(ref: ContextReference) -> list[str]:
count = max(1, min(int(ref.target or "1"), 10))
return ["log", f"-{count}", "-p"]
# Git-backed reference kinds -> f(ref) -> git argv (the label is "git " + argv).
_GIT_REFERENCE_ARGS: dict[str, Callable[[ContextReference], list[str]]] = {
"diff": lambda ref: ["diff"],
"staged": lambda ref: ["diff", "--staged"],
"git": _git_log_args,
}
async def _expand_reference(
ref: ContextReference,
cwd: Path,
@@ -307,13 +320,9 @@ async def _expand_reference(
return _expand_file_reference(ref, cwd, allowed_root=allowed_root)
if ref.kind == "folder":
return _expand_folder_reference(ref, cwd, allowed_root=allowed_root)
if ref.kind == "diff":
return _expand_git_reference(ref, cwd, ["diff"], "git diff")
if ref.kind == "staged":
return _expand_git_reference(ref, cwd, ["diff", "--staged"], "git diff --staged")
if ref.kind == "git":
count = max(1, min(int(ref.target or "1"), 10))
return _expand_git_reference(ref, cwd, ["log", f"-{count}", "-p"], f"git log -{count} -p")
if ref.kind in _GIT_REFERENCE_ARGS:
git_args = _GIT_REFERENCE_ARGS[ref.kind](ref)
return _expand_git_reference(ref, cwd, git_args, "git " + " ".join(git_args))
if ref.kind == "url":
content = await _fetch_url_content(ref.target, url_fetcher=url_fetcher)
if not content:

View File

@@ -395,6 +395,121 @@ def _delegate_action_preview(args: dict) -> str | None:
return None
def _preview_browser_exec(args: dict, max_len: int) -> str | None:
label = _browser_exec_step_label(args)
if label is not None:
return _truncate_preview(label, max_len)
return _truncate_preview(_oneline(str(args.get("code", "") or "")), max_len) or None
def _preview_delegate_task(args: dict, max_len: int) -> str | None:
action_preview = _delegate_action_preview(args)
if action_preview is not None:
return _truncate_preview(action_preview, max_len)
tasks = args.get("tasks")
if tasks and isinstance(tasks, list):
task_count, goals = _delegate_task_goal_parts(tasks, per_goal_len=40)
preview = f"{task_count} tasks: " + " | ".join(goals) if goals else f"{len(tasks)} parallel tasks"
return _truncate_preview(preview, max_len)
goal = args.get("goal", "")
if goal is None:
return None
return _truncate_preview(_oneline(str(goal)), max_len) or None
def _preview_process_manage(args: dict, _max_len: int) -> str | None:
action = args.get("action", "")
sid = args.get("session_id", "")
data = args.get("data", "")
timeout_val = args.get("timeout")
parts = [str(action) if action else ""]
if sid:
parts.append(str(sid)[:16])
if data:
parts.append(f'"{_oneline(str(data)[:20])}"')
if timeout_val and action == "wait":
parts.append(f"{timeout_val}s")
parts = [p for p in parts if p]
return " ".join(parts) if parts else None
def _preview_todo_list(args: dict, _max_len: int) -> str:
todos_arg = args.get("todos")
if todos_arg is None:
return "reading task list"
if args.get("merge", False):
return f"updating {len(todos_arg)} task(s)"
return f"planning {len(todos_arg)} task(s)"
def _preview_shell(key: str):
def _build(args: dict, max_len: int) -> str | None:
command = args.get(key)
if command is None:
return None
return _truncate_preview(summarize_shell_command(str(command)), max_len) or None
return _build
def _preview_read_file(args: dict, max_len: int) -> str | None:
path = args.get("path") or args.get("file") or args.get("filepath")
if path is None:
return None
label = Path(str(path).replace("\\", "/")).name or str(path)
return _truncate_preview(f"{label} {_read_file_line_label(args)}".strip(), max_len) or None
def _preview_session_search(args: dict, _max_len: int) -> str:
query = _oneline(args.get("query", ""))
return f"recall: \"{query[:25]}{'...' if len(query) > 25 else ''}\""
def _preview_memory(args: dict, _max_len: int) -> str:
action = args.get("action", "")
target = args.get("target", "")
if action == "add":
content = _oneline(args.get("content", ""))
return f"+{target}: \"{content[:25]}{'...' if len(content) > 25 else ''}\""
if action in ("replace", "remove"):
old = _oneline(args.get("old_text") or "") or "<missing old_text>"
return f"{'~' if action == 'replace' else '-'}{target}: \"{old[:20]}\""
return action
def _preview_send_message(args: dict, _max_len: int) -> str:
target = args.get("target", "?")
msg = _oneline(args.get("message", ""))
if len(msg) > 20:
msg = msg[:17] + "..."
return f"to {target}: \"{msg}\""
def _preview_skill_view(args: dict, max_len: int) -> str | None:
name = _oneline(str(args.get("name") or ""))
file_path = args.get("file_path")
if file_path:
file_path = _oneline(str(file_path))
return _truncate_preview(f"{name} → {file_path}" if name else file_path, max_len) or None
return _truncate_preview(name, max_len) or None
# Tool-specific preview builders: f(args, max_len) -> preview. Tools not listed
# fall through to the primary-argument lookup in build_tool_preview.
_PREVIEW_BUILDERS = {
"browser_exec": _preview_browser_exec,
"delegate_task": _preview_delegate_task,
"process_manage": _preview_process_manage,
"todo_list": _preview_todo_list,
"terminal": _preview_shell("command"),
"execute_code": _preview_shell("code"),
"read_file": _preview_read_file,
"session_search": _preview_session_search,
"memory": _preview_memory,
"send_message": _preview_send_message,
"skill_view": _preview_skill_view,
}
def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) -> str | None:
"""Build a short preview of a tool call's primary argument for display.
@@ -406,94 +521,9 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) -
return None
args = redact_tool_args_for_display(tool_name, args) or args
def _done(preview: str) -> str | None:
return _truncate_preview(preview, max_len) if preview else None
if tool_name == "browser_exec":
label = _browser_exec_step_label(args)
if label is not None:
return _truncate_preview(label, max_len)
return _done(_oneline(str(args.get("code", "") or "")))
if tool_name == "delegate_task":
action_preview = _delegate_action_preview(args)
if action_preview is not None:
return _truncate_preview(action_preview, max_len)
tasks = args.get("tasks")
if tasks and isinstance(tasks, list):
task_count, goals = _delegate_task_goal_parts(tasks, per_goal_len=40)
preview = f"{task_count} tasks: " + " | ".join(goals) if goals else f"{len(tasks)} parallel tasks"
return _truncate_preview(preview, max_len)
goal = args.get("goal", "")
if goal is None:
return None
return _done(_oneline(str(goal)))
if tool_name == "process_manage":
action = args.get("action", "")
sid = args.get("session_id", "")
data = args.get("data", "")
timeout_val = args.get("timeout")
parts = [str(action) if action else ""]
if sid:
parts.append(str(sid)[:16])
if data:
parts.append(f'"{_oneline(str(data)[:20])}"')
if timeout_val and action == "wait":
parts.append(f"{timeout_val}s")
parts = [p for p in parts if p]
return " ".join(parts) if parts else None
if tool_name == "todo_list":
todos_arg = args.get("todos")
if todos_arg is None:
return "reading task list"
if args.get("merge", False):
return f"updating {len(todos_arg)} task(s)"
return f"planning {len(todos_arg)} task(s)"
if tool_name in {"terminal", "execute_code"}:
command = args.get("code" if tool_name == "execute_code" else "command")
if command is None:
return None
return _done(summarize_shell_command(str(command)))
if tool_name == "read_file":
path = args.get("path") or args.get("file") or args.get("filepath")
if path is None:
return None
label = Path(str(path).replace("\\", "/")).name or str(path)
return _done(f"{label} {_read_file_line_label(args)}".strip())
if tool_name == "session_search":
query = _oneline(args.get("query", ""))
return f"recall: \"{query[:25]}{'...' if len(query) > 25 else ''}\""
if tool_name == "memory":
action = args.get("action", "")
target = args.get("target", "")
if action == "add":
content = _oneline(args.get("content", ""))
return f"+{target}: \"{content[:25]}{'...' if len(content) > 25 else ''}\""
if action in ("replace", "remove"):
old = _oneline(args.get("old_text") or "") or "<missing old_text>"
return f"{'~' if action == 'replace' else '-'}{target}: \"{old[:20]}\""
return action
if tool_name == "send_message":
target = args.get("target", "?")
msg = _oneline(args.get("message", ""))
if len(msg) > 20:
msg = msg[:17] + "..."
return f"to {target}: \"{msg}\""
if tool_name == "skill_view":
name = _oneline(str(args.get("name") or ""))
file_path = args.get("file_path")
if file_path:
file_path = _oneline(str(file_path))
return _done(f"{name} → {file_path}" if name else file_path)
return _done(name)
builder = _PREVIEW_BUILDERS.get(tool_name)
if builder is not None:
return builder(args, max_len)
key = _PRIMARY_ARGS.get(tool_name) or next((k for k in _FALLBACK_PREVIEW_KEYS if k in args), None)
if not key or key not in args:

View File

@@ -220,9 +220,9 @@ def _repair_tool_call_arguments(raw_args: str, tool_name: str = "?") -> str:
json.loads(fixed)
break
except json.JSONDecodeError:
if fixed.endswith('}') and fixed.count('}') > fixed.count('{'):
fixed = fixed[:-1]
elif fixed.endswith(']') and fixed.count(']') > fixed.count('['):
if (fixed.endswith('}') and fixed.count('}') > fixed.count('{')) or (
fixed.endswith(']') and fixed.count(']') > fixed.count('[')
):
fixed = fixed[:-1]
else:
break

View File

@@ -222,9 +222,8 @@ def _extract_item_text(item: Any) -> Optional[str]:
if isinstance(part_text, str) and part_text.strip():
parts.append(part_text.strip())
part_meta = part.get("metadata")
if isinstance(part_meta, dict) and isinstance(part_meta.get("text"), str):
if part_meta["text"].strip():
parts.append(part_meta["text"].strip())
if isinstance(part_meta, dict) and isinstance(part_meta.get("text"), str) and part_meta["text"].strip():
parts.append(part_meta["text"].strip())
text = " ".join(parts)
return text if text.strip() else None
@@ -325,15 +324,17 @@ def prune_pre_checkpoint_items(
summary_remaining = max(0, int(retained_summary_token_budget))
seen_summary_texts: set = set()
def _try_retain_summary(text: Optional[str]) -> Optional[Dict[str, Any]]:
"""Budget/dedup check for a summary; return its cost or None."""
def _retain_summary(text: Optional[str], retained_item: Dict[str, Any]) -> None:
"""Retain a summary whole when it fits the budget and is not a duplicate."""
nonlocal summary_remaining
if not text or summary_remaining <= 0 or text in seen_summary_texts:
return None
return
cost = _approx_tokens(text)
if cost > summary_remaining:
return None # never slice a summary's structural framing
return # never slice a summary's structural framing
seen_summary_texts.add(text)
return {"cost": cost}
retained_reversed.append(retained_item)
summary_remaining -= cost
for item, source in zip(reversed(pre), reversed(pre_sources)):
if not isinstance(item, dict):
@@ -343,15 +344,11 @@ def prune_pre_checkpoint_items(
# when the source itself is a provenance-tagged summary carrier.
if enable_summary_retention and isinstance(source, dict) and _is_summary_item(source):
text = flatten_message_text(source.get("content"))
text = text if text.strip() else None
result = _try_retain_summary(text)
if result:
_src_role = source.get("role")
retained_reversed.append({
"role": _src_role if _src_role in ("user", "assistant") else "assistant",
"content": text,
})
summary_remaining -= result["cost"]
_src_role = source.get("role")
_retain_summary(text if text.strip() else None, {
"role": _src_role if _src_role in ("user", "assistant") else "assistant",
"content": text,
})
continue
# Typed non-message items never carry role=user or a summary flag.
@@ -372,10 +369,7 @@ def prune_pre_checkpoint_items(
text = ""
if is_summary:
result = _try_retain_summary(text)
if result:
retained_reversed.append(item)
summary_remaining -= result["cost"]
_retain_summary(text, item)
elif is_user:
if user_remaining <= 0:
continue
@@ -394,12 +388,8 @@ def prune_pre_checkpoint_items(
logger.debug(
"Pruned pre-checkpoint items: %d input -> %d retained (user_rem=%d, summary_rem=%d)",
len(items),
len(result),
user_remaining,
summary_remaining,
len(items), len(result), user_remaining, summary_remaining,
)
return result

View File

@@ -55,9 +55,12 @@ def find_stable_prefix(content: str) -> Optional[str]:
with _lock:
best: Optional[str] = None
for prefix in _prefixes:
if content.startswith(prefix) and content[len(prefix):].strip():
if best is None or len(prefix) > len(best):
best = prefix
if (
content.startswith(prefix)
and content[len(prefix):].strip()
and (best is None or len(prefix) > len(best))
):
best = prefix
if best is not None:
_prefixes.move_to_end(best) # after the scan: never mutate mid-iteration
return best

View File

@@ -114,6 +114,8 @@ def _can_carry_marker(
if content is None or content == "":
return False
if isinstance(content, list):
# Mirrors _apply_cache_marker (marks only the LAST part): a list whose
# last element isn't a dict cannot receive a marker.
return bool(content) and isinstance(content[-1], dict)
return isinstance(content, str)
@@ -188,6 +190,9 @@ def effective_cache_ttl(
if ttl != "1h":
return ttl or "5m"
if (provider or "").lower() in MEASURED_1H_PROVIDERS:
# Checked BEFORE the generic Qwen clamp (which would swallow every Qwen
# model on this route); the per-model denial stays nested so an
# opencode-go observation cannot reclamp the same model on another route.
return "5m" if _flat_model(model) in NO_1H_TIER_MODELS else "1h"
if is_qwen_model(model) or (provider or "").lower() in ALIBABA_FAMILY_PROVIDERS:
return "5m"

View File

@@ -383,11 +383,12 @@ def _render_skill_block(
loaded: tuple[dict[str, Any], Path | None, str],
activation_note: str,
task_id: str | None,
**message_kwargs: str,
) -> str:
"""Bump usage and build the message block for one loaded skill."""
loaded_skill, skill_dir, skill_name = loaded
_bump_use(skill_name, task_id)
return _build_skill_message(loaded_skill, skill_dir, activation_note, session_id=task_id)
return _build_skill_message(loaded_skill, skill_dir, activation_note, session_id=task_id, **message_kwargs)
def _scaffold_header(
@@ -627,20 +628,13 @@ def build_skill_invocation_message(
if not loaded:
return None
loaded_skill, skill_dir, skill_name = loaded
_bump_use(skill_name, task_id)
activation_note = (
f'[IMPORTANT: The user has invoked the "{skill_name}" skill, indicating they want '
"you to follow its instructions. The full skill content is loaded below.]"
)
return _build_skill_message(
loaded_skill,
skill_dir,
activation_note,
return _render_skill_block(
loaded,
f'[IMPORTANT: The user has invoked the "{loaded[2]}" skill, indicating they want '
"you to follow its instructions. The full skill content is loaded below.]",
task_id,
user_instruction=user_instruction,
runtime_note=runtime_note,
session_id=task_id,
)

View File

@@ -1,16 +1,12 @@
"""Transcript repair and in-place row reconciliation helpers for SessionDB and run_agent.
Extracted from hermes_state.py and run_agent.py to keep the godfiles narrow and bounded
under the 2K invariant (#95514 / PR #95886). Provides focused helpers to:
1. Resolve active assistant rows and watermark compaction clones in SQLite during batch appends.
2. In-place update blank assistant rows or adopt concurrent non-blank winner content without overwrite.
3. Synchronize in-memory message dicts with canonical committed content and row IDs after commit.
"""Transcript repair for SessionDB batch appends: reconcile in-memory assistant
rows with committed SQLite rows (blank-row in-place update, concurrent-winner
adoption, watermark-compaction clone lookup) and sync markers after commit.
"""
from __future__ import annotations
import sqlite3
from typing import Any, Callable, Dict, List, Optional
from typing import Any, Callable, Dict, List
from agent.context_compressor import _DB_PERSISTED_MARKER
@@ -24,12 +20,9 @@ def is_content_blank(content: Any) -> bool:
if isinstance(content, list):
if not content:
return True
texts = [
p.get("text", "")
for p in content
if isinstance(p, dict) and p.get("type") == "text"
]
return not "".join(texts).strip()
return not "".join(
p.get("text", "") for p in content if isinstance(p, dict) and p.get("type") == "text"
).strip()
return False