diff --git a/agent/context_breakdown.py b/agent/context_breakdown.py index 9f9c5da755..238a10b7d2 100644 --- a/agent/context_breakdown.py +++ b/agent/context_breakdown.py @@ -29,15 +29,15 @@ _CATEGORY_COLORS = { def _chars_to_tokens(text: str) -> int: - if not text: - return 0 - return (len(text) + 3) // 4 + return (len(text) + 3) // 4 if text else 0 def _json_tokens(value: Any) -> int: - if not value: - return 0 - return _chars_to_tokens(json.dumps(value, ensure_ascii=False)) + return _chars_to_tokens(json.dumps(value, ensure_ascii=False)) if value else 0 + + +def _bytes_to_tokens(size: Optional[int]) -> Optional[int]: + return None if size is None else (int(size) + 3) // 4 def _tool_name(tool: dict) -> str: @@ -47,6 +47,12 @@ def _tool_name(tool: dict) -> str: return str(tool.get("name") or "") +def _skills_block(stable: str) -> str: + """The live ```` block inside the stable tier, or ''.""" + m = _SKILLS_BLOCK_RE.search(stable) + return m.group(0) if m else "" + + def _split_tools(tools: Sequence[dict]) -> Tuple[List[dict], List[dict], List[dict]]: builtin: List[dict] = [] mcp: List[dict] = [] @@ -63,8 +69,7 @@ def _split_tools(tools: Sequence[dict]) -> Tuple[List[dict], List[dict], List[di def _memory_blocks(agent: Any) -> Tuple[str, str]: - memory_block = "" - user_block = "" + memory_block = user_block = "" store = getattr(agent, "_memory_store", None) if store is None: return memory_block, user_block @@ -98,9 +103,7 @@ def compute_session_context_breakdown( stable = parts.get("stable", "") or "" context = parts.get("context", "") or "" volatile = parts.get("volatile", "") or "" - - skills_match = _SKILLS_BLOCK_RE.search(stable) - skills_index = skills_match.group(0) if skills_match else "" + skills_index = _skills_block(stable) memory_block, user_block = _memory_blocks(agent) memory_text = "\n\n".join(part for part in (memory_block, user_block) if part).strip() @@ -109,11 +112,7 @@ def compute_session_context_breakdown( system_tail = _strip_blocks(volatile, memory_block, user_block) system_prompt_text = "\n\n".join(part for part in (system_core, system_tail) if part).strip() - tools = list(getattr(agent, "tools", None) or []) - builtin_tools, mcp_tools, subagent_tools = _split_tools(tools) - - conversation_tokens = estimate_messages_tokens_rough(messages or []) - + builtin_tools, mcp_tools, subagent_tools = _split_tools(list(getattr(agent, "tools", None) or [])) categories = [ ("system_prompt", "System prompt", _chars_to_tokens(system_prompt_text)), ("tool_definitions", "Tool definitions", _json_tokens(builtin_tools)), @@ -122,43 +121,33 @@ def compute_session_context_breakdown( ("mcp", "MCP", _json_tokens(mcp_tools)), ("subagent_definitions", "Subagent definitions", _json_tokens(subagent_tools)), ("memory", "Memory", _chars_to_tokens(memory_text)), - ("conversation", "Conversation", conversation_tokens), + ("conversation", "Conversation", estimate_messages_tokens_rough(messages or [])), ] - estimated_total = sum(tokens for _, _, tokens in categories) comp = getattr(agent, "context_compressor", None) context_max = int(getattr(comp, "context_length", 0) or 0) if comp else 0 - # Prefer the usage-anchored figure: provider-exact prompt+completion of - # the last response plus a delta estimate of anything appended since — - # fresher than the raw last_prompt_tokens (which lags messages appended - # after the response) and far more accurate than the heuristic total. + # Usage-anchored figure (provider-exact tokens of a response + delta of what + # was appended since) beats last_prompt_tokens (lags) and the heuristic. + # Prefer the turn-base anchor: on reasoning models later same-turn + # responses inflate prompt_tokens with replayed thinking that evaporates at + # the turn boundary, so anchoring on the LAST response makes the meter + # sawtooth. Fall back to last-response anchor, then measured/estimated. from agent.model_metadata import anchored_context_tokens - # Prefer the turn-base anchor (first response of the current turn): on - # reasoning models, later same-turn responses inflate prompt_tokens with - # replayed thinking that evaporates at the turn boundary, so anchoring on - # the LAST response makes the meter sawtooth. Fall back to the last- - # response anchor, then to measured/estimated figures. anchored_used = anchored_context_tokens( messages or [], getattr(agent, "_turn_base_usage_anchor", None), charge_stale_thinking=False, ) if anchored_used is None: - anchored_used = anchored_context_tokens( - messages or [], getattr(agent, "_usage_anchor", None) - ) + anchored_used = anchored_context_tokens(messages or [], getattr(agent, "_usage_anchor", None)) measured_used = int(getattr(comp, "last_prompt_tokens", 0) or 0) if comp else 0 if anchored_used is not None: context_used = anchored_used else: context_used = measured_used if measured_used > 0 else estimated_total - context_percent = ( - max(0, min(100, round(context_used / context_max * 100))) - if context_max - else 0 - ) + context_percent = max(0, min(100, round(context_used / context_max * 100))) if context_max else 0 return { "categories": [ @@ -180,10 +169,8 @@ def compute_session_context_breakdown( # ── /context rendering (CLI + gateway) ────────────────────────────────────── -# -# Pure text renderers over the payload above. The CLI shows a glyph block-grid -# plus a category table; the gateway uses the same table without the grid -# (proportional monospace is not guaranteed on messaging platforms). +# Pure text renderers over the payload above. The gateway skips the glyph grid +# (monospace is not guaranteed on messaging platforms). _CATEGORY_GLYPHS = { "system_prompt": "■", @@ -198,66 +185,43 @@ _CATEGORY_GLYPHS = { _FREE_GLYPH = "·" _GRID_COLUMNS = 20 _GRID_ROWS = 5 # 100 cells → 1 cell per percent of the context window - -# Human-readable tables cap the expanded listings; nothing is dropped from -# the underlying data. -_DETAILS_TABLE_LIMIT = 15 - - -def _bytes_to_tokens(size: Optional[int]) -> Optional[int]: - if size is None: - return None - return (int(size) + 3) // 4 +_DETAILS_TABLE_LIMIT = 15 # display cap only; the underlying data keeps everything def compute_context_details(agent: Any) -> Dict[str, Any]: - """Expanded per-skill / per-toolset cost listing for ``/context all``. - - Reuses the ``hermes prompt-size`` attribution mechanism (PR #66656): - per-skill index-line bytes parsed from the live ```` - block, and per-toolset schema bytes attributed via the tool registry's - canonical tool→toolset map. Byte figures are converted to the same - chars/4 token heuristic the categories above use. - """ + """Expanded per-skill / per-toolset cost listing for ``/context all``, + reusing the ``hermes prompt-size`` attribution (index-line bytes from the + live skills block; schema bytes via the registry's tool→toolset map).""" from hermes_cli.prompt_size import ( _compute_skills_breakdown, _compute_toolsets_breakdown, ) from agent.system_prompt import build_system_prompt_parts - parts = build_system_prompt_parts(agent) - stable = parts.get("stable", "") or "" - skills_match = _SKILLS_BLOCK_RE.search(stable) - skills_block = skills_match.group(0) if skills_match else "" - - skills: List[Dict[str, Any]] = [] - if skills_block: - for entry in _compute_skills_breakdown(skills_block): - skills.append({ - "name": entry.get("name", ""), - "index_tokens": _bytes_to_tokens(entry.get("index_line_bytes")) or 0, - "skill_md_tokens": _bytes_to_tokens(entry.get("skill_md_bytes")), - }) - - toolsets: List[Dict[str, Any]] = [] + skills_block = _skills_block(build_system_prompt_parts(agent).get("stable", "") or "") + skills = [ + { + "name": entry.get("name", ""), + "index_tokens": _bytes_to_tokens(entry.get("index_line_bytes")) or 0, + "skill_md_tokens": _bytes_to_tokens(entry.get("skill_md_bytes")), + } + for entry in (_compute_skills_breakdown(skills_block) if skills_block else []) + ] tools = list(getattr(agent, "tools", None) or []) - if tools: - for group in _compute_toolsets_breakdown(tools): - toolsets.append({ - "toolset": group.get("toolset", ""), - "tool_count": int(group.get("tool_count", 0) or 0), - "schema_tokens": _bytes_to_tokens(group.get("json_bytes")) or 0, - }) - + toolsets = [ + { + "toolset": group.get("toolset", ""), + "tool_count": int(group.get("tool_count", 0) or 0), + "schema_tokens": _bytes_to_tokens(group.get("json_bytes")) or 0, + } + for group in (_compute_toolsets_breakdown(tools) if tools else []) + ] return {"skills": skills, "toolsets": toolsets} def render_context_grid(payload: Dict[str, Any]) -> List[str]: - """Render the payload as a Claude Code-style glyph block grid. - - 100 cells (5×20), each one percent of the model context window. Categories - fill in declaration order; the remainder renders as free space. - """ + """Glyph block grid: 100 cells, one per percent of the context window; + categories fill in declaration order, the remainder is free space.""" context_max = int(payload.get("context_max") or 0) categories = payload.get("categories") or [] total_cells = _GRID_COLUMNS * _GRID_ROWS @@ -307,6 +271,12 @@ def render_context_category_lines(payload: Dict[str, Any]) -> List[str]: return lines +def _append_overflow(lines: List[str], count: int) -> None: + remaining = count - _DETAILS_TABLE_LIMIT + if remaining > 0: + lines.append(f" … and {remaining} more") + + def render_context_details_lines(details: Dict[str, Any]) -> List[str]: """Render the expanded ``/context all`` per-skill / per-toolset tables.""" lines: List[str] = [] @@ -319,9 +289,7 @@ def render_context_details_lines(details: Dict[str, Any]) -> List[str]: f" {group['toolset']:<24} {group['tool_count']:>3} tools" f" {group['schema_tokens']:>8,} tokens" ) - remaining = len(toolsets) - _DETAILS_TABLE_LIMIT - if remaining > 0: - lines.append(f" … and {remaining} more") + _append_overflow(lines, len(toolsets)) skills = details.get("skills") or [] if skills: @@ -338,9 +306,7 @@ def render_context_details_lines(details: Dict[str, Any]) -> List[str]: f" {name:<28} index {entry['index_tokens']:>6,}" f" SKILL.md {md_str} tokens" ) - remaining = len(skills) - _DETAILS_TABLE_LIMIT - if remaining > 0: - lines.append(f" … and {remaining} more") + _append_overflow(lines, len(skills)) return lines @@ -351,12 +317,8 @@ def render_context_breakdown_lines( details: Optional[Dict[str, Any]] = None, grid: bool = True, ) -> List[str]: - """Render the full /context view as plain-text lines. - - ``grid=True`` (CLI) prepends the glyph block grid; the gateway passes - ``grid=False`` and keeps its own gauge. ``details`` (from - :func:`compute_context_details`) appends the expanded listings. - """ + """Full /context view. ``grid`` prepends the glyph grid (CLI; the gateway + keeps its own gauge); ``details`` appends the expanded listings.""" lines: List[str] = [] if grid: lines.extend(render_context_grid(payload)) @@ -368,9 +330,7 @@ def render_context_breakdown_lines( if context_max > 0: pct = int(payload.get("context_percent") or 0) lines.append("") - lines.append( - f"Context window: {context_used:,} / {context_max:,} tokens ({pct}%)" - ) + lines.append(f"Context window: {context_used:,} / {context_max:,} tokens ({pct}%)") if details is not None: detail_lines = render_context_details_lines(details) diff --git a/agent/context_references.py b/agent/context_references.py index 90552285ab..b8ce46d670 100644 --- a/agent/context_references.py +++ b/agent/context_references.py @@ -294,6 +294,19 @@ async def preprocess_context_references_async( ) +def _git_log_args(ref: ContextReference) -> list[str]: + count = max(1, min(int(ref.target or "1"), 10)) + return ["log", f"-{count}", "-p"] + + +# Git-backed reference kinds -> f(ref) -> git argv (the label is "git " + argv). +_GIT_REFERENCE_ARGS: dict[str, Callable[[ContextReference], list[str]]] = { + "diff": lambda ref: ["diff"], + "staged": lambda ref: ["diff", "--staged"], + "git": _git_log_args, +} + + async def _expand_reference( ref: ContextReference, cwd: Path, @@ -307,13 +320,9 @@ async def _expand_reference( return _expand_file_reference(ref, cwd, allowed_root=allowed_root) if ref.kind == "folder": return _expand_folder_reference(ref, cwd, allowed_root=allowed_root) - if ref.kind == "diff": - return _expand_git_reference(ref, cwd, ["diff"], "git diff") - if ref.kind == "staged": - return _expand_git_reference(ref, cwd, ["diff", "--staged"], "git diff --staged") - if ref.kind == "git": - count = max(1, min(int(ref.target or "1"), 10)) - return _expand_git_reference(ref, cwd, ["log", f"-{count}", "-p"], f"git log -{count} -p") + if ref.kind in _GIT_REFERENCE_ARGS: + git_args = _GIT_REFERENCE_ARGS[ref.kind](ref) + return _expand_git_reference(ref, cwd, git_args, "git " + " ".join(git_args)) if ref.kind == "url": content = await _fetch_url_content(ref.target, url_fetcher=url_fetcher) if not content: diff --git a/agent/display.py b/agent/display.py index 109f3f205f..482fc2b3b6 100644 --- a/agent/display.py +++ b/agent/display.py @@ -395,6 +395,121 @@ def _delegate_action_preview(args: dict) -> str | None: return None +def _preview_browser_exec(args: dict, max_len: int) -> str | None: + label = _browser_exec_step_label(args) + if label is not None: + return _truncate_preview(label, max_len) + return _truncate_preview(_oneline(str(args.get("code", "") or "")), max_len) or None + + +def _preview_delegate_task(args: dict, max_len: int) -> str | None: + action_preview = _delegate_action_preview(args) + if action_preview is not None: + return _truncate_preview(action_preview, max_len) + tasks = args.get("tasks") + if tasks and isinstance(tasks, list): + task_count, goals = _delegate_task_goal_parts(tasks, per_goal_len=40) + preview = f"{task_count} tasks: " + " | ".join(goals) if goals else f"{len(tasks)} parallel tasks" + return _truncate_preview(preview, max_len) + goal = args.get("goal", "") + if goal is None: + return None + return _truncate_preview(_oneline(str(goal)), max_len) or None + + +def _preview_process_manage(args: dict, _max_len: int) -> str | None: + action = args.get("action", "") + sid = args.get("session_id", "") + data = args.get("data", "") + timeout_val = args.get("timeout") + parts = [str(action) if action else ""] + if sid: + parts.append(str(sid)[:16]) + if data: + parts.append(f'"{_oneline(str(data)[:20])}"') + if timeout_val and action == "wait": + parts.append(f"{timeout_val}s") + parts = [p for p in parts if p] + return " ".join(parts) if parts else None + + +def _preview_todo_list(args: dict, _max_len: int) -> str: + todos_arg = args.get("todos") + if todos_arg is None: + return "reading task list" + if args.get("merge", False): + return f"updating {len(todos_arg)} task(s)" + return f"planning {len(todos_arg)} task(s)" + + +def _preview_shell(key: str): + def _build(args: dict, max_len: int) -> str | None: + command = args.get(key) + if command is None: + return None + return _truncate_preview(summarize_shell_command(str(command)), max_len) or None + return _build + + +def _preview_read_file(args: dict, max_len: int) -> str | None: + path = args.get("path") or args.get("file") or args.get("filepath") + if path is None: + return None + label = Path(str(path).replace("\\", "/")).name or str(path) + return _truncate_preview(f"{label} {_read_file_line_label(args)}".strip(), max_len) or None + + +def _preview_session_search(args: dict, _max_len: int) -> str: + query = _oneline(args.get("query", "")) + return f"recall: \"{query[:25]}{'...' if len(query) > 25 else ''}\"" + + +def _preview_memory(args: dict, _max_len: int) -> str: + action = args.get("action", "") + target = args.get("target", "") + if action == "add": + content = _oneline(args.get("content", "")) + return f"+{target}: \"{content[:25]}{'...' if len(content) > 25 else ''}\"" + if action in ("replace", "remove"): + old = _oneline(args.get("old_text") or "") or "" + return f"{'~' if action == 'replace' else '-'}{target}: \"{old[:20]}\"" + return action + + +def _preview_send_message(args: dict, _max_len: int) -> str: + target = args.get("target", "?") + msg = _oneline(args.get("message", "")) + if len(msg) > 20: + msg = msg[:17] + "..." + return f"to {target}: \"{msg}\"" + + +def _preview_skill_view(args: dict, max_len: int) -> str | None: + name = _oneline(str(args.get("name") or "")) + file_path = args.get("file_path") + if file_path: + file_path = _oneline(str(file_path)) + return _truncate_preview(f"{name} → {file_path}" if name else file_path, max_len) or None + return _truncate_preview(name, max_len) or None + + +# Tool-specific preview builders: f(args, max_len) -> preview. Tools not listed +# fall through to the primary-argument lookup in build_tool_preview. +_PREVIEW_BUILDERS = { + "browser_exec": _preview_browser_exec, + "delegate_task": _preview_delegate_task, + "process_manage": _preview_process_manage, + "todo_list": _preview_todo_list, + "terminal": _preview_shell("command"), + "execute_code": _preview_shell("code"), + "read_file": _preview_read_file, + "session_search": _preview_session_search, + "memory": _preview_memory, + "send_message": _preview_send_message, + "skill_view": _preview_skill_view, +} + + def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) -> str | None: """Build a short preview of a tool call's primary argument for display. @@ -406,94 +521,9 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) - return None args = redact_tool_args_for_display(tool_name, args) or args - def _done(preview: str) -> str | None: - return _truncate_preview(preview, max_len) if preview else None - - if tool_name == "browser_exec": - label = _browser_exec_step_label(args) - if label is not None: - return _truncate_preview(label, max_len) - return _done(_oneline(str(args.get("code", "") or ""))) - - if tool_name == "delegate_task": - action_preview = _delegate_action_preview(args) - if action_preview is not None: - return _truncate_preview(action_preview, max_len) - tasks = args.get("tasks") - if tasks and isinstance(tasks, list): - task_count, goals = _delegate_task_goal_parts(tasks, per_goal_len=40) - preview = f"{task_count} tasks: " + " | ".join(goals) if goals else f"{len(tasks)} parallel tasks" - return _truncate_preview(preview, max_len) - goal = args.get("goal", "") - if goal is None: - return None - return _done(_oneline(str(goal))) - - if tool_name == "process_manage": - action = args.get("action", "") - sid = args.get("session_id", "") - data = args.get("data", "") - timeout_val = args.get("timeout") - parts = [str(action) if action else ""] - if sid: - parts.append(str(sid)[:16]) - if data: - parts.append(f'"{_oneline(str(data)[:20])}"') - if timeout_val and action == "wait": - parts.append(f"{timeout_val}s") - parts = [p for p in parts if p] - return " ".join(parts) if parts else None - - if tool_name == "todo_list": - todos_arg = args.get("todos") - if todos_arg is None: - return "reading task list" - if args.get("merge", False): - return f"updating {len(todos_arg)} task(s)" - return f"planning {len(todos_arg)} task(s)" - - if tool_name in {"terminal", "execute_code"}: - command = args.get("code" if tool_name == "execute_code" else "command") - if command is None: - return None - return _done(summarize_shell_command(str(command))) - - if tool_name == "read_file": - path = args.get("path") or args.get("file") or args.get("filepath") - if path is None: - return None - label = Path(str(path).replace("\\", "/")).name or str(path) - return _done(f"{label} {_read_file_line_label(args)}".strip()) - - if tool_name == "session_search": - query = _oneline(args.get("query", "")) - return f"recall: \"{query[:25]}{'...' if len(query) > 25 else ''}\"" - - if tool_name == "memory": - action = args.get("action", "") - target = args.get("target", "") - if action == "add": - content = _oneline(args.get("content", "")) - return f"+{target}: \"{content[:25]}{'...' if len(content) > 25 else ''}\"" - if action in ("replace", "remove"): - old = _oneline(args.get("old_text") or "") or "" - return f"{'~' if action == 'replace' else '-'}{target}: \"{old[:20]}\"" - return action - - if tool_name == "send_message": - target = args.get("target", "?") - msg = _oneline(args.get("message", "")) - if len(msg) > 20: - msg = msg[:17] + "..." - return f"to {target}: \"{msg}\"" - - if tool_name == "skill_view": - name = _oneline(str(args.get("name") or "")) - file_path = args.get("file_path") - if file_path: - file_path = _oneline(str(file_path)) - return _done(f"{name} → {file_path}" if name else file_path) - return _done(name) + builder = _PREVIEW_BUILDERS.get(tool_name) + if builder is not None: + return builder(args, max_len) key = _PRIMARY_ARGS.get(tool_name) or next((k for k in _FALLBACK_PREVIEW_KEYS if k in args), None) if not key or key not in args: diff --git a/agent/message_sanitization.py b/agent/message_sanitization.py index d4deab39c0..d8ce465004 100644 --- a/agent/message_sanitization.py +++ b/agent/message_sanitization.py @@ -220,9 +220,9 @@ def _repair_tool_call_arguments(raw_args: str, tool_name: str = "?") -> str: json.loads(fixed) break except json.JSONDecodeError: - if fixed.endswith('}') and fixed.count('}') > fixed.count('{'): - fixed = fixed[:-1] - elif fixed.endswith(']') and fixed.count(']') > fixed.count('['): + if (fixed.endswith('}') and fixed.count('}') > fixed.count('{')) or ( + fixed.endswith(']') and fixed.count(']') > fixed.count('[') + ): fixed = fixed[:-1] else: break diff --git a/agent/native_compaction.py b/agent/native_compaction.py index 30e6efdb99..d63d22f49c 100644 --- a/agent/native_compaction.py +++ b/agent/native_compaction.py @@ -222,9 +222,8 @@ def _extract_item_text(item: Any) -> Optional[str]: if isinstance(part_text, str) and part_text.strip(): parts.append(part_text.strip()) part_meta = part.get("metadata") - if isinstance(part_meta, dict) and isinstance(part_meta.get("text"), str): - if part_meta["text"].strip(): - parts.append(part_meta["text"].strip()) + if isinstance(part_meta, dict) and isinstance(part_meta.get("text"), str) and part_meta["text"].strip(): + parts.append(part_meta["text"].strip()) text = " ".join(parts) return text if text.strip() else None @@ -325,15 +324,17 @@ def prune_pre_checkpoint_items( summary_remaining = max(0, int(retained_summary_token_budget)) seen_summary_texts: set = set() - def _try_retain_summary(text: Optional[str]) -> Optional[Dict[str, Any]]: - """Budget/dedup check for a summary; return its cost or None.""" + def _retain_summary(text: Optional[str], retained_item: Dict[str, Any]) -> None: + """Retain a summary whole when it fits the budget and is not a duplicate.""" + nonlocal summary_remaining if not text or summary_remaining <= 0 or text in seen_summary_texts: - return None + return cost = _approx_tokens(text) if cost > summary_remaining: - return None # never slice a summary's structural framing + return # never slice a summary's structural framing seen_summary_texts.add(text) - return {"cost": cost} + retained_reversed.append(retained_item) + summary_remaining -= cost for item, source in zip(reversed(pre), reversed(pre_sources)): if not isinstance(item, dict): @@ -343,15 +344,11 @@ def prune_pre_checkpoint_items( # when the source itself is a provenance-tagged summary carrier. if enable_summary_retention and isinstance(source, dict) and _is_summary_item(source): text = flatten_message_text(source.get("content")) - text = text if text.strip() else None - result = _try_retain_summary(text) - if result: - _src_role = source.get("role") - retained_reversed.append({ - "role": _src_role if _src_role in ("user", "assistant") else "assistant", - "content": text, - }) - summary_remaining -= result["cost"] + _src_role = source.get("role") + _retain_summary(text if text.strip() else None, { + "role": _src_role if _src_role in ("user", "assistant") else "assistant", + "content": text, + }) continue # Typed non-message items never carry role=user or a summary flag. @@ -372,10 +369,7 @@ def prune_pre_checkpoint_items( text = "" if is_summary: - result = _try_retain_summary(text) - if result: - retained_reversed.append(item) - summary_remaining -= result["cost"] + _retain_summary(text, item) elif is_user: if user_remaining <= 0: continue @@ -394,12 +388,8 @@ def prune_pre_checkpoint_items( logger.debug( "Pruned pre-checkpoint items: %d input -> %d retained (user_rem=%d, summary_rem=%d)", - len(items), - len(result), - user_remaining, - summary_remaining, + len(items), len(result), user_remaining, summary_remaining, ) - return result diff --git a/agent/prompt_cache_boundary.py b/agent/prompt_cache_boundary.py index 1d751e61b0..5635d78311 100644 --- a/agent/prompt_cache_boundary.py +++ b/agent/prompt_cache_boundary.py @@ -55,9 +55,12 @@ def find_stable_prefix(content: str) -> Optional[str]: with _lock: best: Optional[str] = None for prefix in _prefixes: - if content.startswith(prefix) and content[len(prefix):].strip(): - if best is None or len(prefix) > len(best): - best = prefix + if ( + content.startswith(prefix) + and content[len(prefix):].strip() + and (best is None or len(prefix) > len(best)) + ): + best = prefix if best is not None: _prefixes.move_to_end(best) # after the scan: never mutate mid-iteration return best diff --git a/agent/prompt_caching.py b/agent/prompt_caching.py index df857bb9eb..f2a04e7600 100644 --- a/agent/prompt_caching.py +++ b/agent/prompt_caching.py @@ -114,6 +114,8 @@ def _can_carry_marker( if content is None or content == "": return False if isinstance(content, list): + # Mirrors _apply_cache_marker (marks only the LAST part): a list whose + # last element isn't a dict cannot receive a marker. return bool(content) and isinstance(content[-1], dict) return isinstance(content, str) @@ -188,6 +190,9 @@ def effective_cache_ttl( if ttl != "1h": return ttl or "5m" if (provider or "").lower() in MEASURED_1H_PROVIDERS: + # Checked BEFORE the generic Qwen clamp (which would swallow every Qwen + # model on this route); the per-model denial stays nested so an + # opencode-go observation cannot reclamp the same model on another route. return "5m" if _flat_model(model) in NO_1H_TIER_MODELS else "1h" if is_qwen_model(model) or (provider or "").lower() in ALIBABA_FAMILY_PROVIDERS: return "5m" diff --git a/agent/skill_commands.py b/agent/skill_commands.py index b623af6bdf..a4ad829510 100644 --- a/agent/skill_commands.py +++ b/agent/skill_commands.py @@ -383,11 +383,12 @@ def _render_skill_block( loaded: tuple[dict[str, Any], Path | None, str], activation_note: str, task_id: str | None, + **message_kwargs: str, ) -> str: """Bump usage and build the message block for one loaded skill.""" loaded_skill, skill_dir, skill_name = loaded _bump_use(skill_name, task_id) - return _build_skill_message(loaded_skill, skill_dir, activation_note, session_id=task_id) + return _build_skill_message(loaded_skill, skill_dir, activation_note, session_id=task_id, **message_kwargs) def _scaffold_header( @@ -627,20 +628,13 @@ def build_skill_invocation_message( if not loaded: return None - loaded_skill, skill_dir, skill_name = loaded - _bump_use(skill_name, task_id) - - activation_note = ( - f'[IMPORTANT: The user has invoked the "{skill_name}" skill, indicating they want ' - "you to follow its instructions. The full skill content is loaded below.]" - ) - return _build_skill_message( - loaded_skill, - skill_dir, - activation_note, + return _render_skill_block( + loaded, + f'[IMPORTANT: The user has invoked the "{loaded[2]}" skill, indicating they want ' + "you to follow its instructions. The full skill content is loaded below.]", + task_id, user_instruction=user_instruction, runtime_note=runtime_note, - session_id=task_id, ) diff --git a/agent/transcript_repair.py b/agent/transcript_repair.py index 545818f5d9..0b790a1376 100644 --- a/agent/transcript_repair.py +++ b/agent/transcript_repair.py @@ -1,16 +1,12 @@ -"""Transcript repair and in-place row reconciliation helpers for SessionDB and run_agent. - -Extracted from hermes_state.py and run_agent.py to keep the godfiles narrow and bounded -under the 2K invariant (#95514 / PR #95886). Provides focused helpers to: -1. Resolve active assistant rows and watermark compaction clones in SQLite during batch appends. -2. In-place update blank assistant rows or adopt concurrent non-blank winner content without overwrite. -3. Synchronize in-memory message dicts with canonical committed content and row IDs after commit. +"""Transcript repair for SessionDB batch appends: reconcile in-memory assistant +rows with committed SQLite rows (blank-row in-place update, concurrent-winner +adoption, watermark-compaction clone lookup) and sync markers after commit. """ from __future__ import annotations import sqlite3 -from typing import Any, Callable, Dict, List, Optional +from typing import Any, Callable, Dict, List from agent.context_compressor import _DB_PERSISTED_MARKER @@ -24,12 +20,9 @@ def is_content_blank(content: Any) -> bool: if isinstance(content, list): if not content: return True - texts = [ - p.get("text", "") - for p in content - if isinstance(p, dict) and p.get("type") == "text" - ] - return not "".join(texts).strip() + return not "".join( + p.get("text", "") for p in content if isinstance(p, dict) and p.get("type") == "text" + ).strip() return False