Files
hermes-agent/tools/todo_tool.py
kshitijk4poor 622a296f7a fix(agent): keep the todo predicate off the model_tools/executor import path
is_todo_tool_call lived in agent/tool_executor.py and went through
canonical_tool_name, which imports model_tools. TUI resume calls it from
_todo_state_from_history on the RPC path, so the first resume in a
gateway loaded ~405 modules (2-3s) synchronously. tui_gateway/server.py
and run_agent.py also imported agent.tool_executor at module level,
adding ~142 modules to every TUI/desktop launch and breaking run_agent's
lazy-forward rule.

The predicate now lives in tools/todo_tool.py, which both startup paths
already load. It matches TODO_TOOL_NAMES ({TODO_SCHEMA name} + the legacy
aliases) and imports the bridge parser only when a tool_call entry's
args mention "todo". model_tools._LEGACY_TOOL_ALIASES derives its todo
entry from TODO_LEGACY_ALIASES, so there is one source of truth ("todo"
is the only alias mapping to todo_list). The live tool.complete path in
tool_progress uses is_todo_tool_name and the hand-kept _TODO_TOOL_NAMES
tuple is gone. The server.py noqa import is replaced by a function-local
import next to MAX_TODO_RESULT_CHARS, so a pruned name can't be swallowed
by the broad except. run_agent imports lazily. The dead TypeError arm is
dropped, and field reads use message_sanitization._tc_field.
agent/tool_executor.py is back to its pre-stack state.

Co-authored-by: JoaoMarcos44 <joaomarcosdias444@gmail.com>
2026-09-27 18:34:12 +05:30

328 lines
15 KiB
Python

"""Todo tool: in-memory, revisioned task list for multi-step work. State lives on the
AIAgent (one per session), is re-injected after context compression, and every write bumps
a monotonic revision so UI clients can reject stale updates. One ``todo_list`` tool: pass
``todos`` to write, omit to read; every call returns the full list. No system-prompt mutation."""
import json
from typing import Any, Dict, List, Optional
VALID_STATUSES = {"pending", "in_progress", "completed", "cancelled"}
# The list is re-read after every compression (format_for_injection), so unbounded
# content/count would defeat the compression it rides through. Caps apply equally to
# model-authored items and caller-replayed API history.
MAX_TODO_CONTENT_CHARS = 4000
MAX_TODO_ITEMS = 256
# Max single todo tool-result payload accepted during history hydration, so a forged
# oversized result is dropped before parsing (AIAgent._hydrate_todo_store).
MAX_TODO_RESULT_CHARS = 512_000
_TRUNCATION_MARKER = "… [truncated]"
# Persisted as ordinary message content; ContextCompressor keys on this stable header to
# tell the synthetic post-compaction row from a real user message.
TODO_INJECTION_HEADER = "[Your active task list was preserved across context compression]"
_STATUS_MARKERS = {"completed": "[x]", "in_progress": "[>]", "pending": "[ ]", "cancelled": "[~]"}
_ACTIVE_STATUSES = {"pending", "in_progress"}
class TodoStore:
"""In-memory todo list, one per AIAgent. List position is priority; items are
``{id, content, status, parent?}`` — ``parent`` nests a subtask."""
def __init__(self):
self._items: List[Dict[str, str]] = []
self._revision = 0
def _fresh_items(self, todos: List[Dict[str, Any]]) -> List[Dict[str, str]]:
"""Validate, dedupe and order a whole new list (replace / restore)."""
return self._normalize_order([self._validate(t) for t in self._dedupe_by_id(todos)])
def write(self, todos: List[Dict[str, Any]], merge: bool = False) -> List[Dict[str, str]]:
"""Replace the list (default) or merge by id; returns the full list after writing."""
before = self.read()
if merge:
self._merge(todos)
else:
self._items = self._fresh_items(todos)
del self._items[MAX_TODO_ITEMS:] # keep the priority head; replays can't grow unbounded
self._sanitize_parents(self._items)
if self._items != before:
self._revision += 1
return self.read()
def _merge(self, todos: List[Dict[str, Any]]) -> None:
"""Update existing items only in the fields provided; append new ones (validated)."""
existing = {item["id"]: item for item in self._items}
for t in self._dedupe_by_id(todos):
item_id = str(t.get("id", "")).strip()
if not item_id:
continue # can't merge without an id
cur = existing.get(item_id)
if cur is None:
validated = self._validate(t)
existing[validated["id"]] = validated
self._items.append(validated)
continue
if t.get("content"):
cur["content"] = self._cap_content(str(t["content"]).strip())
if t.get("status") and str(t["status"]).strip().lower() in VALID_STATUSES:
cur["status"] = str(t["status"]).strip().lower()
if "parent" in t:
parent = str(t["parent"] or "").strip()
if parent:
cur["parent"] = parent
else:
cur.pop("parent", None)
# Rebuild preserving original order for existing items (first occurrence wins).
rebuilt = {item["id"]: existing.get(item["id"], item) for item in self._items}
self._items = self._normalize_order(list(rebuilt.values()))
def read(self) -> List[Dict[str, str]]:
return [item.copy() for item in self._items]
def has_items(self) -> bool:
return bool(self._items)
def snapshot(self) -> Dict[str, Any]:
"""Full state clients can reconcile atomically."""
return {"todos": self.read(), "revision": self._revision}
def restore(self, todos: List[Dict[str, Any]], *, revision: Any = 0) -> List[Dict[str, str]]:
"""Restore a trusted snapshot without manufacturing a new revision."""
self._items = self._fresh_items(todos)[:MAX_TODO_ITEMS]
try:
self._revision = max(0, int(revision or 0))
except (TypeError, ValueError):
self._revision = 0
return self.read()
def format_for_injection(self) -> Optional[str]:
"""Render the list for post-compression injection, or None if nothing active. Only
pending/in_progress items are injected — finished ones make the model re-do work after
compression. A parent is kept (with its real status marker) when any descendant is
active so subtasks keep context."""
if not self._items:
return None
children: Dict[str, List[Dict[str, str]]] = {}
for item in self._items:
if item.get("parent"):
children.setdefault(item["parent"], []).append(item)
def render(item: Dict[str, str], depth: int, out: List[str]) -> bool:
kid_lines: List[str] = []
has_active_kid = False
for kid in children.get(item["id"], []):
has_active_kid |= render(kid, depth + 1, kid_lines)
keep = item["status"] in _ACTIVE_STATUSES or has_active_kid
if keep:
marker = _STATUS_MARKERS.get(item["status"], "[?]")
out.append(f"{' ' * depth}- {marker} {item['id']}. "
f"{item['content']} ({item['status']})")
out.extend(kid_lines)
return keep
lines = [TODO_INJECTION_HEADER]
for item in self._items:
if not item.get("parent"):
render(item, 0, lines)
return "\n".join(lines) if len(lines) > 1 else None
@staticmethod
def _cap_content(content: str) -> str:
"""Truncate to MAX_TODO_CONTENT_CHARS keeping the head (the actionable part) + marker."""
if len(content) > MAX_TODO_CONTENT_CHARS:
return content[:MAX_TODO_CONTENT_CHARS - len(_TRUNCATION_MARKER)] + _TRUNCATION_MARKER
return content
@staticmethod
def _validate(item: Dict[str, Any]) -> Dict[str, str]:
"""Normalize one item to ``{id, content, status, parent?}`` (placeholders when missing)."""
if not isinstance(item, dict):
return {"id": "?", "content": "(invalid item)", "status": "pending"}
item_id = str(item.get("id", "")).strip() or "?"
content = str(item.get("content", "")).strip()
status = str(item.get("status", "pending")).strip().lower()
result = {"id": item_id,
"content": TodoStore._cap_content(content) if content else "(no description)",
"status": status if status in VALID_STATUSES else "pending"}
parent = str(item.get("parent") or "").strip()
if parent and parent != item_id:
result["parent"] = parent
return result
@staticmethod
def _sanitize_parents(items: List[Dict[str, str]]) -> None:
"""Drop dangling parent refs and break cycles in place (such items become roots)."""
by_id = {item["id"]: item for item in items}
for item in items:
if item.get("parent") and item["parent"] not in by_id:
item.pop("parent", None)
for item in items:
seen, node = {item["id"]}, item
while node.get("parent"):
if node["parent"] in seen:
item.pop("parent", None)
break
seen.add(node["parent"])
node = by_id[node["parent"]]
@staticmethod
def _dedupe_by_id(todos: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
"""Collapse duplicate ids, keeping the last occurrence in its position."""
last_index: Dict[str, int] = {}
for i, item in enumerate(todos): # non-dicts get a synthetic key; _validate handles them
key = str(item.get("id", "")).strip() if isinstance(item, dict) else f"__invalid_{i}"
last_index[key or "?"] = i
return [todos[i] for i in sorted(last_index.values())]
@staticmethod
def _normalize_order(items: List[Dict[str, str]]) -> List[Dict[str, str]]:
"""Lift the in_progress step ahead of any earlier pending placeholder. Nested lists
keep authored order — reordering would tear a subtask from its siblings."""
statuses = [item["status"] for item in items]
if any(item.get("parent") for item in items) or "in_progress" not in statuses:
return items
active_index = statuses.index("in_progress")
if "pending" not in statuses[:active_index]:
return items
normalized = items.copy()
normalized.insert(statuses.index("pending"), normalized.pop(active_index))
return normalized
def todo_tool(todos: Optional[List[Dict[str, Any]]] = None, merge: bool = False,
store: Optional[TodoStore] = None) -> str:
"""Write ``todos`` (replace, or ``merge`` by id) or read when None -> list + summary JSON."""
if store is None:
return tool_error("TodoStore not initialized")
if todos is None:
items = store.read()
else:
if isinstance(todos, str): # LLMs sometimes send a JSON string instead of a list
try:
todos = json.loads(todos)
except (json.JSONDecodeError, TypeError):
return tool_error("todos must be a list of objects, got unparseable string")
if not isinstance(todos, list):
return tool_error(f"todos must be a list, got {type(todos).__name__}")
items = store.write(todos, merge)
summary = {"total": len(items)}
for status in ("pending", "in_progress", "completed", "cancelled"):
summary[status] = sum(1 for i in items if i["status"] == status)
return json.dumps({"todos": items, "revision": store.snapshot()["revision"],
"summary": summary}, ensure_ascii=False)
def check_todo_requirements() -> bool:
"""Todo tool has no external requirements -- always available."""
return True
# Behavioral guidance is baked into the (static, cached) description; item shape and merge
# semantics live ONLY in the parameter schema.
TODO_SCHEMA = {
"name": "todo_list",
"description": (
# See #95681.
"Track a task list for multi-step work (3+ steps). Use for complex tasks "
"with 3+ steps or when the user provides multiple tasks. "
"For 'all N items' tasks, enumerate every instance as its own checklist "
"item so none are silently dropped. "
"Call with no parameters to read the current list.\n"
"List order is priority. Only ONE item in_progress at a time. "
"Break large phases into subtasks via parent. "
"Mark an item completed only after the work is verified done, never "
"based on intent. If something fails, cancel it and add a revised "
"item. Always returns the full current list."
),
"parameters": {
"type": "object",
"properties": {
"todos": {
"type": "array",
"description": "Task items to write.",
"items": {
"type": "object",
"properties": {
"id": {
"type": "string"
},
"content": {
"type": "string",
"description": "Task description"
},
"status": {
"type": "string",
"enum": ["pending", "in_progress", "completed", "cancelled"]
},
"parent": {
"type": "string",
"description": "Optional id of another item, making this a nested subtask. Omit for top-level."
}
},
"required": ["id", "content", "status"]
}
},
"merge": {
"type": "boolean",
"description": (
"true: update existing items by id, add new ones. "
"false (default): replace the entire list with a fresh plan."
),
"default": False
}
},
"required": []
}
}
# Pre-rename names that replay as the Todo tool. model_tools._LEGACY_TOOL_ALIASES derives its todo
# entries from this, so the alias map and the transcript/TUI predicates below cannot drift.
TODO_LEGACY_ALIASES = ("todo",)
TODO_TOOL_NAMES = frozenset((TODO_SCHEMA["name"], *TODO_LEGACY_ALIASES))
def is_todo_tool_name(name: Any) -> bool:
"""True for the Todo tool's current name or a legacy alias (an already-unwrapped dispatch name)."""
return name in TODO_TOOL_NAMES
def is_todo_tool_call(tool_call: Any) -> bool:
"""True when a transcript tool_call entry (dict or object) invoked the Todo tool.
Covers the current name, legacy aliases, and the ``tool_call`` bridge (``todo_list`` is deferred by
default, and the transcript keeps the bridge name). The bridge is peeled from the recorded arguments
only, never live tool-search config, and must wrap exactly one call. Lives here, not in
agent.tool_executor, so TUI resume and run_agent never pull model_tools / the executor in to answer it.
"""
from agent.message_sanitization import _tc_field
fn = _tc_field(tool_call, "function")
name, raw_args = _tc_field(fn, "name") or "", _tc_field(fn, "arguments")
if is_todo_tool_name(name):
return True
# Cheap pre-check before the bridge modules load: no "todo" in the raw args means no todo inside.
if isinstance(raw_args, str) and "todo" not in raw_args:
return False
from tools.tool_search_catalog import TOOL_CALL_NAME
if name != TOOL_CALL_NAME:
return False
try:
args = json.loads(raw_args) if isinstance(raw_args, str) else raw_args
except json.JSONDecodeError:
return False
if not isinstance(args, dict):
return False
from tools.tool_search_validation import normalize_tool_call_entries
entries, error = normalize_tool_call_entries(args)
return not error and len(entries) == 1 and is_todo_tool_name(entries[0]["name"])
from tools.registry import registry, tool_error
registry.register(
name="todo_list", toolset="todo", schema=TODO_SCHEMA, check_fn=check_todo_requirements,
handler=lambda args, **kw: todo_tool(
todos=args.get("todos"), merge=args.get("merge", False), store=kw.get("store")),
emoji="📋")