Files
hermes-agent/agent/credential_pool_admin.py
teknium1 a564f77746 fix(auth): hermes auth reset survives a live session's next pool flush
A running gateway/chat holds the credential pool in memory. When the CLI
resets a still-binding cooldown from another process, the reset row on
disk has no status at all, so the recency merge in write_credential_pool
(which ranks by last_status_at) could not tell "reset after my cooldown"
from "never had a status" and let the live pool's stale EXHAUSTED entry
win on its next ordinary flush - a rotation, a token refresh, a sibling
429. `hermes auth reset` printed success and the cooldown came back
(#89415, step B in the thread). status_cleared_ids only protects the
resetting write itself.

reset_status/reset_statuses now stamp status_cleared_at on the cleared
row. The disk-cooldown merge adopts the disk row whenever the in-memory
entry is DEAD/EXHAUSTED and that marker postdates its last_status_at, and
_resync_stale_entry lifts the in-memory cooldown from the same marker, so
the live session serves the credential again without a restart. The
marker is sticky but only outranks OLDER statuses: a fresh exhaustion
after the reset stamps a newer last_status_at and still binds.

Live probe (temp HERMES_HOME, fake Codex entry, process A = live pool,
process B = real auth_reset_command): before, disk read `exhausted`
again after A's flush and A re-selected None; after, disk stays clear and
A re-selects the entry (mem ok). Control: reset then a new 402 stays
exhausted through flush and re-select.
2026-09-19 09:40:56 -07:00

118 lines
5.1 KiB
Python

"""Locked credential-pool administration and target resolution."""
from __future__ import annotations
import time
from dataclasses import replace
from typing import Any, Optional, Tuple, TYPE_CHECKING
if TYPE_CHECKING:
from agent.credential_pool import PooledCredential
def _cleared_status_copy(entry: PooledCredential) -> PooledCredential:
from agent.credential_pool import _CLEAR_STATUS
# The reset marker lets a live pool in another process tell "reset after my cooldown" from
# "never had a status" — both read as bare None on disk (#89415).
return replace(entry, **_CLEAR_STATUS, model_cooldowns=None, status_cleared_at=time.time(),
extra={k: v for k, v in entry.extra.items() if k != "failure_reason"})
class CredentialPoolAdminMixin:
def reset_status(self, credential_id: str) -> Optional[PooledCredential]:
"""Clear only the target's local error state, preserving sibling cooldowns."""
with self._lock:
entry = self._find(lambda e: e.id == credential_id)
if entry is None:
return None
cleared = _cleared_status_copy(entry)
self._replace_entry(entry, cleared)
self._persist(status_cleared_ids=[cleared.id])
return cleared
def reset_statuses(self) -> int:
"""Clear exhaustion state on every entry. Returns how many were cleared.
``failure_reason`` lives in ``extra``, not a dataclass field, so it is
stripped explicitly. The persist declares the cleared ids because the
disk-recency merge reads a cleared ``last_status_at`` (None -> epoch 0)
as a stale snapshot and would copy a still-binding cooldown back.
"""
from agent.credential_pool import _CLEAR_STATUS
with self._lock:
stale = [
e for e in self._entries
if e.last_status or e.last_status_at or e.last_error_code or e.failure_reason or e.model_cooldowns
]
if stale:
stale_ids = {e.id for e in stale}
self._entries = [
_cleared_status_copy(e) if e.id in stale_ids else e
for e in self._entries
]
self._persist(status_cleared_ids=list(stale_ids))
return len(stale)
def remove_index(self, index: int) -> Optional[PooledCredential]:
with self._lock:
if index < 1 or index > len(self._entries):
return None
removed = self._entries.pop(index - 1)
self._entries = [replace(e, priority=p) for p, e in enumerate(self._entries)]
self._persist(removed_ids=[removed.id])
if self._current_id == removed.id:
self._current_id = None
return removed
def move_entry(self, credential_id: str, priority: int) -> Optional[PooledCredential]:
"""Place an entry at a clamped zero-based position and persist contiguous priorities."""
from agent.credential_pool import _normalize_pool_priorities
with self._lock:
entry = self._find(lambda e: e.id == credential_id)
if entry is None:
return None
others = [e for e in self._entries if e.id != credential_id]
others.insert(max(0, min(int(priority), len(others))), entry)
entries = [replace(e, priority=p) for p, e in enumerate(others)]
# Apply load-time ordering now so the reported position survives reload.
_normalize_pool_priorities(self.provider, entries)
self._entries = sorted(entries, key=lambda e: e.priority)
self._persist()
return self._find(lambda e: e.id == credential_id)
def resolve_target(self, target: Any) -> Tuple[Optional[int], Optional[PooledCredential], Optional[str]]:
raw = str(target or "").strip()
if not raw:
return None, None, "No credential target provided."
with self._lock:
for idx, entry in enumerate(self._entries, start=1):
if entry.id == raw:
return idx, entry, None
label_matches = [
(idx, entry)
for idx, entry in enumerate(self._entries, start=1)
if entry.label.strip().lower() == raw.lower()
]
if len(label_matches) == 1:
return label_matches[0][0], label_matches[0][1], None
if len(label_matches) > 1:
return None, None, f'Ambiguous credential label "{raw}". Use the numeric index or entry id instead.'
if raw.isdigit():
index = int(raw)
if 1 <= index <= len(self._entries):
return index, self._entries[index - 1], None
return None, None, f"No credential #{index}."
return None, None, f'No credential matching "{raw}".'
def add_entry(self, entry: PooledCredential) -> PooledCredential:
from agent.credential_pool import _next_priority
with self._lock:
entry = replace(entry, priority=_next_priority(self._entries))
self._entries.append(entry)
self._persist()
return entry