fix(gateway): agent-cache pressure valve measures the cgroup's anon charge, not just the gateway's own RSS

The budget in _sweep_agent_cache_under_pressure comes from the gateway's own
cgroup memory.high/memory.max, but read_anon_rss_mb() read /proc/self/status
RssAnon: only the main process. Every child in the unit (execute_code kernels,
terminal commands) is charged against the same limit, so a kernel at 4.7 GiB
pushed the unit to MemoryHigh while the valve saw <1.6 GiB and never fired;
systemd's stop then SIGKILLed the gateway mid-flush (the #80764 signature).

Under a capped cgroup v2, read own memory.stat `anon` (the same scope as the
budget); uncapped, or when the file is unreadable, keep the self reading.
_finite_limit is factored out of _cgroup_limit_bytes so the cap check and
the budget agree on what "unlimited" means.

Fixes #110549
This commit is contained in:
teknium1
2026-09-14 22:14:45 -07:00
committed by Teknium
parent 645b9526a2
commit 7b44106174
4 changed files with 93 additions and 13 deletions

View File

@@ -54,13 +54,22 @@ def _positive(value: Any, cast: Callable[[Any], Any] = int) -> Any:
return parsed if parsed is not None and parsed > 0 else None
def _finite_limit(path: Path) -> Optional[int]:
"""A cgroup memory limit file's value when it is a real cap; None for unreadable, empty,
``max``, or the v1 near-2^63 sentinel (all mean unlimited)."""
try:
limit = int(path.read_text(encoding="utf-8").strip())
except (OSError, ValueError):
return None
return limit if 0 < limit < (1 << 62) else None
def _cgroup_limit_bytes() -> Optional[int]:
"""Memory limit this process runs under, if cgroup-capped.
Prefers v2 ``memory.high`` (the throttling point) over ``memory.max``, then v1.
Own cgroup first (where a systemd unit's ``MemoryHigh=``/``MemoryMax=`` lands —
root reads ``max`` there), then root for container-style limits. ``max`` and
the v1 near-2^63 sentinel mean unlimited.
root reads ``max`` there), then root for container-style limits.
"""
if sys.platform != "linux":
return None
@@ -72,11 +81,8 @@ def _cgroup_limit_bytes() -> Optional[int]:
own = None
roots = ([f"/sys/fs/cgroup{own}"] if own and own != "/" else []) + ["/sys/fs/cgroup"]
for candidate in [f"{r}/memory.{f}" for r in roots for f in ("high", "max")] + ["/sys/fs/cgroup/memory/memory.limit_in_bytes"]:
try:
limit = int(Path(candidate).read_text(encoding="utf-8").strip())
except (OSError, ValueError): # unreadable, empty, or "max"
continue
if 0 < limit < (1 << 62):
limit = _finite_limit(Path(candidate))
if limit is not None:
return limit
return None
@@ -133,9 +139,44 @@ def resolve_agent_cache_bounds(config: Any) -> AgentCacheBounds:
)
def _cgroup_anon_bytes() -> Optional[int]:
"""Anonymous memory charged to this process's own cgroup v2 (``memory.stat`` ``anon``), or None.
The budget is derived from the same cgroup's ``memory.high``/``memory.max``, and the kernel
charges every process in the unit against it — execute_code kernels, terminal children — so a
self-only reading under-counts by exactly the children's share (#110549). Anon, not
``memory.current``: the module's signal is heap, and reclaimable page cache is noise.
Only for a *capped* cgroup: an uncapped one (a plain login session) is the whole user
slice, and its budget is total RAM — self RSS stays the right scope there.
"""
if sys.platform != "linux":
return None
try:
from gateway.cgroup_cleanup import _own_cgroup_path
own = _own_cgroup_path()
if not own or own == "/":
return None
root = Path(f"/sys/fs/cgroup{own}")
capped = any(_finite_limit(root / f"memory.{f}") for f in ("high", "max"))
text = root.joinpath("memory.stat").read_text(encoding="utf-8") if capped else ""
except (OSError, ValueError, ImportError):
return None
for line in text.splitlines():
key, _, value = line.partition(" ")
if key == "anon" and value.strip().isdigit():
return int(value)
return None
def read_anon_rss_mb() -> Optional[int]:
"""Anonymous RSS in MB (where cached transcripts live; file-backed pages are noise),
or None. ``/proc/self/status`` first; psutil covers other platforms (total RSS only)."""
"""Anonymous memory in MB (where cached transcripts live; file-backed pages are noise),
or None. Own cgroup's ``memory.stat`` anon first — the scope the budget is charged
against, so same-unit child processes count; then ``/proc/self/status``; psutil covers
other platforms (total RSS only)."""
charged = _cgroup_anon_bytes()
if charged:
return charged // _BYTES_PER_MB
try:
from hermes_cli.mem_trim import collect_memory_snapshot

View File

@@ -98,6 +98,43 @@ class TestMemoryBudgetResolution:
assert resolve_memory_high_mb("auto") is None
class TestPressureSignalScope:
"""The budget is the unit's cgroup limit, so the signal must be the unit's anon charge:
an execute_code kernel in the same cgroup counts even while the gateway itself is small (#110549)."""
def _cgroup(self, monkeypatch, tmp_path, stat_text):
import gateway.agent_cache_pressure as acp
import gateway.cgroup_cleanup as cleanup
monkeypatch.setattr(acp.sys, "platform", "linux")
monkeypatch.setattr(cleanup, "_own_cgroup_path", lambda: "/hermes.service")
real_read_text = acp.Path.read_text
def read_text(self, *args, **kwargs):
if str(self) == "/sys/fs/cgroup/hermes.service/memory.high":
return "6979321856\n"
if str(self) == "/sys/fs/cgroup/hermes.service/memory.stat":
if stat_text is None:
raise OSError("restricted /sys mount")
return stat_text
return real_read_text(self, *args, **kwargs)
monkeypatch.setattr(acp.Path, "read_text", read_text)
return acp
def test_same_cgroup_child_anon_counts_against_the_budget(self, monkeypatch, tmp_path):
acp = self._cgroup(monkeypatch, tmp_path, "anon 4928307200\nfile 5426061312\nkernel 3629735936\n")
monkeypatch.setattr("hermes_cli.mem_trim.collect_memory_snapshot", lambda: {"rss_anon_kib": 1_600 * 1024})
assert acp.read_anon_rss_mb() == 4_700
def test_unreadable_cgroup_stat_keeps_the_self_reading(self, monkeypatch, tmp_path):
acp = self._cgroup(monkeypatch, tmp_path, None)
monkeypatch.setattr("hermes_cli.mem_trim.collect_memory_snapshot", lambda: {"rss_anon_kib": 1_600 * 1024})
assert acp.read_anon_rss_mb() == 1_600
class TestPersistenceGuard:
"""Soft eviction drops the transcript, so it may only run once the
transcript is durable. Exercised against the real AIAgent flush."""

View File

@@ -538,8 +538,10 @@ resident (agents that took a turn within the TTL are never idle-swept), so RSS c
the cgroup throttles and SIGTERM can no longer flush inside systemd's stop timeout
(#80764).
`_sweep_agent_cache_under_pressure()` is the valve. Each watcher tick it compares the
process's anonymous RSS against `memory_high_mb`; over budget, it evicts LRU agents
`_sweep_agent_cache_under_pressure()` is the valve. Each watcher tick it compares anonymous
memory against `memory_high_mb` — the cgroup's own `memory.stat` `anon` when the gateway runs
under a cgroup limit (the scope the budget is charged against, so same-unit children such as
`execute_code` kernels count; #110549), otherwise the process's own anonymous RSS; over budget, it evicts LRU agents
through the same soft path the cap enforcer uses (`_commit_then_release_soft`), then
runs `malloc_trim` so the freed arenas actually return to the OS. Evicted sessions
rebuild their transcript from the persisted session on the next turn.

View File

@@ -1123,9 +1123,9 @@ agent:
protect_recent: 8
```
`max_size` and `idle_ttl_secs` bound the cache by count and by time. Neither knows how many bytes it holds, so `memory_high_mb` adds a third bound: once the gateway's own anonymous resident memory crosses the budget, it sheds least-recently-used transcripts, which reload from the stored session on the next turn. Lower it if the gateway is competing for memory with other services; raise it (or set `0` to switch the pass off) if you would rather keep every prefix warm.
`max_size` and `idle_ttl_secs` bound the cache by count and by time. Neither knows how many bytes it holds, so `memory_high_mb` adds a third bound: once anonymous memory crosses the budget, it sheds least-recently-used transcripts, which reload from the stored session on the next turn. Lower it if the gateway is competing for memory with other services; raise it (or set `0` to switch the pass off) if you would rather keep every prefix warm.
`auto` derives the budget from the memory limit the gateway actually runs under — the cgroup limit for a container or systemd unit, total RAM otherwise — so a `MemoryMax`/`MemoryHigh` on the unit is respected without a second number to keep in sync.
`auto` derives the budget from the memory limit the gateway actually runs under — the cgroup limit for a container or systemd unit, total RAM otherwise — so a `MemoryMax`/`MemoryHigh` on the unit is respected without a second number to keep in sync. Under such a limit the measurement is scoped the same way: the cgroup's own anonymous charge (`memory.stat` `anon`), which includes child processes such as `execute_code` kernels and terminal commands that count against the unit's limit. Uncapped, the gateway's own anonymous RSS is measured.
Sessions that are mid-turn, the `protect_recent` most recently used ones, and any session whose transcript has not finished being written to disk are never shed. Eviction is logged at WARNING with the measured RSS and the sessions dropped: