From ada8e50f2dae4cce75a00bd3a06097dc8998ea38 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Mon, 7 Sep 2026 02:43:29 -0700 Subject: [PATCH] test: verify completion barriers and document local delivery proof --- evals/completion_backlog_probe.py | 26 +++++++++- tests/cli/test_completion_backlog.py | 7 ++- .../completion-backlog-delivery.md | 52 +++++++++++++++++++ website/sidebars.ts | 1 + 4 files changed, 83 insertions(+), 3 deletions(-) create mode 100644 website/docs/developer-guide/completion-backlog-delivery.md diff --git a/evals/completion_backlog_probe.py b/evals/completion_backlog_probe.py index 3e6bf61979..0eab6e78d1 100644 --- a/evals/completion_backlog_probe.py +++ b/evals/completion_backlog_probe.py @@ -99,6 +99,22 @@ def probe(surface, scenario, directory): time.sleep(.01) assert registry.completion_queue.qsize() == count raw = list(registry.completion_queue.queue) + delegation = None + if scenario == 'mixed': + from tools import async_delegation as ad + delegation = {'type': 'async_delegation', 'delegation_id': f'deleg-{surface}', + 'session_key': 'backlog-owner', 'goal': 'Fixture delegation', + 'status': 'completed', 'summary': 'DELEGATION_RESULT', + 'dispatched_at': time.time(), 'completed_at': time.time()} + ad._persist_dispatch(delegation) + ad._persist_completion(delegation, {'status': 'completed'}) + watch = {'type': 'watch_match', 'session_key': 'backlog-owner', + 'session_id': processes[0].id, 'pattern': 'READY', 'output': 'WATCH_READY'} + while not registry.completion_queue.empty(): + registry.completion_queue.get_nowait() + raw = raw[:4] + [watch] + raw[4:8] + [delegation] + raw[8:] + for event in raw: + registry.completion_queue.put(event) expected = [format_process_notification(event) for event in raw] if scenario == 'foreign': session['session_key'] = cli.session_id = 'another-owner' @@ -136,8 +152,16 @@ def probe(surface, scenario, directory): poller.join(10) assert not poller.is_alive() texts = [item['text'] for item in received] + delegation_delivered = None + if delegation: + with ad._transaction() as conn: + row = conn.execute("SELECT delivery_state, delivery_attempts FROM async_delegations WHERE delegation_id=?", + (delegation['delegation_id'],)).fetchone() + delegation_delivered = tuple(row) == ('delivered', 1) return {'surface': surface, 'scenario': scenario, 'children': count, 'wire_turns': len(texts), 'texts': texts, + 'delegation_delivered_once': delegation_delivered, + 'payload_order': [next((i for i, turn in enumerate(texts) if payload in turn), -1) for payload in expected], 'all_payloads_preserved': all(any(text in turn for turn in texts) for text in expected), 'single_exact': texts == expected if count == 1 else None, 'queue_remaining': registry.completion_queue.qsize(), @@ -162,7 +186,7 @@ def main(): os.environ['HOME'] = home results = [probe(surface, scenario, Path(home) / surface / scenario) for surface in ('cli', 'poller', 'post-turn') - for scenario in ('backlog', 'single', 'consumed', 'foreign')] + for scenario in ('backlog', 'single', 'consumed', 'foreign', 'mixed')] Path(args.output).write_text(json.dumps(results, indent=2), encoding="utf-8") print(json.dumps([{k: v for k, v in row.items() if k != 'texts'} for row in results], indent=2)) diff --git a/tests/cli/test_completion_backlog.py b/tests/cli/test_completion_backlog.py index 20f5e3e757..08ff1a5fe7 100644 --- a/tests/cli/test_completion_backlog.py +++ b/tests/cli/test_completion_backlog.py @@ -4,9 +4,12 @@ from evals.completion_backlog_probe import probe def test_ready_completions_share_one_turn_across_interactive_routes(tmp_path): for surface in ("cli", "poller", "post-turn"): - for scenario in ("backlog", "single"): + for scenario in ("backlog", "single", "mixed"): result = probe(surface, scenario, tmp_path / surface / scenario) - assert result["wire_turns"] == 1, result + assert result["wire_turns"] == (5 if scenario == "mixed" else 1), result + assert result["payload_order"] == sorted(result["payload_order"]), result + if scenario == "mixed": + assert result["delegation_delivered_once"], result assert result["all_payloads_preserved"], result if scenario == "single": assert result["single_exact"], result diff --git a/website/docs/developer-guide/completion-backlog-delivery.md b/website/docs/developer-guide/completion-backlog-delivery.md new file mode 100644 index 0000000000..8a516ce476 --- /dev/null +++ b/website/docs/developer-guide/completion-backlog-delivery.md @@ -0,0 +1,52 @@ +--- +title: Background completion backlogs +sidebar_label: Completion backlogs +--- + +# Background completion backlogs + +Interactive surfaces coalesce consecutive background-process completions that are +already ready for one conversation into a single notification turn. This does not +add a delay or promise to combine jobs that finish at different times. Failures and +successful outputs remain in the batch; one completion keeps its original text. + +Process identity remains available until dispatch. Explicit `process_manage` +wait/log/kill consumption can therefore suppress a CLI completion even after it +has left the process registry and entered the input queue. An entirely consumed +batch starts no turn. Watching output and async-delegation results remain separate +notifications, in their original order; they are not folded into completion batches. + +## Consumers and ownership + +- **Classic CLI:** `hermes_cli/cli_process_notifications.py` owns the idle/post-turn + drain, compression-aware ownership and final input unwrapping. +- **TUI and Desktop:** `tui_gateway/session_notifications.py` groups the poller's + ready snapshot after checking ownership. Each process still emits its own UI + status. Busy sessions requeue structured events, not rendered batch strings. +- **Post-turn TUI safety net:** `tui_gateway/prompt_turn.py` uses the same routing + and rendering path. Desktop and dashboard chat clients share this backend. +- **Messaging gateway:** `gateway/run_notifications.py` already uses its own + route-keyed short-window batching. This interactive-backlog change does not + replace that mechanism or alter adapter sends. +- **Noninteractive/headless consumers:** this change does not create a new + autonomous notification loop for an API request, ACP client, or one-shot CLI. + +The shared renderer is `tools/process_registry_notifications.py::ProcessNotificationBatch`. +Neither a batch nor its delivery status is persisted into the system prompt. +Addressed events still require a provable owner; another live session cannot adopt +them. Delegation delivery continues through its existing durable claim/complete +ledger, once per delegation rather than once per process batch. + +## Local validation and its limits + +`evals/completion_backlog_probe.py REPO OUTPUT.json` starts real local shell children +in temporary directories, reads their real completion events, and drives the +production CLI, TUI poller and post-turn notification routes. A loopback HTTP turn +sink replaces `chat` / `_run_prompt_submit`; it records actual dispatches but does +**not** exercise model inference, native renderer interaction or a hosted platform. +Synthetic watch and delegation envelopes are labeled fixtures; delegation claims +use the real temporary SQLite ledger. The backlog case includes a nonzero exit. + +The probe checks a ready backlog, a single exact payload, explicitly consumed +results, foreign ownership, and watches/delegations interleaved with completions. +It measures turn admission at the notification boundary, not model token savings. diff --git a/website/sidebars.ts b/website/sidebars.ts index fbe83d78c6..4bb9ca86bd 100644 --- a/website/sidebars.ts +++ b/website/sidebars.ts @@ -764,6 +764,7 @@ const sidebars: SidebarsConfig = { 'developer-guide/prompt-assembly', 'developer-guide/context-compression-and-caching', 'developer-guide/gateway-internals', + 'developer-guide/completion-backlog-delivery', 'developer-guide/session-storage', 'developer-guide/provider-runtime', 'developer-guide/programmatic-integration',