From 65728d7ef1ea943c25b78e8b6fd7793770aa6e0d Mon Sep 17 00:00:00 2001 From: ethernet Date: Tue, 22 Sep 2026 20:08:52 -0400 Subject: [PATCH] fix(providers): publish home layer before auth sync --- .github/workflows/tests.yml | 14 +++++----- agent/moa_loop.py | 29 ++++++++++++++------ providers/__init__.py | 7 +++-- tests/hermes_platform/test_declaration.py | 2 +- tests/hermes_platform/test_facts.py | 6 ++-- tests/hermes_platform/test_resolver_core.py | 12 ++++---- tests/tui_gateway/test_profiles_configure.py | 2 +- 7 files changed, 43 insertions(+), 29 deletions(-) diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 1c4c813485..d638ae574e 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -76,7 +76,8 @@ jobs: env: # This is the maximum number of test FILES that run together. # run_tests_parallel.py starts one pytest subprocess for each file - # from a single ThreadPoolExecutor, so this value IS the limit. + # from a single ThreadPoolExecutor, so this value IS the limit. The + # default is cpu_count*2, which is 192 here. # # Measured on this runner (96-core EPYC 7763, 377GB). Whole suite, # two repetitions for each value. See run 32549672063: @@ -89,12 +90,11 @@ jobs: # 240 2.5x 140s # 288 3.0x 142s # - # The suite has since grown to more than 5,000 files. At 96 workers, - # two consecutive runs left dozens of first-wave files unable even - # to collect before the five-minute cap. Use the measured half-core - # point: it was only 12s behind the old winner and avoids launching - # one cold pytest import per core at once. - HERMES_TEST_WORKERS: 48 + # One worker for each core wins. The curve is shallow: 126s to 142s + # across a 6x range. The suite has sufficient concurrency at this + # size. The remaining time is the slowest files plus the setup. + # Workers above the core count only add contention. + HERMES_TEST_WORKERS: 96 # Ensure tests don't accidentally call real APIs OPENROUTER_API_KEY: "" OPENAI_API_KEY: "" diff --git a/agent/moa_loop.py b/agent/moa_loop.py index 031c530d6b..ccce40a741 100644 --- a/agent/moa_loop.py +++ b/agent/moa_loop.py @@ -496,6 +496,7 @@ def _trim_messages_for_reference( _REFERENCE_POLL_INTERVAL_S = 5.0 +_REFERENCE_INTERRUPT_SETTLE_S = 0.05 # Sentinel for a reference aborted by user interrupt; the facade must never cache it. _INTERRUPTED_REFERENCE_NOTE = "[skipped: interrupted by user]" @@ -566,6 +567,19 @@ def _run_references_parallel( # Shared per-fan-out context-length cache (dict get/set is GIL-atomic). ctx_len_cache: dict[tuple[str, str], int | None] = {} cache_disabled, cache_ttl = _agent_cache_opts(agent) + + def collect(done: set[Any]) -> None: + nonlocal completed + for future in done: + idx = futures[future] + results[idx] = future.result() + completed += 1 + if progress_callback is not None: + try: + progress_callback(completed, total, _slot_label(reference_models[idx])) + except Exception as exc: # pragma: no cover - display must never break + logger.debug("MoA progress_callback failed: %s", exc) + try: for idx, slot in enumerate(reference_models): if slot.get("provider") == "moa": @@ -581,16 +595,13 @@ def _run_references_parallel( pending = set(futures) while pending: done, pending = _futures_wait(pending, timeout=_REFERENCE_POLL_INTERVAL_S) - for future in done: - idx = futures[future] - results[idx] = future.result() - completed += 1 - if progress_callback is not None: - try: - progress_callback(completed, total, _slot_label(reference_models[idx])) - except Exception as exc: # pragma: no cover - display must never break - logger.debug("MoA progress_callback failed: %s", exc) + collect(done) if pending and agent is not None and getattr(agent, "_interrupt_requested", False): + # A worker can raise the interrupt immediately before publishing + # its own result. Give concurrently-finishing work one scheduler + # turn so completed output is not replaced by an interrupt note. + done, pending = _futures_wait(pending, timeout=_REFERENCE_INTERRUPT_SETTLE_S) + collect(done) interrupted = True _settle_interrupted(futures, results, reference_models, late_accounting_sink) break diff --git a/providers/__init__.py b/providers/__init__.py index 92f309f1a5..1827fac013 100644 --- a/providers/__init__.py +++ b/providers/__init__.py @@ -231,6 +231,11 @@ def _home_layer() -> _HomeLayer: if home is not None and (stamps := _plugin_dir_stamps(home)) != layer.stamps: _scan_home_layer(layer, key) layer.stamps = stamps + # Publish the completed layer only after its stamp is current. Auth sync + # calls list_providers(), which re-enters this function; syncing from + # _scan_home_layer before this assignment recursively rescanned forever. + if _discovered and not _discovering: + _sync_auth_registry() return layer @@ -333,8 +338,6 @@ def _scan_home_layer(layer: _HomeLayer, key: str) -> None: finally: _REGISTRATION_TARGET.reset(token) _discovering = prior_discovering - if _discovered and not _discovering: - _sync_auth_registry() def _user_module_name(plugin_dir: Path, home_key: str) -> str: diff --git a/tests/hermes_platform/test_declaration.py b/tests/hermes_platform/test_declaration.py index 26084a21a5..54c823d0c7 100644 --- a/tests/hermes_platform/test_declaration.py +++ b/tests/hermes_platform/test_declaration.py @@ -85,7 +85,7 @@ def test_availability_unsupported_os_when_no_block_for_host(): assert availability(decl, os_family="darwin").state == "unsupported_os" -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_availability_missing_then_present_then_version_gate(tmp_path): location = tmp_path / "Applications" / "Thing.app" decl = parse_declaration( diff --git a/tests/hermes_platform/test_facts.py b/tests/hermes_platform/test_facts.py index 0741f8aea1..66ab8d9b9c 100644 --- a/tests/hermes_platform/test_facts.py +++ b/tests/hermes_platform/test_facts.py @@ -74,7 +74,7 @@ def cleared_fact_caches(): facts.clear_caches() -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_windows_native_arch_matches_registry_identifier(cleared_fact_caches) -> None: winreg = importlib.import_module("winreg") with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, facts._CPU_KEY) as key: @@ -84,13 +84,13 @@ def test_windows_native_arch_matches_registry_identifier(cleared_fact_caches) -> assert facts.native_arch() == expected -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_macos_live_cpu_facts(cleared_fact_caches) -> None: assert facts.cpu_model() assert facts.native_arch() in {"arm64", "amd64"} -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_linux_live_cpu_facts_match_cpuinfo(cleared_fact_caches) -> None: cpuinfo = Path("/proc/cpuinfo").read_text(encoding="utf-8", errors="replace") diff --git a/tests/hermes_platform/test_resolver_core.py b/tests/hermes_platform/test_resolver_core.py index 1049945e67..4c926c0123 100644 --- a/tests/hermes_platform/test_resolver_core.py +++ b/tests/hermes_platform/test_resolver_core.py @@ -16,7 +16,7 @@ def _make_exe(path): return path -@pytest.mark.skipif(sys.platform == "win32", reason="POSIX executable bits") +@pytest.mark.platforms("posix") def test_path_hit_comes_first_and_known_dirs_are_still_recorded(tmp_path): on_path = _make_exe(tmp_path / "pathbin" / "tool") in_known = _make_exe(tmp_path / "known" / "tool") @@ -27,7 +27,7 @@ def test_path_hit_comes_first_and_known_dirs_are_still_recorded(tmp_path): assert [c.value for c in res.present] == [str(on_path), str(in_known)] -@pytest.mark.skipif(sys.platform == "win32", reason="POSIX executable bits") +@pytest.mark.platforms("posix") def test_known_dir_hit_when_path_misses(tmp_path): in_known = _make_exe(tmp_path / "known" / "tool") res = locate_command("tool", LookupContext(path=""), known_dirs=(str(in_known.parent),)) @@ -50,7 +50,7 @@ def test_absent_and_none_both_mean_ambient(): assert LookupContext(path="/x").effective_path() == "/x" -@pytest.mark.skipif(sys.platform == "win32", reason="POSIX executable bits") +@pytest.mark.platforms("posix") def test_explicit_path_bypasses_search(tmp_path): exe = _make_exe(tmp_path / "bin" / "tool") res = locate_command(str(exe), LookupContext(path="")) @@ -60,7 +60,7 @@ def test_explicit_path_bypasses_search(tmp_path): assert missing.kind == "missing" and missing.candidates[0].source == "explicit" -@pytest.mark.skipif(sys.platform == "win32", reason="POSIX executable bits") +@pytest.mark.platforms("posix") def test_locate_never_searches_the_working_directory(tmp_path, monkeypatch): _make_exe(tmp_path / "tool") monkeypatch.chdir(tmp_path) @@ -69,7 +69,7 @@ def test_locate_never_searches_the_working_directory(tmp_path, monkeypatch): assert relative.kind == "missing" and relative.candidates[0].present is False -@pytest.mark.skipif(sys.platform == "win32", reason="POSIX executable bits") +@pytest.mark.platforms("posix") def test_known_dir_expands_home_and_env(tmp_path, monkeypatch): exe = _make_exe(tmp_path / "home" / ".local" / "bin" / "tool") monkeypatch.setenv("HOME", str(tmp_path / "home")) @@ -79,7 +79,7 @@ def test_known_dir_expands_home_and_env(tmp_path, monkeypatch): assert res.command == (str(exe),), spec -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_windows_pathext_is_honored_without_mutating_environ(tmp_path, monkeypatch): exe = tmp_path / "bin" / "tool.cmd" exe.parent.mkdir() diff --git a/tests/tui_gateway/test_profiles_configure.py b/tests/tui_gateway/test_profiles_configure.py index 97ebfc74c9..95bfe250f3 100644 --- a/tests/tui_gateway/test_profiles_configure.py +++ b/tests/tui_gateway/test_profiles_configure.py @@ -11,7 +11,7 @@ from __future__ import annotations from pathlib import Path import pytest -import yaml +import hermes_yaml as yaml import tui_gateway.server as server