diff --git a/tests/website/test_fetch_plugin_stars.py b/tests/website/test_fetch_plugin_stars.py index 32b8c31e38..2da7f54a2d 100644 --- a/tests/website/test_fetch_plugin_stars.py +++ b/tests/website/test_fetch_plugin_stars.py @@ -52,7 +52,7 @@ def test_deploy_reuses_the_cache_without_any_github_call(mod, tmp_path, monkeypa assert json.loads(out.read_text())["stars"] == {"a/one": 7} -def test_probe_is_one_graphql_request_and_a_failure_keeps_previous_counts(mod, tmp_path, monkeypatch): +def test_probe_is_one_graphql_request_for_a_small_catalog_and_a_failure_keeps_previous_counts(mod, tmp_path, monkeypatch): cat = _catalog(tmp_path, "https://github.com/a/one", "https://github.com/b/two", "https://gitlab.com/c/three") out = tmp_path / "plugin-stars.json" out.write_text(json.dumps({"fetched_at": "2026-01-01T00:00:00+00:00", "stars": {"a/one": 7, "b/two": 9}}), @@ -95,3 +95,22 @@ def test_failed_probe_keeps_the_previous_timestamp_and_warns(mod, tmp_path, monk data = json.loads(out.read_text()) assert data == {"fetched_at": "2026-09-16T18:40:16+00:00", "stars": {"a/one": 7}} assert "::warning::" in capsys.readouterr().out + + +def test_probe_batches_large_catalogs_and_every_repo_gets_a_count(mod, monkeypatch): + # One request for the whole catalog tripped GitHub's per-query resource limit at ~300 + # repos and silently left the tail without counts; batches must cover every slug and the + # alias index must restart per request. + slugs = [f"o/r{i}" for i in range(mod._BATCH * 2 + 5)] + seen: list[int] = [] + + def per_batch(query, token): + n = query.count("repository(") + seen.append(n) + return {"data": {f"r{i}": {"stargazerCount": i} for i in range(n)}} + + monkeypatch.setattr(mod, "_graphql", per_batch) + stars, probed = mod.probe_stars(slugs, {}, token="t") + assert probed and len(stars) == len(slugs) + assert seen == [mod._BATCH, mod._BATCH, 5] + assert stars[f"o/r{mod._BATCH}"] == 0 and stars[f"o/r{mod._BATCH + 1}"] == 1 diff --git a/website/scripts/fetch-plugin-stars.py b/website/scripts/fetch-plugin-stars.py index a7263dadb5..687e06ecee 100644 --- a/website/scripts/fetch-plugin-stars.py +++ b/website/scripts/fetch-plugin-stars.py @@ -95,6 +95,9 @@ def load_previous(output: Path, live_url: str | None) -> dict: _GRAPHQL_URL = "https://api.github.com/graphql" +# Repository aliases per GraphQL request. One request for the whole catalog exceeded GitHub's +# per-query resource limit at ~300 repos: partial data + an error, stale counts kept silently. +_BATCH = 100 def _graphql(query: str, token: str) -> dict: @@ -115,7 +118,7 @@ def stars_query(slugs: list[str]) -> str: def probe_stars(slugs: list[str], previous: dict[str, int], token: str | None) -> tuple[dict[str, int], bool]: - """One GraphQL request for all repos -> ``(stars, probed)``. On failure keep every previous + """GraphQL probe in ``_BATCH``-sized requests -> ``(stars, probed)``. On failure keep every previous count (never regress to 0) and report ``probed=False`` so the caller does not restamp ``fetched_at`` over counts that are days old (#118113).""" kept = {s: previous[s] for s in slugs if s in previous} @@ -124,21 +127,23 @@ def probe_stars(slugs: list[str], previous: dict[str, int], token: str | None) - if not token: _log("no GITHUB_TOKEN; keeping previous counts without probing") return kept, False - try: - payload = _graphql(stars_query(slugs), token) - except (urllib.error.URLError, OSError, ValueError) as e: - _log(f"GraphQL probe failed ({e}); keeping previous counts") - return kept, False - data = payload.get("data") or {} - for err in payload.get("errors") or []: - _log(f"GraphQL: {err.get('message')}") # e.g. a renamed/deleted repo; its previous count is kept stars: dict[str, int] = {} - for i, slug in enumerate(slugs): - node = data.get(f"r{i}") - if isinstance(node, dict) and isinstance(node.get("stargazerCount"), int): - stars[slug] = node["stargazerCount"] - elif slug in previous: - stars[slug] = previous[slug] + for start in range(0, len(slugs), _BATCH): + batch = slugs[start:start + _BATCH] + try: + payload = _graphql(stars_query(batch), token) + except (urllib.error.URLError, OSError, ValueError) as e: + _log(f"GraphQL probe failed ({e}); keeping previous counts") + return kept, False + data = payload.get("data") or {} + for err in payload.get("errors") or []: + _log(f"GraphQL: {err.get('message')}") # e.g. a renamed/deleted repo; its previous count is kept + for i, slug in enumerate(batch): + node = data.get(f"r{i}") + if isinstance(node, dict) and isinstance(node.get("stargazerCount"), int): + stars[slug] = node["stargazerCount"] + elif slug in previous: + stars[slug] = previous[slug] return stars, True @@ -167,7 +172,7 @@ def main(catalog_dir: Path = DEFAULT_CATALOG_DIR, output: Path = DEFAULT_OUTPUT, f"{missing} of {len(slugs)} catalog repos have no star count") print(f"Probe failed; wrote {len(stars)} cached star counts (as of {fetched_at}) to {output}") return 0 - print(f"Probed {len(slugs)} repos in one GraphQL request, wrote {len(stars)} star counts to {output}") + print(f"Probed {len(slugs)} repos in {-(-len(slugs) // _BATCH)} GraphQL request(s), wrote {len(stars)} star counts to {output}") return 0 @@ -176,7 +181,7 @@ if __name__ == "__main__": parser.add_argument("--catalog-dir", type=Path, default=DEFAULT_CATALOG_DIR) parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT) parser.add_argument("--probe", action="store_true", - help="call GitHub (one GraphQL request); only the scheduled skills-index run does this") + help="call GitHub (batched GraphQL requests); only the scheduled skills-index run does this") parser.add_argument("--no-live", action="store_true", help="do not consult the live site's cache") args = parser.parse_args() sys.exit(main(catalog_dir=args.catalog_dir, output=args.output, probe=args.probe,