fix(website): batch the plugin stars probe at 100 repos per GraphQL request
A single request for all 317 catalog repos now exceeds GitHub's per-query resource limit: the reply carries partial data plus an error, the script prints 'Probed 317', and every repo past the limit keeps its stale count (hindsight showed 26.5k stars while GitHub had 34.5k). Batch at 100; probed live: 317/317 repos in 4 requests, no errors.
This commit is contained in:
@@ -52,7 +52,7 @@ def test_deploy_reuses_the_cache_without_any_github_call(mod, tmp_path, monkeypa
|
|||||||
assert json.loads(out.read_text())["stars"] == {"a/one": 7}
|
assert json.loads(out.read_text())["stars"] == {"a/one": 7}
|
||||||
|
|
||||||
|
|
||||||
def test_probe_is_one_graphql_request_and_a_failure_keeps_previous_counts(mod, tmp_path, monkeypatch):
|
def test_probe_is_one_graphql_request_for_a_small_catalog_and_a_failure_keeps_previous_counts(mod, tmp_path, monkeypatch):
|
||||||
cat = _catalog(tmp_path, "https://github.com/a/one", "https://github.com/b/two", "https://gitlab.com/c/three")
|
cat = _catalog(tmp_path, "https://github.com/a/one", "https://github.com/b/two", "https://gitlab.com/c/three")
|
||||||
out = tmp_path / "plugin-stars.json"
|
out = tmp_path / "plugin-stars.json"
|
||||||
out.write_text(json.dumps({"fetched_at": "2026-01-01T00:00:00+00:00", "stars": {"a/one": 7, "b/two": 9}}),
|
out.write_text(json.dumps({"fetched_at": "2026-01-01T00:00:00+00:00", "stars": {"a/one": 7, "b/two": 9}}),
|
||||||
@@ -95,3 +95,22 @@ def test_failed_probe_keeps_the_previous_timestamp_and_warns(mod, tmp_path, monk
|
|||||||
data = json.loads(out.read_text())
|
data = json.loads(out.read_text())
|
||||||
assert data == {"fetched_at": "2026-09-16T18:40:16+00:00", "stars": {"a/one": 7}}
|
assert data == {"fetched_at": "2026-09-16T18:40:16+00:00", "stars": {"a/one": 7}}
|
||||||
assert "::warning::" in capsys.readouterr().out
|
assert "::warning::" in capsys.readouterr().out
|
||||||
|
|
||||||
|
|
||||||
|
def test_probe_batches_large_catalogs_and_every_repo_gets_a_count(mod, monkeypatch):
|
||||||
|
# One request for the whole catalog tripped GitHub's per-query resource limit at ~300
|
||||||
|
# repos and silently left the tail without counts; batches must cover every slug and the
|
||||||
|
# alias index must restart per request.
|
||||||
|
slugs = [f"o/r{i}" for i in range(mod._BATCH * 2 + 5)]
|
||||||
|
seen: list[int] = []
|
||||||
|
|
||||||
|
def per_batch(query, token):
|
||||||
|
n = query.count("repository(")
|
||||||
|
seen.append(n)
|
||||||
|
return {"data": {f"r{i}": {"stargazerCount": i} for i in range(n)}}
|
||||||
|
|
||||||
|
monkeypatch.setattr(mod, "_graphql", per_batch)
|
||||||
|
stars, probed = mod.probe_stars(slugs, {}, token="t")
|
||||||
|
assert probed and len(stars) == len(slugs)
|
||||||
|
assert seen == [mod._BATCH, mod._BATCH, 5]
|
||||||
|
assert stars[f"o/r{mod._BATCH}"] == 0 and stars[f"o/r{mod._BATCH + 1}"] == 1
|
||||||
|
|||||||
@@ -95,6 +95,9 @@ def load_previous(output: Path, live_url: str | None) -> dict:
|
|||||||
|
|
||||||
|
|
||||||
_GRAPHQL_URL = "https://api.github.com/graphql"
|
_GRAPHQL_URL = "https://api.github.com/graphql"
|
||||||
|
# Repository aliases per GraphQL request. One request for the whole catalog exceeded GitHub's
|
||||||
|
# per-query resource limit at ~300 repos: partial data + an error, stale counts kept silently.
|
||||||
|
_BATCH = 100
|
||||||
|
|
||||||
|
|
||||||
def _graphql(query: str, token: str) -> dict:
|
def _graphql(query: str, token: str) -> dict:
|
||||||
@@ -115,7 +118,7 @@ def stars_query(slugs: list[str]) -> str:
|
|||||||
|
|
||||||
|
|
||||||
def probe_stars(slugs: list[str], previous: dict[str, int], token: str | None) -> tuple[dict[str, int], bool]:
|
def probe_stars(slugs: list[str], previous: dict[str, int], token: str | None) -> tuple[dict[str, int], bool]:
|
||||||
"""One GraphQL request for all repos -> ``(stars, probed)``. On failure keep every previous
|
"""GraphQL probe in ``_BATCH``-sized requests -> ``(stars, probed)``. On failure keep every previous
|
||||||
count (never regress to 0) and report ``probed=False`` so the caller does not restamp
|
count (never regress to 0) and report ``probed=False`` so the caller does not restamp
|
||||||
``fetched_at`` over counts that are days old (#118113)."""
|
``fetched_at`` over counts that are days old (#118113)."""
|
||||||
kept = {s: previous[s] for s in slugs if s in previous}
|
kept = {s: previous[s] for s in slugs if s in previous}
|
||||||
@@ -124,21 +127,23 @@ def probe_stars(slugs: list[str], previous: dict[str, int], token: str | None) -
|
|||||||
if not token:
|
if not token:
|
||||||
_log("no GITHUB_TOKEN; keeping previous counts without probing")
|
_log("no GITHUB_TOKEN; keeping previous counts without probing")
|
||||||
return kept, False
|
return kept, False
|
||||||
try:
|
|
||||||
payload = _graphql(stars_query(slugs), token)
|
|
||||||
except (urllib.error.URLError, OSError, ValueError) as e:
|
|
||||||
_log(f"GraphQL probe failed ({e}); keeping previous counts")
|
|
||||||
return kept, False
|
|
||||||
data = payload.get("data") or {}
|
|
||||||
for err in payload.get("errors") or []:
|
|
||||||
_log(f"GraphQL: {err.get('message')}") # e.g. a renamed/deleted repo; its previous count is kept
|
|
||||||
stars: dict[str, int] = {}
|
stars: dict[str, int] = {}
|
||||||
for i, slug in enumerate(slugs):
|
for start in range(0, len(slugs), _BATCH):
|
||||||
node = data.get(f"r{i}")
|
batch = slugs[start:start + _BATCH]
|
||||||
if isinstance(node, dict) and isinstance(node.get("stargazerCount"), int):
|
try:
|
||||||
stars[slug] = node["stargazerCount"]
|
payload = _graphql(stars_query(batch), token)
|
||||||
elif slug in previous:
|
except (urllib.error.URLError, OSError, ValueError) as e:
|
||||||
stars[slug] = previous[slug]
|
_log(f"GraphQL probe failed ({e}); keeping previous counts")
|
||||||
|
return kept, False
|
||||||
|
data = payload.get("data") or {}
|
||||||
|
for err in payload.get("errors") or []:
|
||||||
|
_log(f"GraphQL: {err.get('message')}") # e.g. a renamed/deleted repo; its previous count is kept
|
||||||
|
for i, slug in enumerate(batch):
|
||||||
|
node = data.get(f"r{i}")
|
||||||
|
if isinstance(node, dict) and isinstance(node.get("stargazerCount"), int):
|
||||||
|
stars[slug] = node["stargazerCount"]
|
||||||
|
elif slug in previous:
|
||||||
|
stars[slug] = previous[slug]
|
||||||
return stars, True
|
return stars, True
|
||||||
|
|
||||||
|
|
||||||
@@ -167,7 +172,7 @@ def main(catalog_dir: Path = DEFAULT_CATALOG_DIR, output: Path = DEFAULT_OUTPUT,
|
|||||||
f"{missing} of {len(slugs)} catalog repos have no star count")
|
f"{missing} of {len(slugs)} catalog repos have no star count")
|
||||||
print(f"Probe failed; wrote {len(stars)} cached star counts (as of {fetched_at}) to {output}")
|
print(f"Probe failed; wrote {len(stars)} cached star counts (as of {fetched_at}) to {output}")
|
||||||
return 0
|
return 0
|
||||||
print(f"Probed {len(slugs)} repos in one GraphQL request, wrote {len(stars)} star counts to {output}")
|
print(f"Probed {len(slugs)} repos in {-(-len(slugs) // _BATCH)} GraphQL request(s), wrote {len(stars)} star counts to {output}")
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
|
||||||
@@ -176,7 +181,7 @@ if __name__ == "__main__":
|
|||||||
parser.add_argument("--catalog-dir", type=Path, default=DEFAULT_CATALOG_DIR)
|
parser.add_argument("--catalog-dir", type=Path, default=DEFAULT_CATALOG_DIR)
|
||||||
parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT)
|
parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT)
|
||||||
parser.add_argument("--probe", action="store_true",
|
parser.add_argument("--probe", action="store_true",
|
||||||
help="call GitHub (one GraphQL request); only the scheduled skills-index run does this")
|
help="call GitHub (batched GraphQL requests); only the scheduled skills-index run does this")
|
||||||
parser.add_argument("--no-live", action="store_true", help="do not consult the live site's cache")
|
parser.add_argument("--no-live", action="store_true", help="do not consult the live site's cache")
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
sys.exit(main(catalog_dir=args.catalog_dir, output=args.output, probe=args.probe,
|
sys.exit(main(catalog_dir=args.catalog_dir, output=args.output, probe=args.probe,
|
||||||
|
|||||||
Reference in New Issue
Block a user