feat(plugin-catalog): rank entries by GitHub stars, probed at most once a day

Catalog entries sort official → stars desc → name, both in browse shelves and
filtered grids, with a ★ pill on each card linking to the repo's stargazers.

Rate-limit discipline is the design constraint: the docs site deploys many
times a day and shares one GitHub App API budget with every other workflow
(tonight's merge train got rate-limited on unrelated uploads). So
website/scripts/fetch-plugin-stars.py first fetches the live site's own
plugin-stars.json (a CDN GET, not the API); if that cache is under 24h old it
is reused verbatim and GitHub is never called. Only a stale cache triggers one
GET /repos/{owner}/{repo} per unique catalog repo, and a 403/429 mid-run keeps
the previous counts instead of zeroing them. extract-plugins.py merges the
cache into plugins.json (`stars`) and plugins-meta.json (`starsFetchedAt`), and
the page footnote says when the ranking was last refreshed.
This commit is contained in:
teknium1
2026-09-14 21:24:33 -07:00
parent 437116f949
commit 3e2e2c50eb
9 changed files with 361 additions and 11 deletions

View File

@@ -164,6 +164,14 @@ jobs:
- name: Extract skill metadata for dashboard
run: python3 website/scripts/extract-skills.py
# Star counts drive the catalog ranking. The script reuses the live site's cache when it
# is < 24h old (one CDN GET, zero API calls); only a stale cache triggers ~1 API call per
# catalog repo. Never touches the API on the many same-day deploys.
- name: Refresh plugin GitHub stars (daily cache)
env:
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
run: python3 website/scripts/fetch-plugin-stars.py
- name: Extract plugin catalog for the Plugins page
run: python3 website/scripts/extract-plugins.py

1
.gitignore vendored
View File

@@ -137,6 +137,7 @@ website/static/api/skills-meta.json
# plugins.json + plugins-meta.json are build artifacts emitted by
# website/scripts/extract-plugins.py during prebuild (Plugin Catalog page).
website/static/api/plugins.json
website/static/api/plugin-stars.json
website/static/api/plugin-catalog.json
website/static/api/plugins-meta.json
# automation-blueprints-index.json is a build artifact emitted by

View File

@@ -142,6 +142,12 @@ def test_main_writes_catalog_and_meta(mod, tmp_path):
catalog.mkdir()
_write_entry(catalog, "alpha", tier="official", category="memory")
_write_entry(catalog, "beta") # no category → default "desktop" shelf
_write_entry(catalog, "gamma")
# Star cache from fetch-plugin-stars.py: gamma outranks beta within the community tier.
(tmp_path / "api").mkdir()
(tmp_path / "api" / "plugin-stars.json").write_text(json.dumps({
"fetched_at": "2026-09-15T00:00:00+00:00",
"stars": {"example/gamma": 50, "example/beta": 3}}), encoding="utf-8")
(catalog / "removed.yaml").write_text(
"removed:\n - name: gone\n", encoding="utf-8"
)
@@ -152,17 +158,19 @@ def test_main_writes_catalog_and_meta(mod, tmp_path):
assert rc == 0
plugins = json.loads((out_dir / "plugins.json").read_text(encoding="utf-8"))
meta = json.loads((out_dir / "plugins-meta.json").read_text(encoding="utf-8"))
assert [p["name"] for p in plugins] == ["alpha", "beta"]
assert meta["total"] == 2
assert meta["byTier"] == {"official": 1, "community": 1}
assert {p["name"]: p["category"] for p in plugins} == {"alpha": "memory", "beta": "desktop"}
assert meta["byCategory"] == {"desktop": 1, "memory": 1}
assert [p["name"] for p in plugins] == ["alpha", "gamma", "beta"] # official first, then stars desc
assert {p["name"]: p["stars"] for p in plugins} == {"alpha": None, "gamma": 50, "beta": 3}
assert meta["total"] == 3
assert meta["byTier"] == {"official": 1, "community": 2}
assert {p["name"]: p["category"] for p in plugins} == {"alpha": "memory", "beta": "desktop", "gamma": "desktop"}
assert meta["byCategory"] == {"desktop": 2, "memory": 1}
assert meta["starsFetchedAt"] == "2026-09-15T00:00:00+00:00"
assert meta["removedCount"] == 1
assert meta["generatedAt"]
# The live-refresh document consumed by installed clients: loader-schema entries + the kill list.
from hermes_cli.plugin_catalog import entry_from_mapping
live = json.loads((out_dir / "plugin-catalog.json").read_text(encoding="utf-8"))
assert [entry_from_mapping(raw, "live").name for raw in live["entries"]] == ["alpha", "beta"]
assert [entry_from_mapping(raw, "live").name for raw in live["entries"]] == ["alpha", "beta", "gamma"]
assert live["removed"] == [{"name": "gone"}]

View File

@@ -0,0 +1,75 @@
"""fetch-plugin-stars.py: the daily GitHub-stars cache behind catalog ranking.
The contract under test is rate-limit discipline, not the numbers: a fresh cache must never
reach GitHub, and a rate-limited probe must keep the previous counts rather than zeroing them.
"""
from __future__ import annotations
import importlib.util
import json
import urllib.error
from datetime import datetime, timedelta, timezone
from pathlib import Path
import pytest
REPO_ROOT = Path(__file__).resolve().parents[2]
SCRIPT = REPO_ROOT / "website" / "scripts" / "fetch-plugin-stars.py"
@pytest.fixture(scope="module")
def mod():
spec = importlib.util.spec_from_file_location("fetch_plugin_stars", SCRIPT)
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
def _catalog(tmp_path: Path, *repos: str) -> Path:
import yaml
cat = tmp_path / "plugin-catalog"
cat.mkdir()
for i, repo in enumerate(repos):
(cat / f"p{i}.yaml").write_text(yaml.safe_dump({
"name": f"p{i}", "repo": repo, "sha": "38fe0fb53eff98d477f807432e965429e665ca33",
"description": "d", "maintainer": "m"}), encoding="utf-8")
return cat
def test_fresh_cache_is_reused_without_any_github_call(mod, tmp_path, monkeypatch):
cat = _catalog(tmp_path, "https://github.com/a/one")
out = tmp_path / "plugin-stars.json"
recent = (datetime.now(timezone.utc) - timedelta(hours=2)).isoformat()
out.write_text(json.dumps({"fetched_at": recent, "stars": {"a/one": 7}}), encoding="utf-8")
def boom(*a, **k):
raise AssertionError("GitHub must not be called while the cache is fresh")
monkeypatch.setattr(mod, "_http_json", boom)
assert mod.main(catalog_dir=cat, output=out, max_age_hours=24, live_url=None) == 0
assert json.loads(out.read_text())["stars"] == {"a/one": 7}
def test_stale_cache_probes_and_rate_limit_keeps_previous_counts(mod, tmp_path, monkeypatch):
cat = _catalog(tmp_path, "https://github.com/a/one", "https://github.com/b/two", "https://gitlab.com/c/three")
out = tmp_path / "plugin-stars.json"
old = (datetime.now(timezone.utc) - timedelta(days=3)).isoformat()
out.write_text(json.dumps({"fetched_at": old, "stars": {"a/one": 7, "b/two": 9}}), encoding="utf-8")
calls: list[str] = []
def fake(url, headers, timeout=15.0):
calls.append(url)
if url.endswith("/repos/a/one"):
return {"stargazers_count": 42}
raise urllib.error.HTTPError(url, 403, "rate limited", hdrs=None, fp=None)
monkeypatch.setattr(mod, "_http_json", fake)
assert mod.main(catalog_dir=cat, output=out, max_age_hours=24, live_url=None) == 0
data = json.loads(out.read_text())
# a/one refreshed; b/two kept its old count instead of dropping to 0; gitlab never probed.
assert data["stars"] == {"a/one": 42, "b/two": 9}
assert calls == ["https://api.github.com/repos/a/one", "https://api.github.com/repos/b/two"]
assert data["fetched_at"] > old

View File

@@ -38,6 +38,9 @@ import yaml
REPO_ROOT = Path(__file__).resolve().parents[2]
DEFAULT_CATALOG_DIR = REPO_ROOT / "plugin-catalog"
DEFAULT_OUTPUT_DIR = REPO_ROOT / "website" / "static" / "api"
# Written by fetch-plugin-stars.py (at most one GitHub probe per day); absent → no ranking data.
DEFAULT_STARS_FILE = DEFAULT_OUTPUT_DIR / "plugin-stars.json"
_GITHUB_REPO_RE = re.compile(r"^https://github\.com/([^/\s]+)/([^/\s#?]+?)(?:\.git)?/?$")
CATALOG_TIERS = ("official", "community")
CATALOG_CATEGORIES = ("desktop", "memory", "platform", "web", "tools", "voice", "automation", "models", "general")
@@ -66,13 +69,29 @@ def _normalize_capabilities(raw) -> dict:
}
def load_catalog_entries(catalog_dir: Path) -> list[dict]:
def load_stars(stars_file: Path) -> dict[str, int]:
"""``{"owner/repo": stars}`` from fetch-plugin-stars.py's cache; empty when missing/unreadable."""
try:
data = json.loads(stars_file.read_text(encoding="utf-8"))
except (OSError, ValueError):
return {}
raw = data.get("stars") if isinstance(data, dict) else None
return {str(k): int(v) for k, v in raw.items() if isinstance(v, (int, float))} if isinstance(raw, dict) else {}
def _repo_stars(repo: str, stars: dict[str, int]) -> int | None:
m = _GITHUB_REPO_RE.match(repo)
return stars.get(f"{m.group(1)}/{m.group(2)}") if m else None
def load_catalog_entries(catalog_dir: Path, stars: dict[str, int] | None = None) -> list[dict]:
"""Parse all ``*.yaml`` files (except removed.yaml) into page entries.
Entries missing any of name/repo/sha are skipped with a stderr log —
a malformed community entry must never break the docs deploy.
"""
entries: list[dict] = []
stars = stars or {}
if not catalog_dir.is_dir():
return entries
@@ -127,12 +146,22 @@ def load_catalog_entries(catalog_dir: Path) -> list[dict]:
"capabilities": _normalize_capabilities(raw.get("capabilities")),
"docsUrl": str(raw.get("docs_url") or "").strip(),
"installCommand": f"hermes plugins install {name}",
"stars": _repo_stars(repo, stars),
})
entries.sort(key=lambda e: (0 if e["tier"] == "official" else 1, e["name"]))
# Official first, then by stars (unknown = 0), then name so the order is stable.
entries.sort(key=lambda e: (0 if e["tier"] == "official" else 1, -(e["stars"] or 0), e["name"]))
return entries
def _stars_fetched_at(stars_file: Path) -> str | None:
try:
data = json.loads(stars_file.read_text(encoding="utf-8"))
except (OSError, ValueError):
return None
return str(data.get("fetched_at")) if isinstance(data, dict) and data.get("fetched_at") else None
def load_removed(catalog_dir: Path) -> list[dict]:
"""``removed:`` list from plugin-catalog/removed.yaml (mappings only)."""
removed_path = catalog_dir / "removed.yaml"
@@ -168,14 +197,17 @@ def load_raw_entries(catalog_dir: Path) -> list[dict]:
return entries
def main(catalog_dir: Path = DEFAULT_CATALOG_DIR, output_dir: Path = DEFAULT_OUTPUT_DIR) -> int:
def main(catalog_dir: Path = DEFAULT_CATALOG_DIR, output_dir: Path = DEFAULT_OUTPUT_DIR,
stars_file: Path | None = None) -> int:
if not catalog_dir.is_dir():
_log(
f"plugin-catalog directory not found at {catalog_dir}; "
"emitting empty catalog (this is expected until the catalog lands)"
)
entries = load_catalog_entries(catalog_dir)
stars_path = stars_file if stars_file is not None else output_dir / "plugin-stars.json"
stars = load_stars(stars_path)
entries = load_catalog_entries(catalog_dir, stars)
removed_count = count_removed(catalog_dir)
by_tier = Counter(e["tier"] for e in entries)
@@ -185,6 +217,7 @@ def main(catalog_dir: Path = DEFAULT_CATALOG_DIR, output_dir: Path = DEFAULT_OUT
"total": len(entries),
"byTier": {tier: by_tier.get(tier, 0) for tier in CATALOG_TIERS},
"byCategory": {c: by_category.get(c, 0) for c in CATALOG_CATEGORIES if by_category.get(c)},
"starsFetchedAt": _stars_fetched_at(stars_path),
"removedCount": removed_count,
}
@@ -209,5 +242,7 @@ if __name__ == "__main__":
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--catalog-dir", type=Path, default=DEFAULT_CATALOG_DIR)
parser.add_argument("--output-dir", type=Path, default=DEFAULT_OUTPUT_DIR)
parser.add_argument("--stars-file", type=Path, default=None,
help="plugin-stars.json from fetch-plugin-stars.py (default: <output-dir>/plugin-stars.json)")
args = parser.parse_args()
sys.exit(main(catalog_dir=args.catalog_dir, output_dir=args.output_dir))
sys.exit(main(catalog_dir=args.catalog_dir, output_dir=args.output_dir, stars_file=args.stars_file))

View File

@@ -0,0 +1,169 @@
#!/usr/bin/env python3
"""Refresh GitHub star counts for plugin-catalog repos, at most once a day.
Writes ``website/static/api/plugin-stars.json``::
{"fetched_at": "<ISO-8601 UTC>", "stars": {"owner/repo": 123, ...}}
``extract-plugins.py`` merges these into ``plugins.json`` so the catalog page can rank
entries by stars.
Rate-limit discipline (the whole point of this file): the docs site deploys many times a
day and GitHub's per-installation API budget is shared with every other workflow. So this
script NEVER calls the GitHub API unless the cache is stale:
1. Download the live site's current ``plugin-stars.json`` (one CDN GET, not the API).
2. If its ``fetched_at`` is younger than ``--max-age-hours`` (default 24), write it back to
disk unchanged and exit. Zero GitHub calls.
3. Otherwise probe ``GET /repos/{owner}/{repo}`` once per unique repo. A 403/429 or any
network error keeps the previous count for that repo rather than dropping it.
Local ``npm run build`` without network or token degrades to whatever is on disk / an
empty map; the page then simply ranks alphabetically.
"""
from __future__ import annotations
import argparse
import json
import os
import re
import sys
import urllib.error
import urllib.request
from datetime import datetime, timedelta, timezone
from pathlib import Path
import yaml
REPO_ROOT = Path(__file__).resolve().parents[2]
DEFAULT_CATALOG_DIR = REPO_ROOT / "plugin-catalog"
DEFAULT_OUTPUT = REPO_ROOT / "website" / "static" / "api" / "plugin-stars.json"
LIVE_URL = "https://hermes-agent.nousresearch.com/docs/api/plugin-stars.json"
_GITHUB_REPO_RE = re.compile(r"^https://github\.com/([^/\s]+)/([^/\s#?]+?)(?:\.git)?/?$")
def _log(msg: str) -> None:
print(f"[fetch-plugin-stars] {msg}", file=sys.stderr)
def github_slug(repo_url: str) -> str | None:
"""``owner/repo`` for a github.com URL, else None (non-GitHub hosts are never probed)."""
m = _GITHUB_REPO_RE.match(repo_url.strip())
return f"{m.group(1)}/{m.group(2)}" if m else None
def catalog_slugs(catalog_dir: Path) -> list[str]:
slugs: set[str] = set()
for path in sorted(catalog_dir.glob("*.yaml")):
if path.name == "removed.yaml":
continue
try:
raw = yaml.safe_load(path.read_text(encoding="utf-8"))
except (yaml.YAMLError, OSError):
continue
slug = github_slug(str((raw or {}).get("repo") or "")) if isinstance(raw, dict) else None
if slug:
slugs.add(slug)
return sorted(slugs)
def _http_json(url: str, headers: dict[str, str], timeout: float = 15.0):
req = urllib.request.Request(url, headers={"User-Agent": "hermes-agent-docs", **headers})
with urllib.request.urlopen(req, timeout=timeout) as resp:
return json.loads(resp.read().decode("utf-8"))
def load_previous(output: Path, live_url: str | None) -> dict:
"""Newest of {live site copy, on-disk copy}; ``{}`` when neither exists."""
candidates: list[dict] = []
if live_url:
try:
data = _http_json(live_url, {})
if isinstance(data, dict) and isinstance(data.get("stars"), dict):
candidates.append(data)
except (urllib.error.URLError, OSError, ValueError) as e:
_log(f"live cache unavailable ({e}); continuing without it")
if output.is_file():
try:
data = json.loads(output.read_text(encoding="utf-8"))
if isinstance(data, dict) and isinstance(data.get("stars"), dict):
candidates.append(data)
except (OSError, ValueError):
pass
return max(candidates, key=lambda d: str(d.get("fetched_at") or ""), default={})
def is_fresh(previous: dict, max_age: timedelta, now: datetime) -> bool:
try:
fetched = datetime.fromisoformat(str(previous.get("fetched_at")))
except (TypeError, ValueError):
return False
if fetched.tzinfo is None:
fetched = fetched.replace(tzinfo=timezone.utc)
return now - fetched < max_age
def probe_stars(slugs: list[str], previous: dict[str, int], token: str | None) -> dict[str, int]:
"""One ``GET /repos/{slug}`` each; on any failure keep the previous count (never regress to 0)."""
headers = {"Accept": "application/vnd.github+json"}
if token:
headers["Authorization"] = f"Bearer {token}"
stars: dict[str, int] = {}
rate_limited = False
for slug in slugs:
if rate_limited:
if slug in previous:
stars[slug] = previous[slug]
continue
try:
data = _http_json(f"https://api.github.com/repos/{slug}", headers)
stars[slug] = int(data.get("stargazers_count") or 0)
except urllib.error.HTTPError as e:
if e.code in (403, 429):
_log(f"rate limited at {slug} (HTTP {e.code}); keeping previous counts for the rest")
rate_limited = True
else:
_log(f"{slug}: HTTP {e.code}; keeping previous count")
if slug in previous:
stars[slug] = previous[slug]
except (urllib.error.URLError, OSError, ValueError) as e:
_log(f"{slug}: {e}; keeping previous count")
if slug in previous:
stars[slug] = previous[slug]
return stars
def main(catalog_dir: Path = DEFAULT_CATALOG_DIR, output: Path = DEFAULT_OUTPUT,
max_age_hours: float = 24.0, force: bool = False, live_url: str | None = LIVE_URL,
token: str | None = None) -> int:
now = datetime.now(timezone.utc)
previous = load_previous(output, live_url)
output.parent.mkdir(parents=True, exist_ok=True)
if not force and is_fresh(previous, timedelta(hours=max_age_hours), now):
output.write_text(json.dumps(previous, separators=(",", ":")), encoding="utf-8")
print(f"Reused plugin stars from {previous.get('fetched_at')} "
f"({len(previous.get('stars', {}))} repos, no GitHub calls)")
return 0
slugs = catalog_slugs(catalog_dir)
prev_stars = {k: int(v) for k, v in (previous.get("stars") or {}).items() if isinstance(v, (int, float))}
stars = probe_stars(slugs, prev_stars, token or os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN"))
fetched_at = now.isoformat() if stars else str(previous.get("fetched_at") or "")
output.write_text(json.dumps({"fetched_at": fetched_at, "stars": stars}, separators=(",", ":")),
encoding="utf-8")
print(f"Probed {len(slugs)} repos, wrote {len(stars)} star counts to {output}")
return 0
if __name__ == "__main__":
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--catalog-dir", type=Path, default=DEFAULT_CATALOG_DIR)
parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT)
parser.add_argument("--max-age-hours", type=float, default=24.0)
parser.add_argument("--force", action="store_true", help="probe GitHub even if the cache is fresh")
parser.add_argument("--no-live", action="store_true", help="do not consult the live site's cache")
args = parser.parse_args()
sys.exit(main(catalog_dir=args.catalog_dir, output=args.output, max_age_hours=args.max_age_hours,
force=args.force, live_url=None if args.no_live else LIVE_URL))

View File

@@ -33,6 +33,7 @@ const extractScript = join(scriptDir, "extract-skills.py");
const llmsScript = join(scriptDir, "generate-llms-txt.py");
const cronBlueprintsScript = join(scriptDir, "extract-automation-blueprints.py");
const pluginsScript = join(scriptDir, "extract-plugins.py");
const pluginStarsScript = join(scriptDir, "fetch-plugin-stars.py");
const outputFile = join(websiteDir, "static", "api", "skills.json");
const pluginsOutputFile = join(websiteDir, "static", "api", "plugins.json");
const pluginsMetaOutputFile = join(websiteDir, "static", "api", "plugins-meta.json");
@@ -147,6 +148,11 @@ runPython(llmsScript, "generate-llms-txt.py");
// renders an empty state if the generator can't run.
runPython(cronBlueprintsScript, "extract-automation-blueprints.py");
// 4a) plugin-stars.json — GitHub star counts for catalog ranking. Reuses the live
// site's daily cache (one CDN GET); only probes the API when that is >24h old,
// and never fails the build (no token / offline → whatever is cached or nothing).
runPython(pluginStarsScript, "fetch-plugin-stars.py");
// 4) plugins.json + plugins-meta.json — Plugin Catalog page. The script itself
// degrades gracefully (empty catalog, exit 0) when plugin-catalog/ is absent;
// if python3 is missing entirely, write the same empty fallback so the page

View File

@@ -25,6 +25,8 @@ interface CatalogPlugin {
capabilities?: PluginCapabilities;
docsUrl?: string;
installCommand: string;
/** GitHub stargazers at the last daily probe; null when the repo is not on GitHub or unprobed. */
stars?: number | null;
/** Lowercase pre-joined haystack for the search filter (built at load). */
_search?: string;
}
@@ -35,6 +37,7 @@ interface CatalogMeta {
byTier?: Record<string, number>;
byCategory?: Record<string, number>;
removedCount?: number;
starsFetchedAt?: string | null;
}
// Routes Docusaurus serves the static API JSON from. `baseUrl` is `/docs/`,
@@ -100,6 +103,10 @@ function formatRelativeTime(iso?: string): string | null {
return `${months} month${months === 1 ? "" : "s"} ago`;
}
function formatStars(n: number): string {
return n >= 1000 ? `${(n / 1000).toFixed(n >= 10_000 ? 0 : 1)}k` : String(n);
}
function highlightMatch(text: string, query: string): React.ReactNode {
if (!query || !text) return text;
const idx = text.toLowerCase().indexOf(query.toLowerCase());
@@ -203,6 +210,18 @@ function PluginCard({
>
{tier.icon} {tier.label}
</span>
{typeof plugin.stars === "number" && (
<a
className={styles.starPill}
href={`${plugin.repo.replace(/\.git$/, "").replace(/\/$/, "")}/stargazers`}
target="_blank"
rel="noopener noreferrer"
onClick={(e) => e.stopPropagation()}
title={`${plugin.stars.toLocaleString()} GitHub stars`}
>
{"\u2605"} {formatStars(plugin.stars)}
</a>
)}
</div>
</div>
@@ -547,6 +566,14 @@ export default function PluginCatalogPage() {
<span title={meta.generatedAt}>
{formatRelativeTime(meta.generatedAt) || "recently"}
</span>
{meta.starsFetchedAt && (
<>
{" · "}ranked by GitHub stars as of{" "}
<span title={meta.starsFetchedAt}>
{formatRelativeTime(meta.starsFetchedAt) || "recently"}
</span>
</>
)}
</p>
)}

View File

@@ -367,6 +367,27 @@
text-decoration: underline;
}
.starPill {
display: inline-flex;
align-items: center;
gap: 0.25rem;
padding: 0.1rem 0.5rem;
border: 1px solid rgba(255, 215, 0, 0.25);
border-radius: 10px;
background: rgba(255, 215, 0, 0.06);
color: #ffd700;
font-family: "JetBrains Mono", monospace;
font-size: 0.72rem;
text-decoration: none;
white-space: nowrap;
}
.starPill:hover {
background: rgba(255, 215, 0, 0.14);
text-decoration: none;
color: #ffd700;
}
.categoryChip {
border: 1px solid rgba(125, 211, 252, 0.25);
background: rgba(125, 211, 252, 0.07);