From 471d8a427ab252369809273c57425500699bd880 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Tue, 15 Sep 2026 04:31:29 -0700 Subject: [PATCH] ci(plugin-catalog): probe stars only from the scheduled skills-index run, in one GraphQL request MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Aligns the star ranking with the rule the skills index already follows: GitHub is consulted only by the twice-daily skills-index.yml schedule, whose artifact every docs deploy reuses. deploy-site.yml now runs fetch-plugin-stars.py without --probe (reuse-only: artifact → live site copy → disk → empty) and cannot call the API at all, so a same-day merge train adds zero requests regardless of cache age. The probe itself collapses from one REST call per repo (~26 today, growing with the catalog) to a single GraphQL query with aliased repository fields, so the scheduled run costs one request no matter how big the catalog gets. A failed probe (rate limit, renamed repo, bad token) keeps the previous counts. --- .github/workflows/deploy-site.yml | 13 +-- .github/workflows/skills-index.yml | 12 ++- tests/website/test_fetch_plugin_stars.py | 51 +++++----- website/scripts/fetch-plugin-stars.py | 114 ++++++++++++----------- website/scripts/prebuild.mjs | 6 +- 5 files changed, 107 insertions(+), 89 deletions(-) diff --git a/.github/workflows/deploy-site.yml b/.github/workflows/deploy-site.yml index a8816da897..83e95c0b6a 100644 --- a/.github/workflows/deploy-site.yml +++ b/.github/workflows/deploy-site.yml @@ -139,6 +139,8 @@ jobs: echo "Downloading skills-index artifact from run $SKILLS_INDEX_RUN_ID" if gh run download "$SKILLS_INDEX_RUN_ID" --name skills-index --dir "$tmpdir"; then candidate="$(find "$tmpdir" -name skills-index.json -type f | head -n 1 || true)" + stars="$(find "$tmpdir" -name plugin-stars.json -type f | head -n 1 || true)" + [ -n "$stars" ] && cp "$stars" website/static/api/plugin-stars.json if [ -n "$candidate" ]; then cp "$candidate" "$INDEX_PATH" if validate_index; then @@ -164,12 +166,11 @@ jobs: - name: Extract skill metadata for dashboard run: python3 website/scripts/extract-skills.py - # Star counts drive the catalog ranking. The script reuses the live site's cache when it - # is < 24h old (one CDN GET, zero API calls); only a stale cache triggers ~1 API call per - # catalog repo. Never touches the API on the many same-day deploys. - - name: Refresh plugin GitHub stars (daily cache) - env: - GITHUB_TOKEN: ${{ steps.app-token.outputs.token }} + # Star counts drive the catalog ranking. Same rule as the skills index: GitHub is only + # probed by the scheduled skills-index run; deploys reuse its artifact (downloaded above + # into website/static/api/ when SKILLS_INDEX_RUN_ID is set) or the live site's copy. + # Never calls the GitHub API. + - name: Reuse plugin GitHub stars (no API calls) run: python3 website/scripts/fetch-plugin-stars.py - name: Extract plugin catalog for the Plugins page diff --git a/.github/workflows/skills-index.yml b/.github/workflows/skills-index.yml index 2599ec9e77..dde5b7ab82 100644 --- a/.github/workflows/skills-index.yml +++ b/.github/workflows/skills-index.yml @@ -49,11 +49,21 @@ jobs: GITHUB_TOKEN: ${{ steps.app-token.outputs.token }} run: python scripts/build_skills_index.py + # Plugin-catalog star counts follow the same rule as the index: GitHub is consulted + # only here, on the schedule (one GraphQL request for every catalog repo), and docs + # deploys reuse the artifact / live copy without touching the API. + - name: Probe plugin catalog stars + env: + GITHUB_TOKEN: ${{ steps.app-token.outputs.token }} + run: python website/scripts/fetch-plugin-stars.py --probe + - name: Upload index artifact uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: skills-index - path: website/static/api/skills-index.json + path: | + website/static/api/skills-index.json + website/static/api/plugin-stars.json retention-days: 7 # Re-trigger the docs deploy so the refreshed index lands on the live site. diff --git a/tests/website/test_fetch_plugin_stars.py b/tests/website/test_fetch_plugin_stars.py index 90431355ca..b0d1134494 100644 --- a/tests/website/test_fetch_plugin_stars.py +++ b/tests/website/test_fetch_plugin_stars.py @@ -1,7 +1,8 @@ -"""fetch-plugin-stars.py: the daily GitHub-stars cache behind catalog ranking. +"""fetch-plugin-stars.py: plugin-catalog star counts, GitHub consulted only from the scheduled run. -The contract under test is rate-limit discipline, not the numbers: a fresh cache must never -reach GitHub, and a rate-limited probe must keep the previous counts rather than zeroing them. +The contract under test is rate-limit discipline, not the numbers: a deploy (no ``--probe``) +must never reach GitHub, the scheduled probe must be ONE request for every repo, and a failed +probe must keep the previous counts rather than zeroing them. """ from __future__ import annotations @@ -9,7 +10,6 @@ from __future__ import annotations import importlib.util import json import urllib.error -from datetime import datetime, timedelta, timezone from pathlib import Path import pytest @@ -38,38 +38,43 @@ def _catalog(tmp_path: Path, *repos: str) -> Path: return cat -def test_fresh_cache_is_reused_without_any_github_call(mod, tmp_path, monkeypatch): +def test_deploy_reuses_the_cache_without_any_github_call(mod, tmp_path, monkeypatch): cat = _catalog(tmp_path, "https://github.com/a/one") out = tmp_path / "plugin-stars.json" - recent = (datetime.now(timezone.utc) - timedelta(hours=2)).isoformat() - out.write_text(json.dumps({"fetched_at": recent, "stars": {"a/one": 7}}), encoding="utf-8") + out.write_text(json.dumps({"fetched_at": "2026-01-01T00:00:00+00:00", "stars": {"a/one": 7}}), encoding="utf-8") def boom(*a, **k): - raise AssertionError("GitHub must not be called while the cache is fresh") + raise AssertionError("GitHub must not be called without --probe") + monkeypatch.setattr(mod, "_graphql", boom) monkeypatch.setattr(mod, "_http_json", boom) - assert mod.main(catalog_dir=cat, output=out, max_age_hours=24, live_url=None) == 0 + assert mod.main(catalog_dir=cat, output=out, probe=False, live_url=None) == 0 assert json.loads(out.read_text())["stars"] == {"a/one": 7} -def test_stale_cache_probes_and_rate_limit_keeps_previous_counts(mod, tmp_path, monkeypatch): +def test_probe_is_one_graphql_request_and_a_failure_keeps_previous_counts(mod, tmp_path, monkeypatch): cat = _catalog(tmp_path, "https://github.com/a/one", "https://github.com/b/two", "https://gitlab.com/c/three") out = tmp_path / "plugin-stars.json" - old = (datetime.now(timezone.utc) - timedelta(days=3)).isoformat() - out.write_text(json.dumps({"fetched_at": old, "stars": {"a/one": 7, "b/two": 9}}), encoding="utf-8") - + out.write_text(json.dumps({"fetched_at": "2026-01-01T00:00:00+00:00", "stars": {"a/one": 7, "b/two": 9}}), + encoding="utf-8") calls: list[str] = [] - def fake(url, headers, timeout=15.0): - calls.append(url) - if url.endswith("/repos/a/one"): - return {"stargazers_count": 42} - raise urllib.error.HTTPError(url, 403, "rate limited", hdrs=None, fp=None) - monkeypatch.setattr(mod, "_http_json", fake) + def one_request(query, token): + calls.append(query) + # b/two errored (renamed repo): its node is null, previous count must survive. + return {"data": {"r0": {"stargazerCount": 42}, "r1": None}, + "errors": [{"message": "Could not resolve to a Repository"}]} + monkeypatch.setattr(mod, "_graphql", one_request) - assert mod.main(catalog_dir=cat, output=out, max_age_hours=24, live_url=None) == 0 + assert mod.main(catalog_dir=cat, output=out, probe=True, live_url=None, token="t") == 0 data = json.loads(out.read_text()) - # a/one refreshed; b/two kept its old count instead of dropping to 0; gitlab never probed. assert data["stars"] == {"a/one": 42, "b/two": 9} - assert calls == ["https://api.github.com/repos/a/one", "https://api.github.com/repos/b/two"] - assert data["fetched_at"] > old + assert len(calls) == 1 and "gitlab" not in calls[0] and 'owner: "a"' in calls[0] and 'owner: "b"' in calls[0] + assert data["fetched_at"] > "2026-01-01" + + # A rate-limited / failed probe keeps everything as it was. + def limited(query, token): + raise urllib.error.HTTPError("u", 403, "rate limited", hdrs=None, fp=None) + monkeypatch.setattr(mod, "_graphql", limited) + assert mod.main(catalog_dir=cat, output=out, probe=True, live_url=None, token="t") == 0 + assert json.loads(out.read_text())["stars"] == {"a/one": 42, "b/two": 9} diff --git a/website/scripts/fetch-plugin-stars.py b/website/scripts/fetch-plugin-stars.py index a45f8d9d1a..e10e665543 100644 --- a/website/scripts/fetch-plugin-stars.py +++ b/website/scripts/fetch-plugin-stars.py @@ -8,18 +8,15 @@ Writes ``website/static/api/plugin-stars.json``:: ``extract-plugins.py`` merges these into ``plugins.json`` so the catalog page can rank entries by stars. -Rate-limit discipline (the whole point of this file): the docs site deploys many times a -day and GitHub's per-installation API budget is shared with every other workflow. So this -script NEVER calls the GitHub API unless the cache is stale: +Rate-limit discipline follows the skills-index rule: GitHub is only ever consulted from the +scheduled ``skills-index.yml`` run (twice daily, ``--probe``), which uploads the result as an +artifact. Docs deploys run with ``--reuse-only`` and NEVER touch the API: they take the +scheduled artifact when given one, else the live site's copy (one CDN GET), else whatever is +on disk, else an empty map (the page then ranks alphabetically). -1. Download the live site's current ``plugin-stars.json`` (one CDN GET, not the API). -2. If its ``fetched_at`` is younger than ``--max-age-hours`` (default 24), write it back to - disk unchanged and exit. Zero GitHub calls. -3. Otherwise probe ``GET /repos/{owner}/{repo}`` once per unique repo. A 403/429 or any - network error keeps the previous count for that repo rather than dropping it. - -Local ``npm run build`` without network or token degrades to whatever is on disk / an -empty map; the page then simply ranks alphabetically. +The probe itself is ONE GraphQL request for every catalog repo (aliased ``repository`` +fields), not one REST call per repo. Any failure (rate limit, network, bad token) keeps the +previous counts instead of regressing them to zero. """ from __future__ import annotations @@ -31,7 +28,7 @@ import re import sys import urllib.error import urllib.request -from datetime import datetime, timedelta, timezone +from datetime import datetime, timezone from pathlib import Path import yaml @@ -94,55 +91,59 @@ def load_previous(output: Path, live_url: str | None) -> dict: return max(candidates, key=lambda d: str(d.get("fetched_at") or ""), default={}) -def is_fresh(previous: dict, max_age: timedelta, now: datetime) -> bool: - try: - fetched = datetime.fromisoformat(str(previous.get("fetched_at"))) - except (TypeError, ValueError): - return False - if fetched.tzinfo is None: - fetched = fetched.replace(tzinfo=timezone.utc) - return now - fetched < max_age +_GRAPHQL_URL = "https://api.github.com/graphql" + + +def _graphql(query: str, token: str) -> dict: + req = urllib.request.Request( + _GRAPHQL_URL, data=json.dumps({"query": query}).encode("utf-8"), method="POST", + headers={"User-Agent": "hermes-agent-docs", "Authorization": f"Bearer {token}", + "Content-Type": "application/json"}) + with urllib.request.urlopen(req, timeout=30.0) as resp: + return json.loads(resp.read().decode("utf-8")) + + +def stars_query(slugs: list[str]) -> str: + fields = "\n".join( + f'r{i}: repository(owner: {json.dumps(slug.split("/", 1)[0])}, name: {json.dumps(slug.split("/", 1)[1])})' + " { stargazerCount }" + for i, slug in enumerate(slugs)) + return "query {\n" + fields + "\n}" def probe_stars(slugs: list[str], previous: dict[str, int], token: str | None) -> dict[str, int]: - """One ``GET /repos/{slug}`` each; on any failure keep the previous count (never regress to 0).""" - headers = {"Accept": "application/vnd.github+json"} - if token: - headers["Authorization"] = f"Bearer {token}" + """One GraphQL request for all repos; on failure keep every previous count (never regress to 0).""" + if not slugs: + return {} + if not token: + _log("no GITHUB_TOKEN; keeping previous counts without probing") + return {s: previous[s] for s in slugs if s in previous} + try: + payload = _graphql(stars_query(slugs), token) + except (urllib.error.URLError, OSError, ValueError) as e: + _log(f"GraphQL probe failed ({e}); keeping previous counts") + return {s: previous[s] for s in slugs if s in previous} + data = payload.get("data") or {} + for err in payload.get("errors") or []: + _log(f"GraphQL: {err.get('message')}") # e.g. a renamed/deleted repo; its previous count is kept stars: dict[str, int] = {} - rate_limited = False - for slug in slugs: - if rate_limited: - if slug in previous: - stars[slug] = previous[slug] - continue - try: - data = _http_json(f"https://api.github.com/repos/{slug}", headers) - stars[slug] = int(data.get("stargazers_count") or 0) - except urllib.error.HTTPError as e: - if e.code in (403, 429): - _log(f"rate limited at {slug} (HTTP {e.code}); keeping previous counts for the rest") - rate_limited = True - else: - _log(f"{slug}: HTTP {e.code}; keeping previous count") - if slug in previous: - stars[slug] = previous[slug] - except (urllib.error.URLError, OSError, ValueError) as e: - _log(f"{slug}: {e}; keeping previous count") - if slug in previous: - stars[slug] = previous[slug] + for i, slug in enumerate(slugs): + node = data.get(f"r{i}") + if isinstance(node, dict) and isinstance(node.get("stargazerCount"), int): + stars[slug] = node["stargazerCount"] + elif slug in previous: + stars[slug] = previous[slug] return stars def main(catalog_dir: Path = DEFAULT_CATALOG_DIR, output: Path = DEFAULT_OUTPUT, - max_age_hours: float = 24.0, force: bool = False, live_url: str | None = LIVE_URL, - token: str | None = None) -> int: - now = datetime.now(timezone.utc) + probe: bool = False, live_url: str | None = LIVE_URL, token: str | None = None) -> int: previous = load_previous(output, live_url) output.parent.mkdir(parents=True, exist_ok=True) - if not force and is_fresh(previous, timedelta(hours=max_age_hours), now): - output.write_text(json.dumps(previous, separators=(",", ":")), encoding="utf-8") + if not probe: + output.write_text(json.dumps(previous or {"fetched_at": None, "stars": {}}, + separators=(",", ":")), encoding="utf-8") print(f"Reused plugin stars from {previous.get('fetched_at')} " f"({len(previous.get('stars', {}))} repos, no GitHub calls)") return 0 @@ -150,10 +151,11 @@ def main(catalog_dir: Path = DEFAULT_CATALOG_DIR, output: Path = DEFAULT_OUTPUT, slugs = catalog_slugs(catalog_dir) prev_stars = {k: int(v) for k, v in (previous.get("stars") or {}).items() if isinstance(v, (int, float))} stars = probe_stars(slugs, prev_stars, token or os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")) - fetched_at = now.isoformat() if stars else str(previous.get("fetched_at") or "") + probed = stars != prev_stars or not previous + fetched_at = datetime.now(timezone.utc).isoformat() if probed or stars else str(previous.get("fetched_at") or "") output.write_text(json.dumps({"fetched_at": fetched_at, "stars": stars}, separators=(",", ":")), encoding="utf-8") - print(f"Probed {len(slugs)} repos, wrote {len(stars)} star counts to {output}") + print(f"Probed {len(slugs)} repos in one GraphQL request, wrote {len(stars)} star counts to {output}") return 0 @@ -161,9 +163,9 @@ if __name__ == "__main__": parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--catalog-dir", type=Path, default=DEFAULT_CATALOG_DIR) parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT) - parser.add_argument("--max-age-hours", type=float, default=24.0) - parser.add_argument("--force", action="store_true", help="probe GitHub even if the cache is fresh") + parser.add_argument("--probe", action="store_true", + help="call GitHub (one GraphQL request); only the scheduled skills-index run does this") parser.add_argument("--no-live", action="store_true", help="do not consult the live site's cache") args = parser.parse_args() - sys.exit(main(catalog_dir=args.catalog_dir, output=args.output, max_age_hours=args.max_age_hours, - force=args.force, live_url=None if args.no_live else LIVE_URL)) + sys.exit(main(catalog_dir=args.catalog_dir, output=args.output, probe=args.probe, + live_url=None if args.no_live else LIVE_URL)) diff --git a/website/scripts/prebuild.mjs b/website/scripts/prebuild.mjs index 67c4c27a2c..98ca1d29c0 100644 --- a/website/scripts/prebuild.mjs +++ b/website/scripts/prebuild.mjs @@ -148,9 +148,9 @@ runPython(llmsScript, "generate-llms-txt.py"); // renders an empty state if the generator can't run. runPython(cronBlueprintsScript, "extract-automation-blueprints.py"); -// 4a) plugin-stars.json — GitHub star counts for catalog ranking. Reuses the live -// site's daily cache (one CDN GET); only probes the API when that is >24h old, -// and never fails the build (no token / offline → whatever is cached or nothing). +// 4a) plugin-stars.json — GitHub star counts for catalog ranking. Reuse-only here +// (live site copy via one CDN GET, else on-disk, else empty); GitHub itself is +// probed only by the scheduled skills-index workflow. Never fails the build. runPython(pluginStarsScript, "fetch-plugin-stars.py"); // 4) plugins.json + plugins-meta.json — Plugin Catalog page. The script itself