ci(plugin-catalog): probe stars only from the scheduled skills-index run, in one GraphQL request

Aligns the star ranking with the rule the skills index already follows:
GitHub is consulted only by the twice-daily skills-index.yml schedule, whose
artifact every docs deploy reuses. deploy-site.yml now runs
fetch-plugin-stars.py without --probe (reuse-only: artifact → live site copy →
disk → empty) and cannot call the API at all, so a same-day merge train adds
zero requests regardless of cache age.

The probe itself collapses from one REST call per repo (~26 today, growing
with the catalog) to a single GraphQL query with aliased repository fields,
so the scheduled run costs one request no matter how big the catalog gets. A
failed probe (rate limit, renamed repo, bad token) keeps the previous counts.
This commit is contained in:
teknium1
2026-09-15 04:31:29 -07:00
committed by Teknium
parent 1847ad2883
commit 471d8a427a
5 changed files with 107 additions and 89 deletions
+7 -6
View File
@@ -139,6 +139,8 @@ jobs:
echo "Downloading skills-index artifact from run $SKILLS_INDEX_RUN_ID"
if gh run download "$SKILLS_INDEX_RUN_ID" --name skills-index --dir "$tmpdir"; then
candidate="$(find "$tmpdir" -name skills-index.json -type f | head -n 1 || true)"
stars="$(find "$tmpdir" -name plugin-stars.json -type f | head -n 1 || true)"
[ -n "$stars" ] && cp "$stars" website/static/api/plugin-stars.json
if [ -n "$candidate" ]; then
cp "$candidate" "$INDEX_PATH"
if validate_index; then
@@ -164,12 +166,11 @@ jobs:
- name: Extract skill metadata for dashboard
run: python3 website/scripts/extract-skills.py
# Star counts drive the catalog ranking. The script reuses the live site's cache when it
# is < 24h old (one CDN GET, zero API calls); only a stale cache triggers ~1 API call per
# catalog repo. Never touches the API on the many same-day deploys.
- name: Refresh plugin GitHub stars (daily cache)
env:
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
# Star counts drive the catalog ranking. Same rule as the skills index: GitHub is only
# probed by the scheduled skills-index run; deploys reuse its artifact (downloaded above
# into website/static/api/ when SKILLS_INDEX_RUN_ID is set) or the live site's copy.
# Never calls the GitHub API.
- name: Reuse plugin GitHub stars (no API calls)
run: python3 website/scripts/fetch-plugin-stars.py
- name: Extract plugin catalog for the Plugins page
+11 -1
View File
@@ -49,11 +49,21 @@ jobs:
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
run: python scripts/build_skills_index.py
# Plugin-catalog star counts follow the same rule as the index: GitHub is consulted
# only here, on the schedule (one GraphQL request for every catalog repo), and docs
# deploys reuse the artifact / live copy without touching the API.
- name: Probe plugin catalog stars
env:
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
run: python website/scripts/fetch-plugin-stars.py --probe
- name: Upload index artifact
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: skills-index
path: website/static/api/skills-index.json
path: |
website/static/api/skills-index.json
website/static/api/plugin-stars.json
retention-days: 7
# Re-trigger the docs deploy so the refreshed index lands on the live site.
+28 -23
View File
@@ -1,7 +1,8 @@
"""fetch-plugin-stars.py: the daily GitHub-stars cache behind catalog ranking.
"""fetch-plugin-stars.py: plugin-catalog star counts, GitHub consulted only from the scheduled run.
The contract under test is rate-limit discipline, not the numbers: a fresh cache must never
reach GitHub, and a rate-limited probe must keep the previous counts rather than zeroing them.
The contract under test is rate-limit discipline, not the numbers: a deploy (no ``--probe``)
must never reach GitHub, the scheduled probe must be ONE request for every repo, and a failed
probe must keep the previous counts rather than zeroing them.
"""
from __future__ import annotations
@@ -9,7 +10,6 @@ from __future__ import annotations
import importlib.util
import json
import urllib.error
from datetime import datetime, timedelta, timezone
from pathlib import Path
import pytest
@@ -38,38 +38,43 @@ def _catalog(tmp_path: Path, *repos: str) -> Path:
return cat
def test_fresh_cache_is_reused_without_any_github_call(mod, tmp_path, monkeypatch):
def test_deploy_reuses_the_cache_without_any_github_call(mod, tmp_path, monkeypatch):
cat = _catalog(tmp_path, "https://github.com/a/one")
out = tmp_path / "plugin-stars.json"
recent = (datetime.now(timezone.utc) - timedelta(hours=2)).isoformat()
out.write_text(json.dumps({"fetched_at": recent, "stars": {"a/one": 7}}), encoding="utf-8")
out.write_text(json.dumps({"fetched_at": "2026-01-01T00:00:00+00:00", "stars": {"a/one": 7}}), encoding="utf-8")
def boom(*a, **k):
raise AssertionError("GitHub must not be called while the cache is fresh")
raise AssertionError("GitHub must not be called without --probe")
monkeypatch.setattr(mod, "_graphql", boom)
monkeypatch.setattr(mod, "_http_json", boom)
assert mod.main(catalog_dir=cat, output=out, max_age_hours=24, live_url=None) == 0
assert mod.main(catalog_dir=cat, output=out, probe=False, live_url=None) == 0
assert json.loads(out.read_text())["stars"] == {"a/one": 7}
def test_stale_cache_probes_and_rate_limit_keeps_previous_counts(mod, tmp_path, monkeypatch):
def test_probe_is_one_graphql_request_and_a_failure_keeps_previous_counts(mod, tmp_path, monkeypatch):
cat = _catalog(tmp_path, "https://github.com/a/one", "https://github.com/b/two", "https://gitlab.com/c/three")
out = tmp_path / "plugin-stars.json"
old = (datetime.now(timezone.utc) - timedelta(days=3)).isoformat()
out.write_text(json.dumps({"fetched_at": old, "stars": {"a/one": 7, "b/two": 9}}), encoding="utf-8")
out.write_text(json.dumps({"fetched_at": "2026-01-01T00:00:00+00:00", "stars": {"a/one": 7, "b/two": 9}}),
encoding="utf-8")
calls: list[str] = []
def fake(url, headers, timeout=15.0):
calls.append(url)
if url.endswith("/repos/a/one"):
return {"stargazers_count": 42}
raise urllib.error.HTTPError(url, 403, "rate limited", hdrs=None, fp=None)
monkeypatch.setattr(mod, "_http_json", fake)
def one_request(query, token):
calls.append(query)
# b/two errored (renamed repo): its node is null, previous count must survive.
return {"data": {"r0": {"stargazerCount": 42}, "r1": None},
"errors": [{"message": "Could not resolve to a Repository"}]}
monkeypatch.setattr(mod, "_graphql", one_request)
assert mod.main(catalog_dir=cat, output=out, max_age_hours=24, live_url=None) == 0
assert mod.main(catalog_dir=cat, output=out, probe=True, live_url=None, token="t") == 0
data = json.loads(out.read_text())
# a/one refreshed; b/two kept its old count instead of dropping to 0; gitlab never probed.
assert data["stars"] == {"a/one": 42, "b/two": 9}
assert calls == ["https://api.github.com/repos/a/one", "https://api.github.com/repos/b/two"]
assert data["fetched_at"] > old
assert len(calls) == 1 and "gitlab" not in calls[0] and 'owner: "a"' in calls[0] and 'owner: "b"' in calls[0]
assert data["fetched_at"] > "2026-01-01"
# A rate-limited / failed probe keeps everything as it was.
def limited(query, token):
raise urllib.error.HTTPError("u", 403, "rate limited", hdrs=None, fp=None)
monkeypatch.setattr(mod, "_graphql", limited)
assert mod.main(catalog_dir=cat, output=out, probe=True, live_url=None, token="t") == 0
assert json.loads(out.read_text())["stars"] == {"a/one": 42, "b/two": 9}
+58 -56
View File
@@ -8,18 +8,15 @@ Writes ``website/static/api/plugin-stars.json``::
``extract-plugins.py`` merges these into ``plugins.json`` so the catalog page can rank
entries by stars.
Rate-limit discipline (the whole point of this file): the docs site deploys many times a
day and GitHub's per-installation API budget is shared with every other workflow. So this
script NEVER calls the GitHub API unless the cache is stale:
Rate-limit discipline follows the skills-index rule: GitHub is only ever consulted from the
scheduled ``skills-index.yml`` run (twice daily, ``--probe``), which uploads the result as an
artifact. Docs deploys run with ``--reuse-only`` and NEVER touch the API: they take the
scheduled artifact when given one, else the live site's copy (one CDN GET), else whatever is
on disk, else an empty map (the page then ranks alphabetically).
1. Download the live site's current ``plugin-stars.json`` (one CDN GET, not the API).
2. If its ``fetched_at`` is younger than ``--max-age-hours`` (default 24), write it back to
disk unchanged and exit. Zero GitHub calls.
3. Otherwise probe ``GET /repos/{owner}/{repo}`` once per unique repo. A 403/429 or any
network error keeps the previous count for that repo rather than dropping it.
Local ``npm run build`` without network or token degrades to whatever is on disk / an
empty map; the page then simply ranks alphabetically.
The probe itself is ONE GraphQL request for every catalog repo (aliased ``repository``
fields), not one REST call per repo. Any failure (rate limit, network, bad token) keeps the
previous counts instead of regressing them to zero.
"""
from __future__ import annotations
@@ -31,7 +28,7 @@ import re
import sys
import urllib.error
import urllib.request
from datetime import datetime, timedelta, timezone
from datetime import datetime, timezone
from pathlib import Path
import yaml
@@ -94,55 +91,59 @@ def load_previous(output: Path, live_url: str | None) -> dict:
return max(candidates, key=lambda d: str(d.get("fetched_at") or ""), default={})
def is_fresh(previous: dict, max_age: timedelta, now: datetime) -> bool:
try:
fetched = datetime.fromisoformat(str(previous.get("fetched_at")))
except (TypeError, ValueError):
return False
if fetched.tzinfo is None:
fetched = fetched.replace(tzinfo=timezone.utc)
return now - fetched < max_age
_GRAPHQL_URL = "https://api.github.com/graphql"
def _graphql(query: str, token: str) -> dict:
req = urllib.request.Request(
_GRAPHQL_URL, data=json.dumps({"query": query}).encode("utf-8"), method="POST",
headers={"User-Agent": "hermes-agent-docs", "Authorization": f"Bearer {token}",
"Content-Type": "application/json"})
with urllib.request.urlopen(req, timeout=30.0) as resp:
return json.loads(resp.read().decode("utf-8"))
def stars_query(slugs: list[str]) -> str:
fields = "\n".join(
f'r{i}: repository(owner: {json.dumps(slug.split("/", 1)[0])}, name: {json.dumps(slug.split("/", 1)[1])})'
" { stargazerCount }"
for i, slug in enumerate(slugs))
return "query {\n" + fields + "\n}"
def probe_stars(slugs: list[str], previous: dict[str, int], token: str | None) -> dict[str, int]:
"""One ``GET /repos/{slug}`` each; on any failure keep the previous count (never regress to 0)."""
headers = {"Accept": "application/vnd.github+json"}
if token:
headers["Authorization"] = f"Bearer {token}"
"""One GraphQL request for all repos; on failure keep every previous count (never regress to 0)."""
if not slugs:
return {}
if not token:
_log("no GITHUB_TOKEN; keeping previous counts without probing")
return {s: previous[s] for s in slugs if s in previous}
try:
payload = _graphql(stars_query(slugs), token)
except (urllib.error.URLError, OSError, ValueError) as e:
_log(f"GraphQL probe failed ({e}); keeping previous counts")
return {s: previous[s] for s in slugs if s in previous}
data = payload.get("data") or {}
for err in payload.get("errors") or []:
_log(f"GraphQL: {err.get('message')}") # e.g. a renamed/deleted repo; its previous count is kept
stars: dict[str, int] = {}
rate_limited = False
for slug in slugs:
if rate_limited:
if slug in previous:
stars[slug] = previous[slug]
continue
try:
data = _http_json(f"https://api.github.com/repos/{slug}", headers)
stars[slug] = int(data.get("stargazers_count") or 0)
except urllib.error.HTTPError as e:
if e.code in (403, 429):
_log(f"rate limited at {slug} (HTTP {e.code}); keeping previous counts for the rest")
rate_limited = True
else:
_log(f"{slug}: HTTP {e.code}; keeping previous count")
if slug in previous:
stars[slug] = previous[slug]
except (urllib.error.URLError, OSError, ValueError) as e:
_log(f"{slug}: {e}; keeping previous count")
if slug in previous:
stars[slug] = previous[slug]
for i, slug in enumerate(slugs):
node = data.get(f"r{i}")
if isinstance(node, dict) and isinstance(node.get("stargazerCount"), int):
stars[slug] = node["stargazerCount"]
elif slug in previous:
stars[slug] = previous[slug]
return stars
def main(catalog_dir: Path = DEFAULT_CATALOG_DIR, output: Path = DEFAULT_OUTPUT,
max_age_hours: float = 24.0, force: bool = False, live_url: str | None = LIVE_URL,
token: str | None = None) -> int:
now = datetime.now(timezone.utc)
probe: bool = False, live_url: str | None = LIVE_URL, token: str | None = None) -> int:
previous = load_previous(output, live_url)
output.parent.mkdir(parents=True, exist_ok=True)
if not force and is_fresh(previous, timedelta(hours=max_age_hours), now):
output.write_text(json.dumps(previous, separators=(",", ":")), encoding="utf-8")
if not probe:
output.write_text(json.dumps(previous or {"fetched_at": None, "stars": {}},
separators=(",", ":")), encoding="utf-8")
print(f"Reused plugin stars from {previous.get('fetched_at')} "
f"({len(previous.get('stars', {}))} repos, no GitHub calls)")
return 0
@@ -150,10 +151,11 @@ def main(catalog_dir: Path = DEFAULT_CATALOG_DIR, output: Path = DEFAULT_OUTPUT,
slugs = catalog_slugs(catalog_dir)
prev_stars = {k: int(v) for k, v in (previous.get("stars") or {}).items() if isinstance(v, (int, float))}
stars = probe_stars(slugs, prev_stars, token or os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN"))
fetched_at = now.isoformat() if stars else str(previous.get("fetched_at") or "")
probed = stars != prev_stars or not previous
fetched_at = datetime.now(timezone.utc).isoformat() if probed or stars else str(previous.get("fetched_at") or "")
output.write_text(json.dumps({"fetched_at": fetched_at, "stars": stars}, separators=(",", ":")),
encoding="utf-8")
print(f"Probed {len(slugs)} repos, wrote {len(stars)} star counts to {output}")
print(f"Probed {len(slugs)} repos in one GraphQL request, wrote {len(stars)} star counts to {output}")
return 0
@@ -161,9 +163,9 @@ if __name__ == "__main__":
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--catalog-dir", type=Path, default=DEFAULT_CATALOG_DIR)
parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT)
parser.add_argument("--max-age-hours", type=float, default=24.0)
parser.add_argument("--force", action="store_true", help="probe GitHub even if the cache is fresh")
parser.add_argument("--probe", action="store_true",
help="call GitHub (one GraphQL request); only the scheduled skills-index run does this")
parser.add_argument("--no-live", action="store_true", help="do not consult the live site's cache")
args = parser.parse_args()
sys.exit(main(catalog_dir=args.catalog_dir, output=args.output, max_age_hours=args.max_age_hours,
force=args.force, live_url=None if args.no_live else LIVE_URL))
sys.exit(main(catalog_dir=args.catalog_dir, output=args.output, probe=args.probe,
live_url=None if args.no_live else LIVE_URL))
+3 -3
View File
@@ -148,9 +148,9 @@ runPython(llmsScript, "generate-llms-txt.py");
// renders an empty state if the generator can't run.
runPython(cronBlueprintsScript, "extract-automation-blueprints.py");
// 4a) plugin-stars.json — GitHub star counts for catalog ranking. Reuses the live
// site's daily cache (one CDN GET); only probes the API when that is >24h old,
// and never fails the build (no token / offline → whatever is cached or nothing).
// 4a) plugin-stars.json — GitHub star counts for catalog ranking. Reuse-only here
// (live site copy via one CDN GET, else on-disk, else empty); GitHub itself is
// probed only by the scheduled skills-index workflow. Never fails the build.
runPython(pluginStarsScript, "fetch-plugin-stars.py");
// 4) plugins.json + plugins-meta.json — Plugin Catalog page. The script itself