ci(plugin-catalog): probe stars only from the scheduled skills-index run, in one GraphQL request
Aligns the star ranking with the rule the skills index already follows: GitHub is consulted only by the twice-daily skills-index.yml schedule, whose artifact every docs deploy reuses. deploy-site.yml now runs fetch-plugin-stars.py without --probe (reuse-only: artifact → live site copy → disk → empty) and cannot call the API at all, so a same-day merge train adds zero requests regardless of cache age. The probe itself collapses from one REST call per repo (~26 today, growing with the catalog) to a single GraphQL query with aliased repository fields, so the scheduled run costs one request no matter how big the catalog gets. A failed probe (rate limit, renamed repo, bad token) keeps the previous counts.
This commit is contained in:
@@ -139,6 +139,8 @@ jobs:
|
||||
echo "Downloading skills-index artifact from run $SKILLS_INDEX_RUN_ID"
|
||||
if gh run download "$SKILLS_INDEX_RUN_ID" --name skills-index --dir "$tmpdir"; then
|
||||
candidate="$(find "$tmpdir" -name skills-index.json -type f | head -n 1 || true)"
|
||||
stars="$(find "$tmpdir" -name plugin-stars.json -type f | head -n 1 || true)"
|
||||
[ -n "$stars" ] && cp "$stars" website/static/api/plugin-stars.json
|
||||
if [ -n "$candidate" ]; then
|
||||
cp "$candidate" "$INDEX_PATH"
|
||||
if validate_index; then
|
||||
@@ -164,12 +166,11 @@ jobs:
|
||||
- name: Extract skill metadata for dashboard
|
||||
run: python3 website/scripts/extract-skills.py
|
||||
|
||||
# Star counts drive the catalog ranking. The script reuses the live site's cache when it
|
||||
# is < 24h old (one CDN GET, zero API calls); only a stale cache triggers ~1 API call per
|
||||
# catalog repo. Never touches the API on the many same-day deploys.
|
||||
- name: Refresh plugin GitHub stars (daily cache)
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
# Star counts drive the catalog ranking. Same rule as the skills index: GitHub is only
|
||||
# probed by the scheduled skills-index run; deploys reuse its artifact (downloaded above
|
||||
# into website/static/api/ when SKILLS_INDEX_RUN_ID is set) or the live site's copy.
|
||||
# Never calls the GitHub API.
|
||||
- name: Reuse plugin GitHub stars (no API calls)
|
||||
run: python3 website/scripts/fetch-plugin-stars.py
|
||||
|
||||
- name: Extract plugin catalog for the Plugins page
|
||||
|
||||
@@ -49,11 +49,21 @@ jobs:
|
||||
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
run: python scripts/build_skills_index.py
|
||||
|
||||
# Plugin-catalog star counts follow the same rule as the index: GitHub is consulted
|
||||
# only here, on the schedule (one GraphQL request for every catalog repo), and docs
|
||||
# deploys reuse the artifact / live copy without touching the API.
|
||||
- name: Probe plugin catalog stars
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
run: python website/scripts/fetch-plugin-stars.py --probe
|
||||
|
||||
- name: Upload index artifact
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: skills-index
|
||||
path: website/static/api/skills-index.json
|
||||
path: |
|
||||
website/static/api/skills-index.json
|
||||
website/static/api/plugin-stars.json
|
||||
retention-days: 7
|
||||
|
||||
# Re-trigger the docs deploy so the refreshed index lands on the live site.
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
"""fetch-plugin-stars.py: the daily GitHub-stars cache behind catalog ranking.
|
||||
"""fetch-plugin-stars.py: plugin-catalog star counts, GitHub consulted only from the scheduled run.
|
||||
|
||||
The contract under test is rate-limit discipline, not the numbers: a fresh cache must never
|
||||
reach GitHub, and a rate-limited probe must keep the previous counts rather than zeroing them.
|
||||
The contract under test is rate-limit discipline, not the numbers: a deploy (no ``--probe``)
|
||||
must never reach GitHub, the scheduled probe must be ONE request for every repo, and a failed
|
||||
probe must keep the previous counts rather than zeroing them.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -9,7 +10,6 @@ from __future__ import annotations
|
||||
import importlib.util
|
||||
import json
|
||||
import urllib.error
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
@@ -38,38 +38,43 @@ def _catalog(tmp_path: Path, *repos: str) -> Path:
|
||||
return cat
|
||||
|
||||
|
||||
def test_fresh_cache_is_reused_without_any_github_call(mod, tmp_path, monkeypatch):
|
||||
def test_deploy_reuses_the_cache_without_any_github_call(mod, tmp_path, monkeypatch):
|
||||
cat = _catalog(tmp_path, "https://github.com/a/one")
|
||||
out = tmp_path / "plugin-stars.json"
|
||||
recent = (datetime.now(timezone.utc) - timedelta(hours=2)).isoformat()
|
||||
out.write_text(json.dumps({"fetched_at": recent, "stars": {"a/one": 7}}), encoding="utf-8")
|
||||
out.write_text(json.dumps({"fetched_at": "2026-01-01T00:00:00+00:00", "stars": {"a/one": 7}}), encoding="utf-8")
|
||||
|
||||
def boom(*a, **k):
|
||||
raise AssertionError("GitHub must not be called while the cache is fresh")
|
||||
raise AssertionError("GitHub must not be called without --probe")
|
||||
monkeypatch.setattr(mod, "_graphql", boom)
|
||||
monkeypatch.setattr(mod, "_http_json", boom)
|
||||
|
||||
assert mod.main(catalog_dir=cat, output=out, max_age_hours=24, live_url=None) == 0
|
||||
assert mod.main(catalog_dir=cat, output=out, probe=False, live_url=None) == 0
|
||||
assert json.loads(out.read_text())["stars"] == {"a/one": 7}
|
||||
|
||||
|
||||
def test_stale_cache_probes_and_rate_limit_keeps_previous_counts(mod, tmp_path, monkeypatch):
|
||||
def test_probe_is_one_graphql_request_and_a_failure_keeps_previous_counts(mod, tmp_path, monkeypatch):
|
||||
cat = _catalog(tmp_path, "https://github.com/a/one", "https://github.com/b/two", "https://gitlab.com/c/three")
|
||||
out = tmp_path / "plugin-stars.json"
|
||||
old = (datetime.now(timezone.utc) - timedelta(days=3)).isoformat()
|
||||
out.write_text(json.dumps({"fetched_at": old, "stars": {"a/one": 7, "b/two": 9}}), encoding="utf-8")
|
||||
|
||||
out.write_text(json.dumps({"fetched_at": "2026-01-01T00:00:00+00:00", "stars": {"a/one": 7, "b/two": 9}}),
|
||||
encoding="utf-8")
|
||||
calls: list[str] = []
|
||||
|
||||
def fake(url, headers, timeout=15.0):
|
||||
calls.append(url)
|
||||
if url.endswith("/repos/a/one"):
|
||||
return {"stargazers_count": 42}
|
||||
raise urllib.error.HTTPError(url, 403, "rate limited", hdrs=None, fp=None)
|
||||
monkeypatch.setattr(mod, "_http_json", fake)
|
||||
def one_request(query, token):
|
||||
calls.append(query)
|
||||
# b/two errored (renamed repo): its node is null, previous count must survive.
|
||||
return {"data": {"r0": {"stargazerCount": 42}, "r1": None},
|
||||
"errors": [{"message": "Could not resolve to a Repository"}]}
|
||||
monkeypatch.setattr(mod, "_graphql", one_request)
|
||||
|
||||
assert mod.main(catalog_dir=cat, output=out, max_age_hours=24, live_url=None) == 0
|
||||
assert mod.main(catalog_dir=cat, output=out, probe=True, live_url=None, token="t") == 0
|
||||
data = json.loads(out.read_text())
|
||||
# a/one refreshed; b/two kept its old count instead of dropping to 0; gitlab never probed.
|
||||
assert data["stars"] == {"a/one": 42, "b/two": 9}
|
||||
assert calls == ["https://api.github.com/repos/a/one", "https://api.github.com/repos/b/two"]
|
||||
assert data["fetched_at"] > old
|
||||
assert len(calls) == 1 and "gitlab" not in calls[0] and 'owner: "a"' in calls[0] and 'owner: "b"' in calls[0]
|
||||
assert data["fetched_at"] > "2026-01-01"
|
||||
|
||||
# A rate-limited / failed probe keeps everything as it was.
|
||||
def limited(query, token):
|
||||
raise urllib.error.HTTPError("u", 403, "rate limited", hdrs=None, fp=None)
|
||||
monkeypatch.setattr(mod, "_graphql", limited)
|
||||
assert mod.main(catalog_dir=cat, output=out, probe=True, live_url=None, token="t") == 0
|
||||
assert json.loads(out.read_text())["stars"] == {"a/one": 42, "b/two": 9}
|
||||
|
||||
@@ -8,18 +8,15 @@ Writes ``website/static/api/plugin-stars.json``::
|
||||
``extract-plugins.py`` merges these into ``plugins.json`` so the catalog page can rank
|
||||
entries by stars.
|
||||
|
||||
Rate-limit discipline (the whole point of this file): the docs site deploys many times a
|
||||
day and GitHub's per-installation API budget is shared with every other workflow. So this
|
||||
script NEVER calls the GitHub API unless the cache is stale:
|
||||
Rate-limit discipline follows the skills-index rule: GitHub is only ever consulted from the
|
||||
scheduled ``skills-index.yml`` run (twice daily, ``--probe``), which uploads the result as an
|
||||
artifact. Docs deploys run with ``--reuse-only`` and NEVER touch the API: they take the
|
||||
scheduled artifact when given one, else the live site's copy (one CDN GET), else whatever is
|
||||
on disk, else an empty map (the page then ranks alphabetically).
|
||||
|
||||
1. Download the live site's current ``plugin-stars.json`` (one CDN GET, not the API).
|
||||
2. If its ``fetched_at`` is younger than ``--max-age-hours`` (default 24), write it back to
|
||||
disk unchanged and exit. Zero GitHub calls.
|
||||
3. Otherwise probe ``GET /repos/{owner}/{repo}`` once per unique repo. A 403/429 or any
|
||||
network error keeps the previous count for that repo rather than dropping it.
|
||||
|
||||
Local ``npm run build`` without network or token degrades to whatever is on disk / an
|
||||
empty map; the page then simply ranks alphabetically.
|
||||
The probe itself is ONE GraphQL request for every catalog repo (aliased ``repository``
|
||||
fields), not one REST call per repo. Any failure (rate limit, network, bad token) keeps the
|
||||
previous counts instead of regressing them to zero.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -31,7 +28,7 @@ import re
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
import yaml
|
||||
@@ -94,55 +91,59 @@ def load_previous(output: Path, live_url: str | None) -> dict:
|
||||
return max(candidates, key=lambda d: str(d.get("fetched_at") or ""), default={})
|
||||
|
||||
|
||||
def is_fresh(previous: dict, max_age: timedelta, now: datetime) -> bool:
|
||||
try:
|
||||
fetched = datetime.fromisoformat(str(previous.get("fetched_at")))
|
||||
except (TypeError, ValueError):
|
||||
return False
|
||||
if fetched.tzinfo is None:
|
||||
fetched = fetched.replace(tzinfo=timezone.utc)
|
||||
return now - fetched < max_age
|
||||
_GRAPHQL_URL = "https://api.github.com/graphql"
|
||||
|
||||
|
||||
def _graphql(query: str, token: str) -> dict:
|
||||
req = urllib.request.Request(
|
||||
_GRAPHQL_URL, data=json.dumps({"query": query}).encode("utf-8"), method="POST",
|
||||
headers={"User-Agent": "hermes-agent-docs", "Authorization": f"Bearer {token}",
|
||||
"Content-Type": "application/json"})
|
||||
with urllib.request.urlopen(req, timeout=30.0) as resp:
|
||||
return json.loads(resp.read().decode("utf-8"))
|
||||
|
||||
|
||||
def stars_query(slugs: list[str]) -> str:
|
||||
fields = "\n".join(
|
||||
f'r{i}: repository(owner: {json.dumps(slug.split("/", 1)[0])}, name: {json.dumps(slug.split("/", 1)[1])})'
|
||||
" { stargazerCount }"
|
||||
for i, slug in enumerate(slugs))
|
||||
return "query {\n" + fields + "\n}"
|
||||
|
||||
|
||||
def probe_stars(slugs: list[str], previous: dict[str, int], token: str | None) -> dict[str, int]:
|
||||
"""One ``GET /repos/{slug}`` each; on any failure keep the previous count (never regress to 0)."""
|
||||
headers = {"Accept": "application/vnd.github+json"}
|
||||
if token:
|
||||
headers["Authorization"] = f"Bearer {token}"
|
||||
"""One GraphQL request for all repos; on failure keep every previous count (never regress to 0)."""
|
||||
if not slugs:
|
||||
return {}
|
||||
if not token:
|
||||
_log("no GITHUB_TOKEN; keeping previous counts without probing")
|
||||
return {s: previous[s] for s in slugs if s in previous}
|
||||
try:
|
||||
payload = _graphql(stars_query(slugs), token)
|
||||
except (urllib.error.URLError, OSError, ValueError) as e:
|
||||
_log(f"GraphQL probe failed ({e}); keeping previous counts")
|
||||
return {s: previous[s] for s in slugs if s in previous}
|
||||
data = payload.get("data") or {}
|
||||
for err in payload.get("errors") or []:
|
||||
_log(f"GraphQL: {err.get('message')}") # e.g. a renamed/deleted repo; its previous count is kept
|
||||
stars: dict[str, int] = {}
|
||||
rate_limited = False
|
||||
for slug in slugs:
|
||||
if rate_limited:
|
||||
if slug in previous:
|
||||
stars[slug] = previous[slug]
|
||||
continue
|
||||
try:
|
||||
data = _http_json(f"https://api.github.com/repos/{slug}", headers)
|
||||
stars[slug] = int(data.get("stargazers_count") or 0)
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code in (403, 429):
|
||||
_log(f"rate limited at {slug} (HTTP {e.code}); keeping previous counts for the rest")
|
||||
rate_limited = True
|
||||
else:
|
||||
_log(f"{slug}: HTTP {e.code}; keeping previous count")
|
||||
if slug in previous:
|
||||
stars[slug] = previous[slug]
|
||||
except (urllib.error.URLError, OSError, ValueError) as e:
|
||||
_log(f"{slug}: {e}; keeping previous count")
|
||||
if slug in previous:
|
||||
stars[slug] = previous[slug]
|
||||
for i, slug in enumerate(slugs):
|
||||
node = data.get(f"r{i}")
|
||||
if isinstance(node, dict) and isinstance(node.get("stargazerCount"), int):
|
||||
stars[slug] = node["stargazerCount"]
|
||||
elif slug in previous:
|
||||
stars[slug] = previous[slug]
|
||||
return stars
|
||||
|
||||
|
||||
def main(catalog_dir: Path = DEFAULT_CATALOG_DIR, output: Path = DEFAULT_OUTPUT,
|
||||
max_age_hours: float = 24.0, force: bool = False, live_url: str | None = LIVE_URL,
|
||||
token: str | None = None) -> int:
|
||||
now = datetime.now(timezone.utc)
|
||||
probe: bool = False, live_url: str | None = LIVE_URL, token: str | None = None) -> int:
|
||||
previous = load_previous(output, live_url)
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
if not force and is_fresh(previous, timedelta(hours=max_age_hours), now):
|
||||
output.write_text(json.dumps(previous, separators=(",", ":")), encoding="utf-8")
|
||||
if not probe:
|
||||
output.write_text(json.dumps(previous or {"fetched_at": None, "stars": {}},
|
||||
separators=(",", ":")), encoding="utf-8")
|
||||
print(f"Reused plugin stars from {previous.get('fetched_at')} "
|
||||
f"({len(previous.get('stars', {}))} repos, no GitHub calls)")
|
||||
return 0
|
||||
@@ -150,10 +151,11 @@ def main(catalog_dir: Path = DEFAULT_CATALOG_DIR, output: Path = DEFAULT_OUTPUT,
|
||||
slugs = catalog_slugs(catalog_dir)
|
||||
prev_stars = {k: int(v) for k, v in (previous.get("stars") or {}).items() if isinstance(v, (int, float))}
|
||||
stars = probe_stars(slugs, prev_stars, token or os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN"))
|
||||
fetched_at = now.isoformat() if stars else str(previous.get("fetched_at") or "")
|
||||
probed = stars != prev_stars or not previous
|
||||
fetched_at = datetime.now(timezone.utc).isoformat() if probed or stars else str(previous.get("fetched_at") or "")
|
||||
output.write_text(json.dumps({"fetched_at": fetched_at, "stars": stars}, separators=(",", ":")),
|
||||
encoding="utf-8")
|
||||
print(f"Probed {len(slugs)} repos, wrote {len(stars)} star counts to {output}")
|
||||
print(f"Probed {len(slugs)} repos in one GraphQL request, wrote {len(stars)} star counts to {output}")
|
||||
return 0
|
||||
|
||||
|
||||
@@ -161,9 +163,9 @@ if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--catalog-dir", type=Path, default=DEFAULT_CATALOG_DIR)
|
||||
parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT)
|
||||
parser.add_argument("--max-age-hours", type=float, default=24.0)
|
||||
parser.add_argument("--force", action="store_true", help="probe GitHub even if the cache is fresh")
|
||||
parser.add_argument("--probe", action="store_true",
|
||||
help="call GitHub (one GraphQL request); only the scheduled skills-index run does this")
|
||||
parser.add_argument("--no-live", action="store_true", help="do not consult the live site's cache")
|
||||
args = parser.parse_args()
|
||||
sys.exit(main(catalog_dir=args.catalog_dir, output=args.output, max_age_hours=args.max_age_hours,
|
||||
force=args.force, live_url=None if args.no_live else LIVE_URL))
|
||||
sys.exit(main(catalog_dir=args.catalog_dir, output=args.output, probe=args.probe,
|
||||
live_url=None if args.no_live else LIVE_URL))
|
||||
|
||||
@@ -148,9 +148,9 @@ runPython(llmsScript, "generate-llms-txt.py");
|
||||
// renders an empty state if the generator can't run.
|
||||
runPython(cronBlueprintsScript, "extract-automation-blueprints.py");
|
||||
|
||||
// 4a) plugin-stars.json — GitHub star counts for catalog ranking. Reuses the live
|
||||
// site's daily cache (one CDN GET); only probes the API when that is >24h old,
|
||||
// and never fails the build (no token / offline → whatever is cached or nothing).
|
||||
// 4a) plugin-stars.json — GitHub star counts for catalog ranking. Reuse-only here
|
||||
// (live site copy via one CDN GET, else on-disk, else empty); GitHub itself is
|
||||
// probed only by the scheduled skills-index workflow. Never fails the build.
|
||||
runPython(pluginStarsScript, "fetch-plugin-stars.py");
|
||||
|
||||
// 4) plugins.json + plugins-meta.json — Plugin Catalog page. The script itself
|
||||
|
||||
Reference in New Issue
Block a user