fix(skills-index): bound ClawHub owner enrichment so the scheduled index build finishes

Every scheduled skills-index.yml run since 2026-07-20 was cancelled at the 15-minute
job timeout, so the live skills-index.json has been frozen at that date and every
`hermes skills search` fell through to live GitHub API calls (~500 inspect requests per
cold search, against a 60/hr unauthenticated budget). The freshness watchdog has been
appending to #66616 four times a day since.

Root cause: enrich_owners() walks every ClawHub skill's detail endpoint (~2s each) to
fetch an owner handle for the "View source" link. The catalog grew from ~50k to 78k
skills, so even at 30 workers that phase alone runs over an hour; nothing bounded it.

- enrich_owners() gains budget_seconds: on expiry it stops and ships the remainder
  without an owner (the link is a nicety; the index is not).
- build_skills_index.py passes an 8-minute budget.
- skills-index.yml build job timeout 15 -> 50 min to cover the measured critical path
  (clawhub walk ~14 min || github taps ~8 min, skills.sh resolve ~6 min, enrichment 8 min).
This commit is contained in:
teknium1
2026-09-12 06:50:58 -07:00
committed by Teknium
parent b309a713a8
commit e7794124da
5 changed files with 47 additions and 6 deletions
+4 -1
View File
@@ -20,7 +20,10 @@ jobs:
# Only run on the upstream repository, not on forks # Only run on the upstream repository, not on forks
if: github.repository == 'NousResearch/hermes-agent' if: github.repository == 'NousResearch/hermes-agent'
runs-on: ubuntu-latest runs-on: ubuntu-latest
timeout-minutes: 15 # Critical path (Sep 2026): clawhub walk ~14 min ∥ github taps ~8 min, then skills.sh
# resolve ~6 min, then a bounded 8 min owner enrichment. 15 min cancelled every
# scheduled run for two months and left the live index frozen.
timeout-minutes: 50
environment: trusted-automation environment: trusted-automation
steps: steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
+5 -1
View File
@@ -41,6 +41,7 @@ import httpx
OUTPUT_PATH = os.path.join(REPO_ROOT, "website", "static", "api", "skills-index.json") OUTPUT_PATH = os.path.join(REPO_ROOT, "website", "static", "api", "skills-index.json")
INDEX_VERSION = 1 INDEX_VERSION = 1
CLAWHUB_ENRICH_BUDGET_SECONDS = 480
def _meta_to_dict(meta: SkillMeta) -> dict: def _meta_to_dict(meta: SkillMeta) -> dict:
@@ -316,7 +317,10 @@ def main():
print(f" Enriching {len(clawhub_metas)} ClawHub skills with owner handles...", print(f" Enriching {len(clawhub_metas)} ClawHub skills with owner handles...",
flush=True) flush=True)
enrich_start = time.time() enrich_start = time.time()
enriched = sources["clawhub"].enrich_owners(clawhub_metas, max_workers=30) # Best-effort: ~2s per detail call means the full catalog would take >1h;
# the un-enriched remainder ships without an owner link (see enrich_owners).
enriched = sources["clawhub"].enrich_owners(
clawhub_metas, max_workers=30, budget_seconds=CLAWHUB_ENRICH_BUDGET_SECONDS)
# Write enriched owner back into the index dicts. # Write enriched owner back into the index dicts.
meta_by_id = {m.identifier: m for m in clawhub_metas} meta_by_id = {m.identifier: m for m in clawhub_metas}
for s in clawhub_skills: for s in clawhub_skills:
@@ -39,7 +39,7 @@ class _FakeSource:
def search(self, query, limit=10): def search(self, query, limit=10):
return [_meta(f"{self._src}-{i}", self._src) for i in range(self._n)] return [_meta(f"{self._src}-{i}", self._src) for i in range(self._n)]
def enrich_owners(self, skills, max_workers=30): def enrich_owners(self, skills, max_workers=30, budget_seconds=None):
# No-op: fake source doesn't need owner enrichment. # No-op: fake source doesn't need owner enrichment.
return 0 return 0
+21
View File
@@ -1,5 +1,6 @@
#!/usr/bin/env python3 #!/usr/bin/env python3
import time
import unittest import unittest
from unittest.mock import patch from unittest.mock import patch
@@ -681,6 +682,26 @@ class TestFetchOwnerHandleRetry(unittest.TestCase):
self.assertEqual(call_count["n"], 6) self.assertEqual(call_count["n"], 6)
mock_sleep.assert_called() mock_sleep.assert_called()
@patch("tools.skills_hub_clawhub.httpx.get")
def test_enrich_owners_budget_stops_early_and_keeps_partial_results(self, mock_get):
"""An exhausted budget ends enrichment early (the un-enriched rest ships without an
owner) instead of walking every remaining skill — the unbounded walk over 78k skills
at ~2s each is what timed out the CI index build for two months."""
def slow_owner(url, *args, **kwargs):
time.sleep(0.05)
return _MockResponse(status_code=200,
json_data={"skill": {"slug": "s"}, "owner": {"handle": "eve"}})
mock_get.side_effect = slow_owner
skills = [SkillMeta(name=f"s{i}", description="", source="clawhub",
identifier=f"s{i}", trust_level="community") for i in range(200)]
enriched = self.src.enrich_owners(skills, max_workers=1, budget_seconds=0.3)
self.assertGreaterEqual(enriched, 1)
self.assertLess(enriched, 200)
self.assertEqual(enriched, sum(1 for s in skills if s.extra.get("owner") == "eve"))
self.assertLess(mock_get.call_count, 200)
if __name__ == "__main__": if __name__ == "__main__":
unittest.main() unittest.main()
+16 -3
View File
@@ -390,10 +390,15 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
time.sleep(delay) time.sleep(delay)
return None return None
def enrich_owners(self, skills: List[SkillMeta], max_workers: int = 30) -> int: def enrich_owners(self, skills: List[SkillMeta], max_workers: int = 30,
budget_seconds: Optional[float] = None) -> int:
"""Batch-fetch owner handles for ClawHub skills missing ``extra["owner"]`` """Batch-fetch owner handles for ClawHub skills missing ``extra["owner"]``
(in-place; returns the number enriched). For the offline index builder: (in-place; returns the number enriched).
the full 50k catalog takes ~5–10 min at 30 workers.
``budget_seconds`` makes this best-effort: the detail API answers in ~2s, so the
full catalog (78k+ skills) needs well over an hour at 30 workers — unbounded, it
was the phase that pushed the index build past its CI timeout for two months.
Skills left un-enriched simply ship without a "View source" owner link.
Safety rails: aborts after 50 consecutive failures (systemic outage), Safety rails: aborts after 50 consecutive failures (systemic outage),
per-request 429 backoff, progress log every 1000 skills. per-request 429 backoff, progress log every 1000 skills.
@@ -403,6 +408,7 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
return 0 return 0
enriched = consecutive_failures = processed = 0 enriched = consecutive_failures = processed = 0
max_consecutive_failures = 50 max_consecutive_failures = 50
deadline = time.monotonic() + budget_seconds if budget_seconds is not None else None
from concurrent.futures import ThreadPoolExecutor, as_completed from concurrent.futures import ThreadPoolExecutor, as_completed
with ThreadPoolExecutor(max_workers=max_workers) as pool: with ThreadPoolExecutor(max_workers=max_workers) as pool:
@@ -410,6 +416,13 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
for future in as_completed(futures): for future in as_completed(futures):
meta = futures[future] meta = futures[future]
processed += 1 processed += 1
if deadline is not None and time.monotonic() > deadline:
logger.warning("ClawHub owner enrichment: budget of %.0fs exhausted after %d/%d "
"(%d enriched) — shipping the rest without owner handles.",
budget_seconds, processed, len(needs_enrichment), enriched)
for f in futures:
f.cancel()
break
try: try:
handle = future.result() handle = future.result()
except Exception: except Exception: