refactor(skills_hub): per-module compaction pass — dedupe builders, collapse defensive layers, tighten docstrings

This commit is contained in:
Teknium
2026-09-02 21:42:18 -07:00
parent 5c1b26f01f
commit d8f8fb58f6
8 changed files with 473 additions and 1267 deletions
+60 -184
View File
@@ -33,28 +33,17 @@ def _search_score(query: str, meta: SkillMeta) -> int:
query_norm = query.strip().lower()
if not query_norm:
return 1
identifier = (meta.identifier or "").lower()
name = (meta.name or "").lower()
identifier, name = (meta.identifier or "").lower(), (meta.name or "").lower()
description = (meta.description or "").lower()
query_terms = _query_terms(query_norm)
identifier_terms = _query_terms(identifier)
name_terms = _query_terms(name)
normalized_identifier = " ".join(identifier_terms)
normalized_name = " ".join(name_terms)
query_terms, identifier_terms, name_terms = _query_terms(query_norm), _query_terms(identifier), _query_terms(name)
normalized_identifier, normalized_name = " ".join(identifier_terms), " ".join(name_terms)
checks = (
(140, query_norm == identifier),
(130, query_norm == name),
(125, normalized_identifier == query_norm),
(120, normalized_name == query_norm),
(95, normalized_identifier.startswith(query_norm)),
(90, normalized_name.startswith(query_norm)),
(140, query_norm == identifier), (130, query_norm == name),
(125, normalized_identifier == query_norm), (120, normalized_name == query_norm),
(95, normalized_identifier.startswith(query_norm)), (90, normalized_name.startswith(query_norm)),
(70, bool(query_terms) and identifier_terms[: len(query_terms)] == query_terms),
(65, bool(query_terms) and name_terms[: len(query_terms)] == query_terms),
(40, query_norm in identifier),
(35, query_norm in name),
(10, query_norm in description),
(40, query_norm in identifier), (35, query_norm in name), (10, query_norm in description),
)
score = sum(points for points, hit in checks if hit)
for term in query_terms:
@@ -68,24 +57,20 @@ def _first_str(*values: Any) -> Optional[str]:
class ClawHubSource(GuardedFetchMixin, SkillSource):
"""ClawHub (clawhub.ai) HTTP API. Every skill is community trust — the
ClawHavoc incident (341 malicious skills, Feb 2026) showed their vetting
is insufficient."""
"""ClawHub (clawhub.ai) HTTP API. Every skill is community trust — the ClawHavoc
incident (341 malicious skills, Feb 2026) showed their vetting is insufficient."""
SOURCE_ID = "clawhub"
BASE_URL = "https://clawhub.ai/api/v1"
# Wall-clock budget for a full catalog walk: 50k+ skills, sequential
# (~250 requests each under timeout=30), so unbounded it blocks for minutes.
CATALOG_WALK_BUDGET_SECONDS = 12
_SLUG_RE = re.compile(r"[A-Za-z0-9][A-Za-z0-9._-]*$")
_query_terms = staticmethod(_query_terms)
_dedupe_results = staticmethod(_dedupe_results)
_search_score = staticmethod(_search_score)
# -- payload helpers ---------------------------------------------------
_get_json = staticmethod(_get_json)
@staticmethod
def _normalize_tags(tags: Any) -> List[str]:
@@ -113,9 +98,7 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
@staticmethod
def _owner_from_payload(data: Optional[Dict[str, Any]]) -> Optional[str]:
if not isinstance(data, dict):
return None
owner = data.get("owner")
owner = data.get("owner") if isinstance(data, dict) else None
if isinstance(owner, dict):
owner = owner.get("handle")
return owner.strip() if isinstance(owner, str) and owner.strip() else None
@@ -137,11 +120,8 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
return SkillMeta(
name=item.get("displayName") or item.get("name") or slug,
description=item.get("summary") or item.get("description") or "",
source="clawhub",
identifier=slug,
trust_level="community",
tags=cls._normalize_tags(item.get("tags", [])),
extra={"owner": owner} if owner else {},
source="clawhub", identifier=slug, trust_level="community",
tags=cls._normalize_tags(item.get("tags", [])), extra={"owner": owner} if owner else {},
)
def _skill_detail(self, identifier: str) -> Optional[Tuple[str, Dict[str, Any]]]:
@@ -156,64 +136,38 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
return None
return slug, data
# -- search / ranking --------------------------------------------------
def _exact_slug_meta(self, query: str) -> Optional[SkillMeta]:
query = query.strip()
parsed = self._parse_identifier(query)
query_terms = _query_terms(query)
candidates: List[str] = []
if parsed:
candidates.append(parsed[0])
elif "/" not in query and self._SLUG_RE.fullmatch(query):
candidates.append(query)
parsed, query_terms = self._parse_identifier(query), _query_terms(query)
candidates = [parsed[0]] if parsed else [query] if "/" not in query and self._SLUG_RE.fullmatch(query) else []
if query_terms:
base_slug = "-".join(query_terms)
if len(query_terms) >= 2:
candidates.extend(
f"{base_slug}-{suffix}" for suffix in ("agent", "skill", "tool", "assistant", "playbook")
)
candidates.extend(f"{base_slug}-{s}" for s in ("agent", "skill", "tool", "assistant", "playbook"))
candidates.append(base_slug)
for candidate in dict.fromkeys(candidates):
meta = self.inspect(candidate)
if meta:
return meta
return None
return next((m for m in map(self.inspect, dict.fromkeys(candidates)) if m), None)
def _finalize_search_results(self, query: str, results: List[SkillMeta], limit: int) -> List[SkillMeta]:
query_norm = query.strip()
if not query_norm:
return _dedupe_results(results)[:limit]
filtered = [meta for meta in results if _search_score(query_norm, meta) > 0]
filtered.sort(key=lambda meta: (-_search_score(query_norm, meta), meta.name.lower(), meta.identifier.lower()))
filtered = _dedupe_results(filtered)
exact = self._exact_slug_meta(query_norm)
if exact:
filtered = [meta for meta in filtered if _search_score(query_norm, meta) >= 20]
filtered = _dedupe_results([exact] + filtered)
if filtered:
filtered = _dedupe_results([exact] + [m for m in filtered if _search_score(query_norm, m) >= 20])
if filtered or re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._/-]*", query_norm):
return filtered[:limit]
if re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._/-]*", query_norm):
return []
return _dedupe_results(results)[:limit]
def search(self, query: str, limit: int = 10) -> List[SkillMeta]:
query = query.strip()
if query:
if len(_query_terms(query)) >= 2:
direct = self._exact_slug_meta(query)
if direct:
return [direct]
results = self._search_catalog(query, limit=limit)
if results:
return results
@@ -222,7 +176,7 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
# is returned whole (caller paginates); on a cold cache the walk is
# bounded to `limit` so browse renders page one without walking
# 50k+ skills (max_items=0 = unbounded, offline index builder only).
catalog = self._load_catalog_index(max_items=limit if limit > 0 else 0)
catalog = self._load_catalog_index(max_items=max(limit, 0))
if catalog:
deduped = _dedupe_results(catalog)
return deduped[:limit] if limit > 0 else deduped
@@ -232,16 +186,11 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
cached = _cached_metas(cache_key)
if cached is not None:
return self._finalize_search_results(query, cached, limit)
data = self._get_json(f"{self.BASE_URL}/skills", timeout=15,
params={"search": query, "limit": limit})
if data is None:
return []
data = self._get_json(f"{self.BASE_URL}/skills", timeout=15, params={"search": query, "limit": limit})
skills_data = data.get("items", data) if isinstance(data, dict) else data
if not isinstance(skills_data, list):
return []
results = [m for m in (self._item_to_meta(item) for item in skills_data[:limit]) if m]
results = [m for m in map(self._item_to_meta, skills_data[:limit]) if m]
final_results = self._finalize_search_results(query, results, limit)
_cache_metas(cache_key, final_results)
return final_results
@@ -259,31 +208,22 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
if not raw:
return None
had_at = raw.startswith("@")
ident = raw[1:] if had_at else raw
if ident.startswith("clawhub/"):
ident = ident[len("clawhub/"):]
parts = [part for part in ident.split("/") if part]
owner = slug = None
parts = [part for part in raw.removeprefix("@").removeprefix("clawhub/").split("/") if part]
if len(parts) == 1:
slug = parts[0]
elif len(parts) == 2 and had_at:
owner, slug = parts
elif len(parts) == 3 and parts[1].lower() == "skills":
owner, _, slug = parts
owner, slug = None, parts[0]
elif (len(parts) == 2 and had_at) or (len(parts) == 3 and parts[1].lower() == "skills"):
owner, slug = parts[0], parts[-1]
else:
return None
if not cls._SLUG_RE.fullmatch(slug) or (owner is not None and not cls._SLUG_RE.fullmatch(owner)):
return None
return slug, owner
# -- fetch / inspect ---------------------------------------------------
def fetch(self, identifier: str) -> Optional[SkillBundle]:
detail = self._skill_detail(identifier)
if detail is None:
return None
slug, skill_data = detail
latest_version = self._resolve_latest_version(slug, skill_data)
if not latest_version:
logger.warning("ClawHub fetch failed for %s: could not resolve latest version", slug)
@@ -296,26 +236,15 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
version_data = self._get_json(f"{self.BASE_URL}/skills/{slug}/versions/{latest_version}")
if isinstance(version_data, dict):
files = self._extract_files(version_data) or files
if "SKILL.md" not in files:
nested = version_data.get("version", {})
if isinstance(nested, dict):
files = self._extract_files(nested) or files
nested = version_data.get("version", {})
if "SKILL.md" not in files and isinstance(nested, dict):
files = self._extract_files(nested) or files
if "SKILL.md" not in files:
logger.warning(
"ClawHub fetch for %s resolved version %s but could not retrieve file content",
slug,
latest_version,
"ClawHub fetch for %s resolved version %s but could not retrieve file content", slug, latest_version,
)
return None
return SkillBundle(
name=slug,
files=files,
source="clawhub",
identifier=slug,
trust_level="community",
)
return SkillBundle(name=slug, files=files, source="clawhub", identifier=slug, trust_level="community")
def inspect(self, identifier: str) -> Optional[SkillMeta]:
detail = self._skill_detail(identifier)
@@ -329,11 +258,9 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
cached = _cached_metas(cache_key)
if cached is not None:
return cached[:limit]
catalog = self._load_catalog_index()
if not catalog:
return []
results = self._finalize_search_results(query, catalog, limit)
_cache_metas(cache_key, results)
return results
@@ -352,7 +279,6 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
cached = _cached_metas(cache_key)
if cached is not None:
return cached
cursor: Optional[str] = None
results: List[SkillMeta] = []
seen: set[str] = set()
@@ -362,57 +288,42 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
# (max_items=0) must walk everything or it trips the deploy health floor.
deadline = time.monotonic() + self.CATALOG_WALK_BUDGET_SECONDS if max_items > 0 else None
partial = False
for _ in range(750):
if deadline is not None and time.monotonic() > deadline:
partial = True
break
params: Dict[str, Any] = {"limit": 200}
if cursor:
params["cursor"] = cursor
params: Dict[str, Any] = {"limit": 200, "cursor": cursor} if cursor else {"limit": 200}
data = self._get_json(f"{self.BASE_URL}/skills", timeout=30, params=params)
items = data.get("items", []) if isinstance(data, dict) else []
if not isinstance(items, list) or not items:
break
for item in items:
slug = item.get("slug")
if not isinstance(slug, str) or not slug or slug in seen:
continue
seen.add(slug)
meta = self._item_to_meta(item)
if meta:
results.append(meta)
if isinstance(slug, str) and slug and slug not in seen:
seen.add(slug)
meta = self._item_to_meta(item)
if meta:
results.append(meta)
cursor = data.get("nextCursor") if isinstance(data, dict) else None
if not isinstance(cursor, str) or not cursor:
break
if max_items > 0 and len(results) >= max_items:
partial = True
break
if not partial:
_cache_metas(cache_key, results)
return results
_get_json = staticmethod(_get_json)
def _resolve_latest_version(self, slug: str, skill_data: Dict[str, Any]) -> Optional[str]:
latest = skill_data.get("latestVersion")
tags = skill_data.get("tags")
latest, tags = skill_data.get("latestVersion"), skill_data.get("tags")
version = _first_str(
latest.get("version") if isinstance(latest, dict) else None,
tags.get("latest") if isinstance(tags, dict) else None,
)
if version:
return version
versions_data = self._get_json(f"{self.BASE_URL}/skills/{slug}/versions")
if isinstance(versions_data, list) and versions_data and isinstance(versions_data[0], dict):
return _first_str(versions_data[0].get("version"))
return None
vd = self._get_json(f"{self.BASE_URL}/skills/{slug}/versions")
return _first_str(vd[0].get("version")) if isinstance(vd, list) and vd and isinstance(vd[0], dict) else None
def _fetch_owner_handle(self, slug: str) -> Optional[str]:
"""Owner handle from the detail API (the listing API lacks it), or None.
@@ -423,7 +334,6 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
"""
url = f"{self.BASE_URL}/skills/{slug}"
max_attempts = 3
for attempt in range(max_attempts):
delay = 2.0 * (2 ** attempt)
try:
@@ -438,9 +348,8 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
return None
return self._owner_from_payload(self._coerce_skill_payload(raw))
if resp.status_code == 429:
retry_after_raw = resp.headers.get("Retry-After")
try:
delay = float(retry_after_raw) if retry_after_raw else delay
delay = float(resp.headers.get("Retry-After") or delay)
except (TypeError, ValueError):
pass
reason = "HTTP 429"
@@ -450,12 +359,9 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
return None # 4xx (non-429): doesn't exist / bad request
if attempt >= max_attempts - 1:
return None
logger.debug(
"_fetch_owner_handle(%s): %s on attempt %d/%d, retrying in %.1fs",
slug, reason, attempt + 1, max_attempts, delay,
)
logger.debug("_fetch_owner_handle(%s): %s on attempt %d/%d, retrying in %.1fs",
slug, reason, attempt + 1, max_attempts, delay)
time.sleep(delay)
return None
def enrich_owners(self, skills: List[SkillMeta], max_workers: int = 30) -> int:
@@ -466,13 +372,9 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
Safety rails: aborts after 50 consecutive failures (systemic outage),
per-request 429 backoff, progress log every 1000 skills.
"""
needs_enrichment = [
s for s in skills
if s.source == "clawhub" and not (s.extra or {}).get("owner")
]
needs_enrichment = [s for s in skills if s.source == "clawhub" and not (s.extra or {}).get("owner")]
if not needs_enrichment:
return 0
enriched = consecutive_failures = processed = 0
max_consecutive_failures = 50
@@ -486,60 +388,44 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
handle = future.result()
except Exception:
handle = None
consecutive_failures = 0 if handle else consecutive_failures + 1
if handle:
if not meta.extra:
meta.extra = {}
meta.extra = meta.extra or {}
meta.extra["owner"] = handle
enriched += 1
consecutive_failures = 0
else:
consecutive_failures += 1
if processed % 1000 == 0:
logger.info(
"ClawHub owner enrichment: %d/%d processed, %d enriched",
processed, len(needs_enrichment), enriched,
)
logger.info("ClawHub owner enrichment: %d/%d processed, %d enriched",
processed, len(needs_enrichment), enriched)
if consecutive_failures >= max_consecutive_failures:
logger.warning(
"ClawHub owner enrichment: %d consecutive failures — "
"aborting early (%d/%d processed, %d enriched). "
"The ClawHub API may be down or rate-limited.",
max_consecutive_failures, processed,
len(needs_enrichment), enriched,
max_consecutive_failures, processed, len(needs_enrichment), enriched,
)
for f in futures:
f.cancel()
break
return enriched
def _extract_files(self, version_data: Dict[str, Any]) -> Dict[str, str]:
files: Dict[str, str] = {}
file_list = version_data.get("files")
if isinstance(file_list, dict):
return {k: v for k, v in file_list.items() if isinstance(v, str)}
if not isinstance(file_list, list):
return files
for file_meta in file_list:
for file_meta in file_list if isinstance(file_list, list) else []:
if not isinstance(file_meta, dict):
continue
fname = file_meta.get("path") or file_meta.get("name")
fname, inline = file_meta.get("path") or file_meta.get("name"), file_meta.get("content")
raw_url = file_meta.get("rawUrl") or file_meta.get("downloadUrl") or file_meta.get("url")
if not fname or not isinstance(fname, str):
continue
inline_content = file_meta.get("content")
if isinstance(inline_content, str):
files[fname] = inline_content
continue
raw_url = file_meta.get("rawUrl") or file_meta.get("downloadUrl") or file_meta.get("url")
if isinstance(raw_url, str) and raw_url.startswith("http"):
if isinstance(inline, str):
files[fname] = inline
elif isinstance(raw_url, str) and raw_url.startswith("http"):
content = self._fetch_text(raw_url)
if content is not None:
files[fname] = content
return files
def _download_zip(self, slug: str, version: str) -> Dict[str, str]:
@@ -551,28 +437,20 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
max_retries = 3
for attempt in range(max_retries):
try:
resp = httpx.get(
f"{self.BASE_URL}/download",
params={"slug": slug, "version": version},
timeout=30,
follow_redirects=True,
)
resp = httpx.get(f"{self.BASE_URL}/download", params={"slug": slug, "version": version},
timeout=30, follow_redirects=True)
if resp.status_code == 429:
try:
retry_after = int(resp.headers.get("retry-after", "5"))
retry_after = min(int(resp.headers.get("retry-after", "5")), 15) # Cap wait time
except (ValueError, TypeError):
retry_after = 5
retry_after = min(retry_after, 15) # Cap wait time
logger.debug(
"ClawHub download rate-limited for %s, retrying in %ds (attempt %d/%d)",
slug, retry_after, attempt + 1, max_retries,
)
logger.debug("ClawHub download rate-limited for %s, retrying in %ds (attempt %d/%d)",
slug, retry_after, attempt + 1, max_retries)
time.sleep(retry_after)
continue
if resp.status_code != 200:
logger.debug("ClawHub ZIP download for %s v%s returned %s", slug, version, resp.status_code)
return files
with zipfile.ZipFile(io.BytesIO(resp.content)) as zf:
for info in zf.infolist():
if info.is_dir():
@@ -590,13 +468,11 @@ class ClawHubSource(GuardedFetchMixin, SkillSource):
except (UnicodeDecodeError, KeyError):
logger.debug("Skipping non-text file in ZIP: %s", name)
return files
except zipfile.BadZipFile:
logger.warning("ClawHub returned invalid ZIP for %s v%s", slug, version)
return files
except httpx.HTTPError as exc:
logger.debug("ClawHub ZIP download failed for %s v%s: %s", slug, version, exc)
return files
logger.debug("ClawHub ZIP download exhausted retries for %s v%s", slug, version)
return files
+115 -274
View File
@@ -20,19 +20,13 @@ from tools.skills_hub_models import (
logger = logging.getLogger("tools.skills_hub")
# Maps a GitHub tap repo (owner/repo) to the provider label used by the
# docs-site catalog (website/scripts/extract-skills.py::GITHUB_TAP_LABELS).
# The runtime index collapses every tap into source="github"; the label in
# ``extra.provider`` keeps per-tap identity searchable/filterable without
# disturbing the dedup / floor / index-skip logic keyed on the bare source id.
# GitHub tap repo (owner/repo) -> provider label used by the docs-site catalog
# (website/scripts/extract-skills.py::GITHUB_TAP_LABELS). The runtime index collapses every tap into
# source="github"; ``extra.provider`` keeps per-tap identity searchable/filterable without disturbing
# dedup / floor / index-skip logic keyed on the bare source id.
GITHUB_TAP_PROVIDERS = {
"openai/skills": "OpenAI",
"anthropics/skills": "Anthropic",
"huggingface/skills": "HuggingFace",
"nvidia/skills": "NVIDIA",
"voltagent/awesome-agent-skills": "VoltAgent",
"garrytan/gstack": "gstack",
"openai/skills": "OpenAI", "anthropics/skills": "Anthropic", "huggingface/skills": "HuggingFace",
"nvidia/skills": "NVIDIA", "voltagent/awesome-agent-skills": "VoltAgent", "garrytan/gstack": "gstack",
"minimax-ai/cli": "MiniMax",
}
@@ -40,20 +34,19 @@ GITHUB_TAP_PROVIDERS = {
# they narrow merged results to GitHub-tap skills carrying that ``extra.provider``.
_PROVIDER_FILTER_VALUES = frozenset(v.lower() for v in GITHUB_TAP_PROVIDERS.values())
_API = "https://api.github.com/repos"
_ACCEPT_JSON = "application/vnd.github.v3+json"
def github_provider_for(repo: str) -> Optional[str]:
"""Provider label for an ``owner/repo`` tap (case-insensitive), or None."""
if not repo:
return None
return GITHUB_TAP_PROVIDERS.get(repo.strip().lower())
return GITHUB_TAP_PROVIDERS.get(repo.strip().lower()) if repo else None
def _filter_results_by_provider(results: List[SkillMeta], provider: str) -> List[SkillMeta]:
"""Keep only results whose ``extra.provider`` matches ``provider``.
An explicit provider filter (``--source nvidia``) narrows to exactly that
provider — the official catalog is NOT injected the way unfiltered browse does.
"""
"""Keep only results whose ``extra.provider`` matches ``provider``. An explicit provider filter
(``--source nvidia``) narrows to exactly that provider — the official catalog is NOT injected the
way unfiltered browse does."""
want = provider.strip().lower()
return [r for r in results if str((r.extra or {}).get("provider", "")).lower() == want]
@@ -65,17 +58,9 @@ def _is_rate_limit_response(resp: httpx.Response) -> bool:
)
# ---------------------------------------------------------------------------
# GitHub Authentication
# ---------------------------------------------------------------------------
class GitHubAuth:
"""GitHub API authentication, tried in priority order:
1. GITHUB_TOKEN / GH_TOKEN (PAT)
2. `gh auth token` (gh CLI)
3. GitHub App JWT + installation token
4. Unauthenticated (60 req/hr, public repos only)
"""
"""GitHub API authentication, tried in priority order: GITHUB_TOKEN / GH_TOKEN (PAT), `gh auth token`
(gh CLI), GitHub App JWT + installation token, then unauthenticated (60 req/hr, public repos only)."""
def __init__(self):
self._cached_token: Optional[str] = None
@@ -84,10 +69,7 @@ class GitHubAuth:
def get_headers(self) -> Dict[str, str]:
token = self._resolve_token()
headers = {"Accept": "application/vnd.github.v3+json"}
if token:
headers["Authorization"] = f"token {token}"
return headers
return {"Accept": _ACCEPT_JSON, **({"Authorization": f"token {token}"} if token else {})}
def is_authenticated(self) -> bool:
return self._resolve_token() is not None
@@ -98,11 +80,8 @@ class GitHubAuth:
return self._cached_method or "anonymous"
def _resolve_token(self) -> Optional[str]:
if self._cached_token and (
self._cached_method != "github-app" or time.time() < self._app_token_expiry
):
if self._cached_token and (self._cached_method != "github-app" or time.time() < self._app_token_expiry):
return self._cached_token
for method, resolve in (
("pat", self._try_pat), ("gh-cli", self._try_gh_cli), ("github-app", self._try_github_app),
):
@@ -112,7 +91,6 @@ class GitHubAuth:
if method == "github-app":
self._app_token_expiry = time.time() + 3500 # ~58 min (tokens last 1 hour)
return token
self._cached_method = "anonymous"
return None
@@ -125,10 +103,8 @@ class GitHubAuth:
def _try_gh_cli(self) -> Optional[str]:
try:
result = subprocess.run(
["gh", "auth", "token"],
capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5,
stdin=subprocess.DEVNULL,
creationflags=windows_hide_flags(),
["gh", "auth", "token"], capture_output=True, text=True, encoding='utf-8', errors='replace',
timeout=5, stdin=subprocess.DEVNULL, creationflags=windows_hide_flags(),
)
if result.returncode == 0 and result.stdout.strip():
return result.stdout.strip()
@@ -138,19 +114,15 @@ class GitHubAuth:
def _try_github_app(self) -> Optional[str]:
from agent.secret_scope import get_secret
app_id = get_secret("GITHUB_APP_ID")
key_path = get_secret("GITHUB_APP_PRIVATE_KEY_PATH")
app_id, key_path = get_secret("GITHUB_APP_ID"), get_secret("GITHUB_APP_PRIVATE_KEY_PATH")
installation_id = get_secret("GITHUB_APP_INSTALLATION_ID")
if not all([app_id, key_path, installation_id]):
return None
try:
import jwt # PyJWT
except ImportError:
logger.debug("PyJWT not installed, skipping GitHub App auth")
return None
try:
key_file = Path(key_path)
if not key_file.exists():
@@ -162,27 +134,19 @@ class GitHubAuth:
)
resp = httpx.post(
f"https://api.github.com/app/installations/{installation_id}/access_tokens",
headers={"Authorization": f"Bearer {encoded_jwt}", "Accept": "application/vnd.github.v3+json"},
timeout=10,
headers={"Authorization": f"Bearer {encoded_jwt}", "Accept": _ACCEPT_JSON}, timeout=10,
)
if resp.status_code == 201:
return resp.json().get("token")
except Exception as e:
logger.debug("GitHub App auth failed: %s", e)
return None
# ---------------------------------------------------------------------------
# GitHub source adapter
# ---------------------------------------------------------------------------
def _split_repo_id(identifier: str) -> Optional[Tuple[str, str]]:
"""``owner/repo/path/to/skill`` -> ``(owner/repo, path/to/skill)``; None when too short."""
parts = identifier.split("/", 2)
if len(parts) < 3:
return None
return f"{parts[0]}/{parts[1]}", parts[2]
return (f"{parts[0]}/{parts[1]}", parts[2]) if len(parts) >= 3 else None
def _skip_bundle_file(rel_path: str) -> bool:
@@ -192,32 +156,27 @@ def _skip_bundle_file(rel_path: str) -> bool:
def _tree_members(entries: List[dict], prefix: str):
"""``(rel_path, item_path, is_regular_blob)`` for every git-tree entry under ``prefix``.
Symlinks (mode 120000) and non-blobs report ``is_regular_blob=False`` so callers
can reject a SKILL.md-linked symlink instead of silently following it.
"""
"""``(rel_path, item_path, is_regular_blob)`` for every git-tree entry under ``prefix``. Symlinks
(mode 120000) and non-blobs report ``is_regular_blob=False`` so callers can reject a SKILL.md-linked
symlink instead of silently following it."""
for item in entries:
item_path = item.get("path", "")
if item_path.startswith(prefix):
regular = item.get("type") == "blob" and item.get("mode") != "120000"
yield item_path[len(prefix):], item_path, regular
yield item_path[len(prefix):], item_path, item.get("type") == "blob" and item.get("mode") != "120000"
class GitHubSource(SkillSource):
"""Fetch skills from GitHub repos via the Contents API."""
DEFAULT_TAPS = [
# openai/skills keeps its content under skills/.curated/ and
# skills/.system/; _list_skills_in_repo skips "."/"_" directories,
# so both entries point at the inner paths directly.
# openai/skills keeps content under skills/.curated/ + skills/.system/; _list_skills_in_repo
# skips "."/"_" directories, so both entries point at the inner paths.
{"repo": "openai/skills", "path": "skills/.curated/"},
{"repo": "openai/skills", "path": "skills/.system/"},
{"repo": "anthropics/skills", "path": "skills/"},
{"repo": "huggingface/skills", "path": "skills/"},
# NVIDIA-verified skills (CUDA-X, NeMo, cuOpt, ...), each with a
# signed skill.oms.sig + governance card; `trusted` via
# tools/skills_guard.py::TRUSTED_REPOS.
# NVIDIA-verified skills (CUDA-X, NeMo, cuOpt, ...), each with a signed skill.oms.sig
# + governance card; `trusted` via tools/skills_guard.py::TRUSTED_REPOS.
{"repo": "NVIDIA/skills", "path": "skills/"},
{"repo": "garrytan/gstack", "path": ""},
]
@@ -227,9 +186,7 @@ class GitHubSource(SkillSource):
def __init__(self, auth: GitHubAuth, extra_taps: Optional[List[Dict]] = None):
self.auth = auth
self.taps = list(self.DEFAULT_TAPS)
if extra_taps:
self.taps.extend(extra_taps)
self.taps = list(self.DEFAULT_TAPS) + list(extra_taps or [])
# Per-instance repo -> (default_branch, tree_entries); lives for one
# search/install flow so repeated tree lookups cost no API calls.
self._tree_cache: Dict[str, Tuple[str, List[dict]]] = {}
@@ -239,22 +196,18 @@ class GitHubSource(SkillSource):
self._rate_limited: bool = False
@property
def is_rate_limited(self) -> bool:
"""Whether GitHub API rate limit was hit during operations."""
def is_rate_limited(self) -> bool: # whether the GitHub API rate limit was hit during operations
return self._rate_limited
def trust_level_for(self, identifier: str) -> str:
# identifier format: "owner/repo/path/to/skill"
parts = identifier.split("/", 2)
if len(parts) >= 2 and f"{parts[0]}/{parts[1]}" in TRUSTED_REPOS:
return "trusted"
return "community"
return "trusted" if len(parts) >= 2 and f"{parts[0]}/{parts[1]}" in TRUSTED_REPOS else "community"
def search(self, query: str, limit: int = 10) -> List[SkillMeta]:
"""Substring-match all taps; dedupe by identifier preferring higher trust."""
results: List[SkillMeta] = []
query_lower = query.lower()
for tap in self.taps:
try:
for skill in self._list_skills_in_repo(tap["repo"], tap.get("path", "")):
@@ -262,22 +215,17 @@ class GitHubSource(SkillSource):
results.append(skill)
except Exception as e:
logger.debug("Failed to search %s: %s", tap['repo'], e)
continue
return _dedupe_by_trust(results)[:limit]
def fetch(self, identifier: str) -> Optional[SkillBundle]:
"""Download a skill; identifier format: "owner/repo/path/to/skill-dir"."""
split = _split_repo_id(identifier)
if split is None:
if (split := _split_repo_id(identifier)) is None:
return None
repo, skill_path = split
skill_dir = skill_path.rstrip("/")
# Resolve the tree FIRST so every byte fetch — SKILL.md included — is
# pinned to the same revision; an unpinned /contents fetch floats to
# HEAD and can serve bytes newer than the tree the paths were
# validated against (TOCTOU). Idempotent + cached.
# Resolve the tree FIRST so every byte fetch — SKILL.md included — is pinned to the
# same revision; an unpinned /contents fetch floats to HEAD and can serve bytes newer
# than the tree the paths were validated against (TOCTOU). Idempotent + cached.
tree = self._get_repo_tree(repo)
pinned_ref = self._tree_revisions.get(repo)
skill_md = self._fetch_file_content(repo, f"{skill_dir}/SKILL.md", ref=pinned_ref)
@@ -286,57 +234,39 @@ class GitHubSource(SkillSource):
referenced = _referenced_support_paths(skill_md)
if referenced is None:
return None
files: Dict[str, Union[str, bytes]] = {"SKILL.md": skill_md}
if tree is not None:
branch, entries = tree
if not self._collect_tree_files(repo, skill_dir, entries, pinned_ref, referenced, files):
if not self._collect_tree_files(repo, skill_dir, tree[1], pinned_ref, referenced, files):
return None
revision = pinned_ref or branch
revision = pinned_ref or tree[0]
else:
for rel_path in referenced:
content = self._fetch_file_bytes(repo, f"{skill_dir}/{rel_path}")
if content is None:
logger.warning("Failed to fetch referenced skill support "
"file; continuing without it: %s", rel_path)
continue
files[rel_path] = content
self._add_support_file(repo, f"{skill_dir}/{rel_path}", rel_path, files, rel_path)
revision = ""
url = f"https://github.com/{repo}/" + (f"tree/{revision}/{skill_path}" if revision else skill_path)
return SkillBundle(
name=skill_dir.split("/")[-1],
files=files,
source="github",
identifier=identifier,
trust_level=self.trust_level_for(identifier),
metadata={
"source_url": (
f"https://github.com/{repo}/tree/{revision}/{skill_path}"
if revision else f"https://github.com/{repo}/{skill_path}"
),
"source_revision": revision,
},
name=skill_dir.split("/")[-1], files=files, source="github", identifier=identifier,
trust_level=self.trust_level_for(identifier), metadata={"source_url": url, "source_revision": revision},
)
def _add_support_file(self, repo: str, item_path: str, rel_path: str, files: dict, shown: str, **kw) -> None:
"""Fetch one support file into ``files``; a failed fetch warns (naming ``shown``) and is skipped."""
content = self._fetch_file_bytes(repo, item_path, **kw)
if content is None:
logger.warning("Failed to fetch referenced skill support file; continuing without it: %s", shown)
else:
files[rel_path] = content
def _collect_tree_files(
self,
repo: str,
skill_path: str,
entries: List[dict],
ref: Optional[str],
referenced: set,
self, repo: str, skill_path: str, entries: List[dict], ref: Optional[str], referenced: set,
files: Dict[str, Union[str, bytes]],
) -> bool:
"""Download the FULL skill directory from the pinned tree into ``files``.
Link-driven fetching silently dropped support files under non-canonical
dirs (``reference/``, ``agents/``, root LICENSE); everything still goes
through quarantine + scan, and the scanner sees MORE this way.
Returns False (bundle rejected) on an unsafe path or a SKILL.md-linked
path that exists in the tree as a symlink/non-blob — that shape is an
escape attempt. A linked path that is simply absent is a dangling link
(repo-only dev tool, prose over-match): warn and install without it.
"""
"""Download the FULL skill directory from the pinned tree into ``files``. Link-driven fetching
silently dropped support files under non-canonical dirs (``reference/``, ``agents/``, root
LICENSE); everything still goes through quarantine + scan, and the scanner sees MORE this way.
Returns False (bundle rejected) on an unsafe path or a SKILL.md-linked path that exists in the
tree as a symlink/non-blob — that shape is an escape attempt. A linked path that is simply absent
is a dangling link (repo-only dev tool, prose over-match): warn and install without it."""
prefix = f"{skill_path}/"
symlinked: set = set()
for rel_path, item_path, regular in _tree_members(entries, prefix):
@@ -350,55 +280,31 @@ class GitHubSource(SkillSource):
except ValueError:
logger.warning("Rejected unsafe file path in skill bundle: %s", item_path)
return False
content = self._fetch_file_bytes(repo, item_path, ref=ref)
if content is None:
logger.warning("Failed to fetch referenced skill support "
"file; continuing without it: %s", item_path)
continue
files[rel_path] = content
self._add_support_file(repo, item_path, rel_path, files, item_path, ref=ref)
for rel_path in sorted(referenced):
if rel_path in symlinked:
logger.warning(
"Rejected non-regular referenced file in skill "
"bundle: %s%s", prefix, rel_path,
)
logger.warning("Rejected non-regular referenced file in skill bundle: %s%s", prefix, rel_path)
return False
if rel_path not in files:
logger.warning(
"Referenced skill support file is missing; "
"continuing without it: %s%s",
prefix, rel_path,
)
"Referenced skill support file is missing; continuing without it: %s%s", prefix, rel_path)
return True
def inspect(self, identifier: str) -> Optional[SkillMeta]:
"""Fetch just the SKILL.md metadata for preview."""
split = _split_repo_id(identifier)
if split is None:
if (split := _split_repo_id(identifier)) is None:
return None
repo, skill_path = split
skill_path = skill_path.rstrip("/")
repo, skill_path = split[0], split[1].rstrip("/")
content = self._fetch_file_content(repo, f"{skill_path}/SKILL.md")
if not content:
return None
fm = _parse_frontmatter(content)
tags = _hermes_tags(fm)
if not tags:
raw_tags = fm.get("tags", [])
tags = raw_tags if isinstance(raw_tags, list) else []
tags = _hermes_tags(fm) or (fm["tags"] if isinstance(fm.get("tags"), list) else [])
provider = github_provider_for(repo)
return SkillMeta(
name=fm.get("name", skill_path.split("/")[-1]),
description=str(fm.get("description", "")),
source="github",
identifier=identifier,
trust_level=self.trust_level_for(identifier),
repo=repo,
path=skill_path,
tags=[str(t) for t in tags],
name=fm.get("name", skill_path.split("/")[-1]), description=str(fm.get("description", "")),
source="github", identifier=identifier, trust_level=self.trust_level_for(identifier),
repo=repo, path=skill_path, tags=[str(t) for t in tags],
extra={"provider": provider} if provider else {},
)
@@ -410,105 +316,69 @@ class GitHubSource(SkillSource):
cached = _cached_metas(cache_key)
if cached is not None:
return cached
url = f"https://api.github.com/repos/{repo}/contents/{path.rstrip('/')}"
resp = self._github_get(url)
resp = self._github_get(f"{_API}/{repo}/contents/{path.rstrip('/')}")
if resp is None or resp.status_code != 200:
return []
entries = resp.json()
if not isinstance(entries, list):
return []
skills: List[SkillMeta] = []
groupings = self._get_skillsh_groupings(repo)
prefix = path.rstrip("/")
for entry in entries:
if entry.get("type") != "dir":
if entry.get("type") != "dir" or entry["name"].startswith((".", "_")):
continue
dir_name = entry["name"]
if dir_name.startswith((".", "_")):
continue
meta = self.inspect(f"{repo}/{prefix}/{dir_name}" if prefix else f"{repo}/{dir_name}")
if meta:
if groupings:
category = groupings.get(meta.name) or groupings.get(dir_name)
if category:
meta.extra["category"] = category
category = groupings and (groupings.get(meta.name) or groupings.get(dir_name))
if category:
meta.extra["category"] = category
skills.append(meta)
_cache_metas(cache_key, skills)
return skills
def _get_repo_tree(self, repo: str) -> Optional[Tuple[str, List[dict]]]:
"""Cached ``(default_branch, tree_entries)`` for a repo, or None.
One install may need the tree several times; caching saves the
``GET /repos/{repo}`` + ``GET .../git/trees/{branch}`` pair each time
(~12 of the 60/hr unauthenticated budget before).
"""
"""Cached ``(default_branch, tree_entries)`` for a repo, or None. One install may need the tree
several times; caching saves the ``GET /repos/{repo}`` + ``GET .../git/trees/{branch}`` pair each
time (~12 of the 60/hr unauthenticated budget before)."""
if repo in self._tree_cache:
return self._tree_cache[repo]
repo_data = self._github_json(f"https://api.github.com/repos/{repo}")
repo_data = self._github_json(f"{_API}/{repo}")
if repo_data is None:
return None
default_branch = repo_data.get("default_branch", "main")
tree_data = self._github_json(
f"https://api.github.com/repos/{repo}/git/trees/{default_branch}",
params={"recursive": "1"}, timeout=30.0,
f"{_API}/{repo}/git/trees/{default_branch}", params={"recursive": "1"}, timeout=30.0,
)
if tree_data is None:
return None
if tree_data.get("truncated"):
logger.debug("Git tree truncated for %s, cannot cache", repo)
return None
entries = tree_data.get("tree", [])
revision = tree_data.get("sha")
if isinstance(revision, str) and revision:
self._tree_revisions[repo] = revision
self._tree_cache[repo] = (default_branch, entries)
return (default_branch, entries)
if isinstance(tree_data.get("sha"), str) and tree_data["sha"]:
self._tree_revisions[repo] = tree_data["sha"]
self._tree_cache[repo] = tree = (default_branch, tree_data.get("tree", []))
return tree
def _github_json(self, url: str, **kwargs) -> Optional[dict]:
"""Decoded JSON body of a 200 ``_github_get`` (which flags rate-limit exhaustion), else None."""
resp = self._github_get(url, **kwargs)
if resp is None or resp.status_code != 200:
return None
try:
return resp.json()
return resp.json() if resp is not None and resp.status_code == 200 else None
except ValueError:
return None
def _check_rate_limit_response(self, resp: httpx.Response) -> None:
"""Flag the instance as rate-limited when GitHub returns 403 + exhausted quota."""
if _is_rate_limit_response(resp):
self._rate_limited = True
logger.warning(
"GitHub API rate limit exhausted (unauthenticated: 60 req/hr). "
"Set GITHUB_TOKEN or install the gh CLI to raise the limit to 5,000/hr."
)
def _github_get(
self,
url: str,
*,
params: Optional[Dict] = None,
headers: Optional[Dict] = None,
timeout: float = 15.0,
max_retries: int = 3,
self, url: str, *, params: Optional[Dict] = None, headers: Optional[Dict] = None,
timeout: float = 15.0, max_retries: int = 3,
) -> Optional[httpx.Response]:
"""GET against the GitHub API with retry/backoff on transient failures.
Returns the final response (caller inspects status) or None when every
attempt raised a transport error. Retries rate-limit 403/429 (waiting
until ``Retry-After``/``X-RateLimit-Reset`` when present, capped 60s —
one shared limit zeroes every GitHub tap at once during an index build),
5xx, and transport errors with exponential backoff. Terminal rate-limit
exhaustion flags the instance so an index build fails loud instead of
silently shipping zero GitHub skills.
"""
"""GET against the GitHub API with retry/backoff on transient failures. Returns the final
response (caller inspects status) or None when every attempt raised a transport error.
Retries rate-limit 403/429 (waiting until ``Retry-After`` / ``X-RateLimit-Reset`` when present,
capped 60s — one shared limit zeroes every GitHub tap at once during an index build), 5xx, and
transport errors with exponential backoff. Terminal rate-limit exhaustion flags the instance so
an index build fails loud instead of silently shipping zero GitHub skills."""
hdrs = headers if headers is not None else self.auth.get_headers()
backoff = 1.0
last_resp: Optional[httpx.Response] = None
@@ -516,13 +386,9 @@ class GitHubSource(SkillSource):
last_attempt = attempt >= max_retries - 1
wait = backoff
try:
resp = httpx.get(
url, params=params, headers=hdrs,
timeout=timeout, follow_redirects=True,
)
resp = httpx.get(url, params=params, headers=hdrs, timeout=timeout, follow_redirects=True)
except httpx.HTTPError as e:
logger.debug("GitHub GET %s failed (attempt %d/%d): %s",
url, attempt + 1, max_retries, e)
logger.debug("GitHub GET %s failed (attempt %d/%d): %s", url, attempt + 1, max_retries, e)
if last_attempt:
return None
else:
@@ -530,8 +396,12 @@ class GitHubSource(SkillSource):
if resp.status_code == 200:
return resp
if resp.status_code in (403, 429):
if not _is_rate_limit_response(resp) or last_attempt:
self._check_rate_limit_response(resp)
limited = _is_rate_limit_response(resp)
if not limited or last_attempt:
if limited: # terminal exhaustion: flag the instance so callers fail loud
self._rate_limited = True
logger.warning("GitHub API rate limit exhausted (unauthenticated: 60 req/hr). "
"Set GITHUB_TOKEN or install the gh CLI to raise the limit to 5,000/hr.")
return resp
reset = resp.headers.get("X-RateLimit-Reset", "")
retry_after = resp.headers.get("Retry-After", "")
@@ -541,35 +411,23 @@ class GitHubSource(SkillSource):
delta = float(reset) - time.time()
if 0 < delta <= 60.0:
wait = delta
logger.debug(
"GitHub rate limited on %s, waiting %.1fs (attempt %d/%d)",
url, wait, attempt + 1, max_retries,
)
logger.debug("GitHub rate limited on %s, waiting %.1fs (attempt %d/%d)",
url, wait, attempt + 1, max_retries)
elif not (500 <= resp.status_code < 600) or last_attempt:
return resp
time.sleep(wait)
backoff = min(backoff * 2, 30.0)
return last_resp
def _find_skill_in_repo_tree(self, repo: str, skill_name: str) -> Optional[str]:
"""Locate ``<skill_name>/SKILL.md`` anywhere in the repo tree (one API call).
Returns the full identifier (``repo/path/to/skill``) or None.
"""
cached = self._get_repo_tree(repo)
if cached is None:
"""Locate ``<skill_name>/SKILL.md`` anywhere in the repo tree (one API call); full identifier or None."""
if (cached := self._get_repo_tree(repo)) is None:
return None
_default_branch, tree_entries = cached
skill_md_suffix = f"/{skill_name}/SKILL.md"
for entry in tree_entries:
if entry.get("type") != "blob":
continue
for entry in cached[1]:
path = entry.get("path", "")
if path.endswith(skill_md_suffix) or path == f"{skill_name}/SKILL.md":
if entry.get("type") == "blob" and (path.endswith(skill_md_suffix) or path == skill_md_suffix[1:]):
return f"{repo}/{path[: -len('/SKILL.md')]}"
return None
def _fetch_file_content(self, repo: str, path: str, ref: Optional[str] = None) -> Optional[str]:
@@ -583,32 +441,20 @@ class GitHubSource(SkillSource):
def _fetch_file_bytes(self, repo: str, path: str, ref: Optional[str] = None) -> Optional[bytes]:
"""Fetch exact file bytes. ``ref`` pins to a tree SHA (see ``fetch`` on
the TOCTOU); None keeps the legacy unpinned behavior."""
encoded_path = quote(path, safe="/")
url = f"https://api.github.com/repos/{repo}/contents/{encoded_path}"
resp = self._github_get(
url,
params={"ref": ref} if ref else None,
f"{_API}/{repo}/contents/{quote(path, safe='/')}", params={"ref": ref} if ref else None,
headers={**self.auth.get_headers(), "Accept": "application/vnd.github.v3.raw"},
)
if resp is not None and resp.status_code == 200:
return resp.content
return None
return resp.content if resp is not None and resp.status_code == 200 else None
def _get_skillsh_groupings(self, repo: str) -> Optional[Dict[str, str]]:
"""Repo-root ``skills.sh.json`` groupings flattened to ``{skill_name: title}``.
``skills.sh.json`` is a cross-ecosystem standard
(``$schema: https://skills.sh/schemas/skills.sh.schema.json``); any tap
shipping it gets category pills for free. None when absent/unparsable.
Cached per repo on the instance.
"""
if repo in self._skillsh_groupings:
return self._skillsh_groupings[repo]
content = self._fetch_file_content(repo, "skills.sh.json")
groupings = self._parse_skillsh_groupings(content) if content else None
self._skillsh_groupings[repo] = groupings
return groupings
"""Repo-root ``skills.sh.json`` groupings flattened to ``{skill_name: title}``. ``skills.sh.json``
is a cross-ecosystem standard (``$schema: https://skills.sh/schemas/skills.sh.schema.json``); any
tap shipping it gets category pills for free. None when absent/unparsable; cached per repo."""
if repo not in self._skillsh_groupings:
content = self._fetch_file_content(repo, "skills.sh.json")
self._skillsh_groupings[repo] = self._parse_skillsh_groupings(content) if content else None
return self._skillsh_groupings[repo]
@staticmethod
def _parse_skillsh_groupings(content: str) -> Optional[Dict[str, str]]:
@@ -617,22 +463,17 @@ class GitHubSource(SkillSource):
data = json.loads(content)
except (json.JSONDecodeError, TypeError):
return None
if not isinstance(data, dict):
return None
groupings = data.get("groupings")
groupings = data.get("groupings") if isinstance(data, dict) else None
if not isinstance(groupings, list):
return None
mapping: Dict[str, str] = {}
for group in groupings:
if not isinstance(group, dict):
continue
title = group.get("title")
members = group.get("skills")
title, members = group.get("title"), group.get("skills")
if not isinstance(title, str) or not isinstance(members, list):
continue
for member in members:
if isinstance(member, str) and member:
mapping.setdefault(member, title) # first grouping wins
return mapping
+30 -93
View File
@@ -17,12 +17,8 @@ from agent.skill_utils import is_excluded_skill_path
from tools.skills_guard import ScanResult, content_hash
from tools.skills_hub_github import GitHubAuth
from tools.skills_hub_models import (
SkillBundle,
SkillSource,
_normalize_lock_install_path,
_validate_bundle_rel_path,
_validate_install_parent_path,
_validate_skill_name,
SkillBundle, SkillSource, _normalize_lock_install_path, _validate_bundle_rel_path,
_validate_install_parent_path, _validate_skill_name,
)
if TYPE_CHECKING: # origin class; runtime use is via the lazy origin import
@@ -49,15 +45,12 @@ def _resolve_lock_install_path(install_path: str, skill_name: str) -> Path:
"""
from tools.skills_hub import _skills_dir
normalized = _normalize_lock_install_path(install_path, skill_name)
skills_dir = _skills_dir()
target = skills_dir = _skills_dir()
skills_root = skills_dir.resolve()
target = skills_dir
for part in normalized.split("/"):
target = target / part
if _is_path_redirect(target):
raise ValueError(f"Unsafe install path: {install_path}")
target = target.resolve()
if target == skills_root or not target.is_relative_to(skills_root):
raise ValueError(f"Unsafe install path: {install_path}")
@@ -70,16 +63,11 @@ def quarantine_bundle(bundle: SkillBundle) -> Path:
ensure_hub_dirs()
skill_name = _validate_skill_name(bundle.name)
# Validate every path before touching disk so a bad member aborts cleanly.
validated_files = [
(_validate_bundle_rel_path(rel_path), file_content)
for rel_path, file_content in bundle.files.items()
]
validated_files = [(_validate_bundle_rel_path(rel_path), content) for rel_path, content in bundle.files.items()]
dest = _quarantine_dir() / skill_name
if dest.exists():
shutil.rmtree(dest)
dest.mkdir(parents=True)
for rel_path, file_content in validated_files:
file_dest = dest.joinpath(*rel_path.split("/"))
file_dest.parent.mkdir(parents=True, exist_ok=True)
@@ -87,7 +75,6 @@ def quarantine_bundle(bundle: SkillBundle) -> Path:
file_dest.write_bytes(file_content)
else:
file_dest.write_text(file_content, encoding="utf-8")
return dest
@@ -100,16 +87,13 @@ def _category_skill_dirs(directory: Path) -> List[str]:
``references/pkg/SKILL.md`` does not make the directory a category.
Shared with ``hermes_cli.skills_hub._existing_categories``.
"""
skill_dirs: List[str] = []
for entry in directory.iterdir():
if not entry.is_dir() or entry.name.startswith("."):
continue
if any(
return [
entry.name for entry in directory.iterdir()
if entry.is_dir() and not entry.name.startswith(".") and any(
not is_excluded_skill_path(skill_md.relative_to(directory), root=directory)
for skill_md in entry.rglob("SKILL.md")
):
skill_dirs.append(entry.name)
return skill_dirs
)
]
def _check_install_target(install_dir: Path) -> None:
@@ -127,38 +111,24 @@ def _check_install_target(install_dir: Path) -> None:
ancestor = install_dir.parent
while ancestor != skills_root and ancestor.is_relative_to(skills_root):
if (ancestor / "SKILL.md").is_file():
raise ValueError(
f"Refusing to install into '{ancestor.name}': it is an "
f"existing skill directory, not a category. Choose a "
f"different category."
)
raise ValueError(f"Refusing to install into '{ancestor.name}': it is an "
f"existing skill directory, not a category. Choose a different category.")
ancestor = ancestor.parent
if not install_dir.exists():
return
if not install_dir.is_dir():
raise ValueError(
f"Refusing to install: '{install_dir.name}' already exists "
f"and is not a directory. Remove it or choose a different "
f"skill name."
)
raise ValueError(f"Refusing to install: '{install_dir.name}' already exists "
f"and is not a directory. Remove it or choose a different skill name.")
if not (install_dir / "SKILL.md").exists():
skill_dirs_in = _category_skill_dirs(install_dir)
if skill_dirs_in:
raise ValueError(
f"Refusing to overwrite category directory '{install_dir}' "
f"which contains {len(skill_dirs_in)} skill(s): "
f"{', '.join(sorted(skill_dirs_in))}. "
f"Use a different --name or install into a subcategory."
)
raise ValueError(f"Refusing to overwrite category directory '{install_dir}' "
f"which contains {len(skill_dirs_in)} skill(s): {', '.join(sorted(skill_dirs_in))}. "
f"Use a different --name or install into a subcategory.")
def install_from_quarantine(
quarantine_path: Path,
skill_name: str,
category: str,
bundle: SkillBundle,
scan_result: ScanResult,
quarantine_path: Path, skill_name: str, category: str, bundle: SkillBundle, scan_result: ScanResult,
scan_provenance: Optional[Dict[str, Any]] = None,
) -> Path:
"""Move a scanned skill from quarantine into the skills directory."""
@@ -186,8 +156,7 @@ def install_from_quarantine(
"Skill '%s' has a large SKILL.md (%s chars). "
"Large skills consume significant context when loaded. "
"Consider asking the author to split it into smaller files.",
safe_skill_name,
f"{skill_size:,}",
safe_skill_name, f"{skill_size:,}",
)
# A symlink in the bundle would copy its target into skills/ and leak it
@@ -202,33 +171,21 @@ def install_from_quarantine(
install_dir.parent.mkdir(parents=True, exist_ok=True)
shutil.move(str(quarantine_path), str(install_dir))
installed_hash = content_hash(install_dir)
HubLockFile().record_install(
name=safe_skill_name,
source=bundle.source,
identifier=bundle.identifier,
trust_level=bundle.trust_level,
scan_verdict=scan_result.verdict,
skill_hash=installed_hash,
name=safe_skill_name, source=bundle.source, identifier=bundle.identifier, trust_level=bundle.trust_level,
scan_verdict=scan_result.verdict, skill_hash=installed_hash,
install_path=install_dir.resolve().relative_to(_skills_dir().resolve()).as_posix(),
files=list(bundle.files.keys()),
metadata=bundle.metadata,
files=list(bundle.files.keys()), metadata=bundle.metadata,
scan_provenance=scan_provenance or getattr(scan_result, "scan_provenance", None),
)
append_audit_log(
"INSTALL", safe_skill_name, bundle.source,
bundle.trust_level, scan_result.verdict,
installed_hash,
)
append_audit_log("INSTALL", safe_skill_name, bundle.source, bundle.trust_level, scan_result.verdict,
installed_hash)
try:
from tools.skill_usage import record_installed
record_installed(safe_skill_name)
except Exception:
logger.debug("Unable to record skill install lifecycle for %s", safe_skill_name, exc_info=True)
return install_dir
@@ -239,20 +196,16 @@ def uninstall_skill(skill_name: str) -> Tuple[bool, str]:
entry = lock.get_installed(skill_name)
if not entry:
return False, f"'{skill_name}' is not a hub-installed skill (may be a builtin)"
# The destructive boundary: whatever reaches rmtree MUST be inside
# SKILLS_DIR and MUST NOT be SKILLS_DIR itself (see _resolve_lock_install_path).
try:
install_path = _resolve_lock_install_path(entry.get("install_path", ""), skill_name)
except ValueError as exc:
return False, f"Refusing to uninstall '{skill_name}': {exc}"
if install_path.exists():
shutil.rmtree(install_path)
lock.record_uninstall(skill_name)
append_audit_log("UNINSTALL", skill_name, entry["source"], entry["trust_level"], "n/a", "user_request")
return True, f"Uninstalled '{skill_name}' from {entry['install_path']}"
@@ -266,10 +219,7 @@ def bundle_content_hash(bundle: SkillBundle) -> str:
path is hashed too so swapping contents between two files changes the hash.
"""
h = hashlib.sha256()
normalized = {
rel_path.replace("\\", "/"): content
for rel_path, content in bundle.files.items()
}
normalized = {rel_path.replace("\\", "/"): content for rel_path, content in bundle.files.items()}
for rel_path in sorted(normalized):
h.update(rel_path.encode("utf-8"))
h.update(b"\x00")
@@ -286,11 +236,8 @@ def _source_matches(source: SkillSource, source_name: str) -> bool:
def check_for_skill_updates(
name: Optional[str] = None,
*,
lock: Optional[HubLockFile] = None,
sources: Optional[List[SkillSource]] = None,
auth: Optional[GitHubAuth] = None,
name: Optional[str] = None, *, lock: Optional[HubLockFile] = None,
sources: Optional[List[SkillSource]] = None, auth: Optional[GitHubAuth] = None,
) -> List[dict]:
"""Check installed hub skills for upstream changes.
@@ -304,16 +251,13 @@ def check_for_skill_updates(
installed = lock.list_installed()
if name:
installed = [entry for entry in installed if entry.get("name") == name]
if sources is None:
sources = create_source_router(auth=auth)
results: List[dict] = []
for entry in installed:
identifier = entry.get("identifier", "")
source_name = entry.get("source", "")
identifier, source_name = entry.get("identifier", ""), entry.get("source", "")
row = {"name": entry.get("name", ""), "identifier": identifier, "source": source_name}
bundle = None
for src in filter(lambda s: _source_matches(s, source_name), sources):
try:
@@ -322,19 +266,12 @@ def check_for_skill_updates(
bundle = None
if bundle:
break
if not bundle:
results.append({**row, "status": "unavailable"})
continue
current_hash = entry.get("content_hash", "")
latest_hash = bundle_content_hash(bundle)
current_hash, latest_hash = entry.get("content_hash", ""), bundle_content_hash(bundle)
results.append({
**row,
"status": "up_to_date" if current_hash == latest_hash else "update_available",
"current_hash": current_hash,
"latest_hash": latest_hash,
"bundle": bundle,
**row, "status": "up_to_date" if current_hash == latest_hash else "update_available",
"current_hash": current_hash, "latest_hash": latest_hash, "bundle": bundle,
})
return results
+54 -116
View File
@@ -1,8 +1,5 @@
"""Skills Hub data models, path validators, and SKILL.md helpers.
Leaf module (no imports from tools.skills_hub) so every source adapter module
can import it at top level without cycles.
"""
"""Skills Hub data models, path validators, and SKILL.md helpers. Leaf module (no imports from
tools.skills_hub) so every source adapter module can import it at top level without cycles."""
import json
import logging
@@ -20,12 +17,8 @@ logger = logging.getLogger("tools.skills_hub")
def hub():
"""``tools.skills_hub`` resolved at call time.
Its cache / HTTP / index helpers are the test-patch targets
(``patch("tools.skills_hub._read_index_cache")`` ...), so adapters look
them up through the module on every call instead of binding at import.
"""
"""``tools.skills_hub`` resolved at call time: its cache / HTTP / index helpers are the test-patch targets
(``patch("tools.skills_hub._read_index_cache")`` ...), so adapters look them up on every call, not at import."""
import tools.skills_hub as mod
return mod
@@ -56,7 +49,6 @@ class SkillBundle:
def _skill_meta_to_dict(meta: SkillMeta) -> dict:
"""Convert a SkillMeta to a dict for caching."""
return dict(vars(meta))
@@ -71,8 +63,8 @@ def _cache_metas(key: str, metas: List[SkillMeta]) -> None:
def _memo_json(key: str, compute: Callable[[], Any], valid: Callable[[Any], bool] = lambda c: c is not None) -> Any:
"""Shared-index-cache memo: a cached value passing ``valid`` is returned as-is;
otherwise ``compute()`` runs and a non-None result is written back."""
"""Shared-index-cache memo: a cached value passing ``valid`` is returned as-is; otherwise ``compute()``
runs and a non-None result is written back."""
cached = hub()._read_index_cache(key)
if valid(cached):
return cached
@@ -86,9 +78,7 @@ def _get_json(url: str, *, timeout: int = 20, **kwargs) -> Optional[Any]:
"""Plain (unguarded) GET + JSON decode; None on non-200 or transport/decode error."""
try:
resp = httpx.get(url, timeout=timeout, **kwargs)
if resp.status_code != 200:
return None
return resp.json()
return resp.json() if resp.status_code == 200 else None
except (httpx.HTTPError, json.JSONDecodeError):
return None
@@ -103,8 +93,7 @@ def _get_text(url: str, *, timeout: int = 20, **kwargs) -> Optional[str]:
def _matches_query(query_lower: str, *fields: Any) -> bool:
"""Case-insensitive substring match of ``query_lower`` against joined fields
(lists are space-joined; an empty query matches everything)."""
"""Case-insensitive substring match against joined fields (lists space-joined; empty query matches all)."""
parts = [" ".join(str(t) for t in f) if isinstance(f, list) else str(f) for f in fields]
return query_lower in " ".join(parts).lower()
@@ -114,10 +103,8 @@ def _first_matching(query_lower: str, items: Iterable[Any], fields_of: Callable[
"""Substring-search ``items`` in order, converting hits with ``to_meta`` until ``limit``."""
results: List[SkillMeta] = []
for item in items:
if _matches_query(query_lower, *fields_of(item)):
meta = to_meta(item)
if meta:
results.append(meta)
if _matches_query(query_lower, *fields_of(item)) and (meta := to_meta(item)):
results.append(meta)
if len(results) >= limit:
break
return results
@@ -127,11 +114,8 @@ TRUST_RANK = {"builtin": 2, "trusted": 1, "community": 0}
def _dedupe_by_trust(results: Iterable[SkillMeta]) -> List[SkillMeta]:
"""Dedupe by identifier, keeping the higher-trust copy (first wins on ties).
identifier is unique per skill; name is not — two taps can publish
same-named skills, and browse-sh reuses task names across sites.
"""
"""Dedupe by identifier, keeping the higher-trust copy (first wins on ties). identifier is unique per
skill; name is not — two taps can publish same-named skills, and browse-sh reuses task names across sites."""
seen: Dict[str, SkillMeta] = {}
for r in results:
kept = seen.get(r.identifier)
@@ -141,11 +125,8 @@ def _dedupe_by_trust(results: Iterable[SkillMeta]) -> List[SkillMeta]:
class SkillSource(ABC):
"""Abstract base for all skill registry adapters.
``SOURCE_ID`` is the unique source id (e.g. 'github', 'clawhub'); ``TRUST_LEVEL``
the trust every identifier gets unless ``trust_level_for`` is overridden.
"""
"""Abstract base for all skill registry adapters. ``SOURCE_ID`` is the unique source id (e.g. 'github',
'clawhub'); ``TRUST_LEVEL`` the trust every identifier gets unless ``trust_level_for`` is overridden."""
SOURCE_ID: str = ""
TRUST_LEVEL: str = "community"
@@ -183,21 +164,15 @@ class GuardedFetchMixin:
return resp.content if resp is not None and resp.status_code == 200 else None
# ---------------------------------------------------------------------------
# SKILL.md frontmatter
# ---------------------------------------------------------------------------
# --- SKILL.md frontmatter ---------------------------------------------------
def _parse_frontmatter(content: str) -> dict:
"""Parse YAML frontmatter from SKILL.md content ({} when absent/invalid)."""
content = content.lstrip("\ufeff") # tolerate UTF-8 BOM (Windows editors)
if not content.startswith("---"):
return {}
match = re.search(r'\n---\s*\n', content[3:])
match = re.search(r'\n---\s*\n', content[3:]) if content.startswith("---") else None
if not match:
return {}
yaml_text = content[3:match.start() + 3]
try:
parsed = yaml.safe_load(yaml_text)
parsed = yaml.safe_load(content[3:match.start() + 3])
return parsed if isinstance(parsed, dict) else {}
except yaml.YAMLError:
return {}
@@ -206,11 +181,8 @@ def _parse_frontmatter(content: str) -> dict:
def _hermes_tags(fm: dict) -> Any:
"""``metadata.hermes.tags`` from parsed frontmatter, or ``[]`` (unvalidated type)."""
metadata = fm.get("metadata", {})
if isinstance(metadata, dict):
hermes_meta = metadata.get("hermes", {})
if isinstance(hermes_meta, dict):
return hermes_meta.get("tags", [])
return []
hermes_meta = metadata.get("hermes", {}) if isinstance(metadata, dict) else None
return hermes_meta.get("tags", []) if isinstance(hermes_meta, dict) else []
def source_url_for_bundle(bundle: SkillBundle) -> str:
@@ -226,34 +198,22 @@ def source_url_for_bundle(bundle: SkillBundle) -> str:
return bundle.identifier
# ---------------------------------------------------------------------------
# Bundle path validation
# ---------------------------------------------------------------------------
# --- Bundle path validation -------------------------------------------------
def _normalize_bundle_path(path_value: str, *, field_name: str, allow_nested: bool) -> str:
"""Normalize and validate bundle-controlled paths before touching disk."""
if not isinstance(path_value, str):
raise ValueError(f"Unsafe {field_name}: expected a string")
raw = path_value.strip()
if not raw:
raise ValueError(f"Unsafe {field_name}: empty path")
normalized = raw.replace("\\", "/")
path = PurePosixPath(normalized)
parts = [part for part in path.parts if part not in {"", "."}]
# A colon in any component is rejected: on Windows it marks a drive
# (``C:foo``) or an NTFS Alternate Data Stream (``file.py:payload`` writes
# scanner-invisible bytes); ``/`` is the only legal separator once normalized.
if (
normalized.startswith("/") or path.is_absolute()
or not parts or any(part == ".." for part in parts)
or any(":" in part for part in parts)
or (not allow_nested and len(parts) != 1)
):
# A colon in any component is rejected: on Windows it marks a drive (``C:foo``) or an NTFS Alternate Data
# Stream (``file.py:payload`` writes scanner-invisible bytes); ``/`` is the only legal separator once normalized.
if (normalized.startswith("/") or path.is_absolute() or not parts or any(part == ".." for part in parts)
or any(":" in part for part in parts) or (not allow_nested and len(parts) != 1)):
raise ValueError(f"Unsafe {field_name}: {path_value}")
return "/".join(parts)
@@ -272,10 +232,9 @@ def _validate_bundle_rel_path(rel_path: str) -> str:
def _normalize_lock_install_path(install_path: str, skill_name: str) -> str:
"""Validate a lock-file ``install_path`` (the ``uninstall_skill`` rmtree target).
Must be relative, traversal-free, and end with ``<skill_name>`` — nested
official skills legitimately live at ``mlops/training/<skill_name>``; an
empty/``"."``/absolute/mismatched entry could point rmtree at the whole
``skills/`` tree or outside it.
Must be relative, traversal-free, and end with ``<skill_name>`` — nested official skills legitimately
live at ``mlops/training/<skill_name>``; an empty/``"."``/absolute/mismatched entry could point rmtree
at the whole ``skills/`` tree or outside it.
"""
safe_skill_name = _validate_skill_name(skill_name)
normalized = _normalize_bundle_path(install_path, field_name="install path", allow_nested=True)
@@ -284,23 +243,16 @@ def _normalize_lock_install_path(install_path: str, skill_name: str) -> str:
return normalized
# ---------------------------------------------------------------------------
# Referenced support-file extraction from SKILL.md
# ---------------------------------------------------------------------------
# --- Referenced support-file extraction from SKILL.md -----------------------
_ALLOWED_SUPPORT_DIRS = frozenset({"references", "templates", "scripts", "assets", "examples"})
_LOCAL_LINK_RE = re.compile(
r"(?:\]\(|`|(?:^|[\s\"']))((?:references|templates|scripts|assets|examples)/[^\s)`\"'<>]+)",
re.MULTILINE,
)
r"(?:\]\(|`|(?:^|[\s\"']))((?:references|templates|scripts|assets|examples)/[^\s)`\"'<>]+)", re.MULTILINE)
_SUSPICIOUS_LOCAL_REF_RE = re.compile(
r"(?:references|templates|scripts|assets|examples)/(?:[^\s)`\"'<>]*/)?\.\.(?:/|$)"
)
r"(?:references|templates|scripts|assets|examples)/(?:[^\s)`\"'<>]*/)?\.\.(?:/|$)")
_VALUELESS_QUERY_FLAG_RE = re.compile(r"(?:[A-Za-z0-9_~-]|%[0-9A-Fa-f]{2})+\Z")
# Same-directory links (``](./FILE.ext)`` / ``](FILE.ext)``): siblings of SKILL.md
# the document links explicitly (e.g. ./CONTEXT-FORMAT.md). Dropping them made the
# install "succeed" with unresolved links. The extension requirement keeps prose
# words out; support-dir links stay on _LOCAL_LINK_RE.
# Same-directory links (``](./FILE.ext)`` / ``](FILE.ext)``): siblings of SKILL.md the document links
# explicitly (e.g. ./CONTEXT-FORMAT.md). Dropping them made the install "succeed" with unresolved links.
# The extension requirement keeps prose words out; support-dir links stay on _LOCAL_LINK_RE.
_SAMEDIR_LINK_RE = re.compile(r"\]\(([^)\s\"'<>]+)")
_SAMEDIR_NAME_RE = re.compile(r"^(?:\./)?[A-Za-z0-9][A-Za-z0-9._-]*$")
@@ -308,16 +260,12 @@ _SAMEDIR_NAME_RE = re.compile(r"^(?:\./)?[A-Za-z0-9][A-Za-z0-9._-]*$")
def _query_is_concrete(query: str) -> bool:
"""Whether a URL query is real URL syntax rather than glob prose.
A non-empty ``key=value`` part is always concrete. Valueless flags are
accepted only when RFC 3986 unreserved-token shaped (percent escapes ok);
``.``, brackets and extra ``?`` are excluded because ``?x.md`` / ``?.md``
is indistinguishable from a single-char glob finishing a filename in prose.
A non-empty ``key=value`` part is always concrete. Valueless flags are accepted only when RFC 3986
unreserved-token shaped (percent escapes ok); ``.``, brackets and extra ``?`` are excluded because
``?x.md`` / ``?.md`` is indistinguishable from a single-char glob finishing a filename in prose.
"""
return all(
("=" in part and bool(part.split("=", 1)[0]))
or bool(_VALUELESS_QUERY_FLAG_RE.fullmatch(part))
for part in query.split("&")
)
return all(("=" in part and bool(part.split("=", 1)[0])) or bool(_VALUELESS_QUERY_FLAG_RE.fullmatch(part))
for part in query.split("&"))
def _referenced_support_paths(skill_md: str) -> Optional[set[str]]:
@@ -328,45 +276,37 @@ def _referenced_support_paths(skill_md: str) -> Optional[set[str]]:
paths: set[str] = set()
for match in _LOCAL_LINK_RE.finditer(normalized):
candidate = match.group(1).rstrip(".,;:")
if candidate.endswith("?"):
continue
parsed = urlsplit(candidate)
raw = unquote(parsed.path)
if any(char in raw for char in "*?[]"):
continue
if parsed.query and not _query_is_concrete(parsed.query):
if (candidate.endswith("?") or any(char in raw for char in "*?[]")
or (parsed.query and not _query_is_concrete(parsed.query))):
continue
try:
safe = _validate_bundle_rel_path(raw)
except ValueError:
return None
if safe.split("/", 1)[0] in _ALLOWED_SUPPORT_DIRS:
# Prose placeholders (``references/type-<name>.md``, truncated at
# ``<`` to ``references/type-``) are instructions, not files: a
# basename ending in a separator is skipped. No extension
# requirement — ``references/LICENSE`` is legitimate.
# Prose placeholders (``references/type-<name>.md``, truncated at ``<`` to
# ``references/type-``) are instructions, not files: a basename ending in a
# separator is skipped. No extension requirement — ``references/LICENSE`` is legitimate.
base = safe.rsplit("/", 1)[-1]
if re.search(r"[*?<>]", safe) or not re.search(r"[A-Za-z0-9]$", base):
continue
paths.add(safe)
for match in _SAMEDIR_LINK_RE.finditer(normalized):
raw = match.group(1).rstrip(".,;:")
# Canonicalize like the support-dir branch (drop query/fragment,
# percent-decode), then strip a leading ``./``.
# Canonicalize like the support-dir branch (drop query/fragment, percent-decode), strip leading ``./``.
name = unquote(urlsplit(raw).path)
name = name[2:] if name.startswith("./") else name
# External URLs, anchors, mailto and site-absolute targets are not
# same-directory file links.
# External URLs, anchors, mailto and site-absolute targets are not same-directory file links.
if not name or "://" in raw or raw.startswith(("mailto:", "#", "/")):
continue
if name.startswith(".."):
return None
# Only unambiguous file links: an extension, no internal slash, never
# SKILL.md itself (casefolded — a ``skill.md`` entry would collide with
# the bundle root on macOS/Windows; skipped, not merged).
if "/" in name or name.casefold() == "skill.md" or "." not in name.lstrip("."):
continue
if not _SAMEDIR_NAME_RE.match(name):
# Only unambiguous file links: an extension, no internal slash, never SKILL.md itself (casefolded —
# a ``skill.md`` entry would collide with the bundle root on macOS/Windows; skipped, not merged).
if ("/" in name or name.casefold() == "skill.md" or "." not in name.lstrip(".")
or not _SAMEDIR_NAME_RE.match(name)):
continue
try:
safe = _validate_bundle_rel_path(name)
@@ -375,12 +315,10 @@ def _referenced_support_paths(skill_md: str) -> Optional[set[str]]:
paths.add(safe)
# Case-folded collisions among accepted same-dir names (``A.md`` + ``a.md``)
# would collide on install — drop the pair rather than guess.
folded: dict[str, str] = {}
folded: dict[str, list[str]] = {}
for p in sorted(paths):
key = p.casefold()
if key in folded:
paths.discard(folded[key])
paths.discard(p)
else:
folded[key] = p
folded.setdefault(p.casefold(), []).append(p)
for group in folded.values():
if len(group) > 1:
paths.difference_update(group)
return paths
+76 -204
View File
@@ -17,31 +17,19 @@ _INDEX_ID_PREFIXES = ("skills-sh/", "skills.sh/", "official/", "github/", "clawh
def _strip_prefix(value: str, prefixes) -> str:
for prefix in prefixes:
if value.startswith(prefix):
return value[len(prefix):]
return value
return next((value[len(p):] for p in prefixes if value.startswith(p)), value)
def _clean_rel_parts(path: str) -> Optional[List[str]]:
"""Split a relative path, dropping ``.``/empty parts; None on traversal or empty."""
parts = [p for p in path.split("/") if p not in ("", ".")]
if not parts or any(p == ".." for p in parts):
return None
return parts
return None if not parts or ".." in parts else parts
# ---------------------------------------------------------------------------
# Official optional skills source adapter
# ---------------------------------------------------------------------------
class OptionalSkillSource(SkillSource):
"""Skills from the repo's ``optional-skills/`` directory.
Official (Nous-maintained) but not activated by default — absent from the
system prompt and not copied to ~/.hermes/skills/ at setup. Discoverable
via the Skills Hub as source "official" with "builtin" trust.
"""
"""Skills from the repo's ``optional-skills/`` directory: official (Nous-maintained) but not
activated by default — absent from the system prompt and not copied to ~/.hermes/skills/ at
setup. Discoverable via the Skills Hub as source "official" with "builtin" trust."""
SOURCE_ID = "official"
TRUST_LEVEL = "builtin"
@@ -53,12 +41,9 @@ class OptionalSkillSource(SkillSource):
def __init__(self, auth: Optional[GitHubAuth] = None):
from hermes_constants import get_optional_skills_dir
self._optional_dir = get_optional_skills_dir(
Path(__file__).parent.parent / "optional-skills"
)
self._optional_dir = get_optional_skills_dir(Path(__file__).parent.parent / "optional-skills")
self._auth = auth
# GitHubSource for the live-repo fallback, created only when a skill is
# missing from the local checkout.
# GitHubSource for the live-repo fallback, created only when a skill is missing locally.
self._github: Optional[GitHubSource] = None
# "category/skill" -> True from the live repo tree; None = not fetched yet.
self._remote_dirs: Optional[Dict[str, bool]] = None
@@ -69,37 +54,25 @@ class OptionalSkillSource(SkillSource):
def _meta(self, rel_dir: str, name: str, description: str, tags: list) -> SkillMeta:
return SkillMeta(
name=name,
description=description,
source="official",
identifier=f"official/{rel_dir}",
trust_level="builtin",
repo=self.OFFICIAL_REPO,
name=name, description=description, source="official", identifier=f"official/{rel_dir}",
trust_level="builtin", repo=self.OFFICIAL_REPO,
# The centralized skills index consumes repo-root-relative paths.
path=f"{self.OPTIONAL_SKILLS_PREFIX}/{rel_dir}",
tags=tags,
path=f"{self.OPTIONAL_SKILLS_PREFIX}/{rel_dir}", tags=tags,
)
def _remote_meta(self, rel_dir: str) -> SkillMeta:
"""Placeholder meta for a skill that exists on live main but not locally."""
return self._meta(
rel_dir, rel_dir.rsplit("/", 1)[-1],
"Official optional skill (from live repo; run install to fetch)", [],
)
desc = "Official optional skill (from live repo; run install to fetch)"
return self._meta(rel_dir, rel_dir.rsplit("/", 1)[-1], desc, [])
@staticmethod
def _bundle(rel_id: str, files: Dict[str, Union[str, bytes]], **kwargs) -> SkillBundle:
return SkillBundle(
name=rel_id.rsplit("/", 1)[-1], files=files, source="official",
identifier=f"official/{rel_id}", trust_level="builtin", **kwargs,
)
# -- search -----------------------------------------------------------
return SkillBundle(name=rel_id.rsplit("/", 1)[-1], files=files, source="official",
identifier=f"official/{rel_id}", trust_level="builtin", **kwargs)
def search(self, query: str, limit: int = 10) -> List[SkillMeta]:
results: List[SkillMeta] = []
query_lower = query.lower()
local_rels: set = set()
for meta in self._scan_all():
local_rels.add(meta.identifier.split("/", 1)[-1] if meta.identifier else "")
@@ -107,7 +80,6 @@ class OptionalSkillSource(SkillSource):
results.append(meta)
if len(results) >= limit:
break
# Also surface skills that landed on live main after this install was cut.
if len(results) < limit:
for rel_dir in sorted(self._list_remote_skill_dirs()):
@@ -116,42 +88,36 @@ class OptionalSkillSource(SkillSource):
results.append(self._remote_meta(rel_dir))
if len(results) >= limit:
break
return results
# -- fetch ------------------------------------------------------------
def fetch(self, identifier: str) -> Optional[SkillBundle]:
# identifier format: "official/category/skill" or "official/skill"
rel = self._rel(identifier)
skill_dir = self._optional_dir / rel
# Guard against path traversal (e.g. "official/../../etc")
try:
resolved = skill_dir.resolve()
resolved = (self._optional_dir / rel).resolve()
optional_root = self._optional_dir.resolve()
if not resolved.is_relative_to(optional_root):
return None
except (OSError, ValueError):
return None
if resolved.is_dir():
skill_dir = resolved
else:
# Try by bare skill name; if still absent, the skill may have
# landed on main after this install was cut — use the live repo.
skill_dir = self._find_skill_dir(rel.rsplit("/", 1)[-1])
if not skill_dir:
return self._fetch_from_live_repo(rel)
# Else try by bare skill name; if still absent, the skill may have landed on main
# after this install was cut — use the live repo.
skill_dir = resolved if resolved.is_dir() else self._find_skill_dir(rel.rsplit("/", 1)[-1])
if not skill_dir:
return self._fetch_from_live_repo(rel)
rel_id = skill_dir.resolve().relative_to(optional_root).as_posix()
# Catalog stubs point at the real skill in an upstream-maintained repo
# (metadata.hermes.upstream); install pulls the live content from there.
upstream = self._upstream_pointer(skill_dir)
try:
skill_md = (skill_dir / "SKILL.md").read_text(encoding="utf-8")
except (OSError, UnicodeDecodeError):
skill_md = None
upstream = None if skill_md is None else self._upstream_pointer_from_content(skill_md)
if upstream is not None:
return self._fetch_from_upstream(upstream, rel_id)
files: Dict[str, Union[str, bytes]] = {}
for f in skill_dir.rglob("*"):
if f.is_file() and not _skip_bundle_file(f.relative_to(skill_dir).as_posix()):
@@ -159,33 +125,21 @@ class OptionalSkillSource(SkillSource):
files[str(f.relative_to(skill_dir))] = f.read_bytes()
except OSError:
continue
return self._bundle(rel_id, files) if files else None
# -- inspect ----------------------------------------------------------
def inspect(self, identifier: str) -> Optional[SkillMeta]:
skill_name = self._rel(identifier).rsplit("/", 1)[-1]
for meta in self._scan_all():
if meta.name == skill_name:
return meta
# Not in the local checkout — check live main.
matches = self._remote_matches(skill_name)
if len(matches) == 1:
return self._remote_meta(matches[0])
return None
# -- catalog ----------------------------------------------------------
matches = self._remote_matches(skill_name) # not in the local checkout — check live main
return self._remote_meta(matches[0]) if len(matches) == 1 else None
def list_local(self) -> List[SkillMeta]:
"""Every optional skill in the local checkout, with frontmatter metadata
(backs the dashboard/desktop "built-in optional skills" catalog)."""
return self._scan_all()
# -- internal helpers -------------------------------------------------
def _get_github(self) -> GitHubSource:
if self._github is None:
self._github = GitHubSource(auth=self._auth or GitHubAuth())
@@ -195,17 +149,13 @@ class OptionalSkillSource(SkillSource):
return [d for d in self._list_remote_skill_dirs() if d.rsplit("/", 1)[-1] == name]
def _fetch_from_live_repo(self, rel: str) -> Optional[SkillBundle]:
"""Fetch an optional skill straight from the live default branch.
Local installs lag ``main``; rather than demanding ``hermes update``
first, resolve against the live repo. ``rel`` is ``category/skill``
(used verbatim) or a bare skill name (located via the repo tree).
"""
"""Fetch an optional skill straight from the live default branch. Local installs lag
``main``; rather than demanding ``hermes update`` first, resolve against the live repo.
``rel`` is ``category/skill`` (used verbatim) or a bare skill name (located via the repo tree)."""
parts = _clean_rel_parts(rel.strip("/"))
if parts is None:
return None
rel = "/".join(parts)
github = self._get_github()
if rel not in self._list_remote_skill_dirs():
# Bare name (or stale category) — locate by final path segment.
@@ -213,16 +163,14 @@ class OptionalSkillSource(SkillSource):
if len(matches) != 1:
return None
rel = matches[0]
repo_path = f"{self.OPTIONAL_SKILLS_PREFIX}/{rel}"
# Download the FULL directory byte-exact (root-level install scripts,
# LICENSE, tests/). GitHubSource.fetch() would only pull SKILL.md +
# referenced support dirs.
# Download the FULL directory byte-exact (root-level install scripts, LICENSE, tests/).
# GitHubSource.fetch() would only pull SKILL.md + referenced support dirs.
tree = github._get_repo_tree(self.OFFICIAL_REPO)
if tree is None:
return None
files: Dict[str, Union[str, bytes]] = {}
for rel_file, item_path, regular in _tree_members(tree[1], f"{repo_path}/"):
for rel_file, item_path, regular in _tree_members(tree[1], f"{self.OPTIONAL_SKILLS_PREFIX}/{rel}/"):
if not regular or _skip_bundle_file(rel_file):
continue
content = github._fetch_file_bytes(self.OFFICIAL_REPO, item_path)
@@ -230,35 +178,27 @@ class OptionalSkillSource(SkillSource):
logger.warning("Live-repo optional skill fetch failed for %s", item_path)
return None
files[rel_file] = content
if "SKILL.md" not in files:
return None
# Live-fetched catalog stubs redirect the same way local ones do.
upstream = self._upstream_pointer_from_content(files["SKILL.md"])
if upstream is not None:
return self._fetch_from_upstream(upstream, rel)
logger.info("Optional skill '%s' fetched from live repo (not in local checkout)", rel)
return self._bundle(rel, files)
def _list_remote_skill_dirs(self) -> Dict[str, bool]:
"""``category/skill`` dirs under optional-skills/ on live main.
One repo-tree call (cached per-process by GitHubSource + the on-disk
index cache). {} when the network/API is unavailable — callers degrade
to local-only.
"""
"""``category/skill`` dirs under optional-skills/ on live main. One repo-tree call (cached
per-process by GitHubSource + the on-disk index cache). {} when the network/API is
unavailable — callers degrade to local-only."""
if self._remote_dirs is not None:
return self._remote_dirs
def compute():
dirs: Dict[str, bool] = {}
tree = self._get_github()._get_repo_tree(self.OFFICIAL_REPO)
if tree is None:
if (tree := self._get_github()._get_repo_tree(self.OFFICIAL_REPO)) is None:
return None
prefix = f"{self.OPTIONAL_SKILLS_PREFIX}/"
suffix = "/SKILL.md"
prefix, suffix = f"{self.OPTIONAL_SKILLS_PREFIX}/", "/SKILL.md"
for item in tree[1]:
path = item.get("path", "")
if item.get("type") == "blob" and path.startswith(prefix) and path.endswith(suffix):
@@ -267,25 +207,13 @@ class OptionalSkillSource(SkillSource):
dirs[rel_dir] = True
return dirs or None
self._remote_dirs = _memo_json(
"official_optional_dirs", compute, valid=lambda c: isinstance(c, dict) and bool(c),
) or {}
self._remote_dirs = _memo_json("official_optional_dirs", compute,
valid=lambda c: isinstance(c, dict) and bool(c)) or {}
return self._remote_dirs
def _upstream_pointer(self, skill_dir: Path) -> Optional[Dict[str, str]]:
"""Upstream pointer for a catalog-stub skill dir, or None for vendored skills.
A stub declares ``metadata.hermes.upstream: {repo: owner/name, path: ...}``
in its SKILL.md frontmatter.
"""
try:
content = (skill_dir / "SKILL.md").read_text(encoding="utf-8")
except (OSError, UnicodeDecodeError):
return None
return self._upstream_pointer_from_content(content)
def _upstream_pointer_from_content(self, content: Union[str, bytes]) -> Optional[Dict[str, str]]:
"""Parse ``metadata.hermes.upstream`` out of SKILL.md content."""
"""Parse ``metadata.hermes.upstream: {repo: owner/name, path: ...}`` out of SKILL.md content
(a catalog stub); None for vendored skills."""
if isinstance(content, bytes):
try:
content = content.decode("utf-8")
@@ -302,51 +230,33 @@ class OptionalSkillSource(SkillSource):
if not repo or repo.count("/") != 1 or not path:
return None
parts = _clean_rel_parts(path)
if parts is None:
return None
return {"repo": repo, "path": "/".join(parts)}
return None if parts is None else {"repo": repo, "path": "/".join(parts)}
def _fetch_from_upstream(self, upstream: Dict[str, str], rel_id: str) -> Optional[SkillBundle]:
"""Fetch an upstream-maintained optional skill via GitHubSource.fetch()
(full-tree download, symlink/unsafe-path rejection, quarantine + scan
downstream) and re-label it as an official catalog entry."""
"""Fetch an upstream-maintained optional skill via GitHubSource.fetch() (full-tree download,
symlink/unsafe-path rejection, quarantine + scan downstream) and re-label it as an official
catalog entry."""
bundle = self._get_github().fetch(f"{upstream['repo']}/{upstream['path']}")
if bundle is None:
logger.warning(
"Upstream fetch failed for optional skill %s (%s:%s)",
rel_id, upstream["repo"], upstream["path"],
)
logger.warning("Upstream fetch failed for optional skill %s (%s:%s)",
rel_id, upstream["repo"], upstream["path"])
return None
return SkillBundle(
name=bundle.name,
files=bundle.files,
source="official",
identifier=f"official/{rel_id}",
name=bundle.name, files=bundle.files, source="official", identifier=f"official/{rel_id}",
# Curated endorsement, but the content is live third-party:
# "trusted", not "builtin", so a dangerous scan verdict still blocks.
trust_level="trusted",
metadata={
**bundle.metadata,
"upstream_repo": upstream["repo"],
"upstream_path": upstream["path"],
},
metadata={**bundle.metadata, "upstream_repo": upstream["repo"], "upstream_path": upstream["path"]},
)
def _local_skill_mds(self):
if not self._optional_dir.is_dir():
return
for skill_md in sorted(self._optional_dir.rglob("SKILL.md")):
if not is_excluded_skill_path(
skill_md.relative_to(self._optional_dir), root=self._optional_dir
):
yield skill_md
root = self._optional_dir
return (md for md in (sorted(root.rglob("SKILL.md")) if root.is_dir() else [])
if not is_excluded_skill_path(md.relative_to(root), root=root))
def _find_skill_dir(self, name: str) -> Optional[Path]:
"""Find a skill directory by name anywhere in optional-skills/."""
for skill_md in self._local_skill_mds():
if skill_md.parent.name == name:
return skill_md.parent
return None
return next((md.parent for md in self._local_skill_mds() if md.parent.name == name), None)
def _scan_all(self) -> List[SkillMeta]:
"""Enumerate all optional skills with metadata."""
@@ -357,30 +267,18 @@ class OptionalSkillSource(SkillSource):
content = skill_md.read_text(encoding="utf-8")
except (OSError, UnicodeDecodeError):
continue
fm = _parse_frontmatter(content)
tags = _hermes_tags(fm)
results.append(self._meta(
parent.relative_to(self._optional_dir).as_posix(),
fm.get("name", parent.name), fm.get("description", "")[:200],
tags if isinstance(tags, list) else [],
))
results.append(self._meta(parent.relative_to(self._optional_dir).as_posix(), fm.get("name", parent.name),
fm.get("description", "")[:200], tags if isinstance(tags, list) else []))
return results
# ---------------------------------------------------------------------------
# Hermes centralized index source
# ---------------------------------------------------------------------------
class HermesIndexSource(SkillSource):
"""Skill source backed by the centralized Hermes Skills Index.
A JSON catalog on the docs site, rebuilt daily by CI, with metadata +
resolved GitHub paths for every skill — search and path discovery cost
zero GitHub API calls. When unavailable every method returns empty/None
so downstream sources take over transparently.
"""
"""Skill source backed by the centralized Hermes Skills Index: a JSON catalog on the docs site,
rebuilt daily by CI, with metadata + resolved GitHub paths for every skill — search and path
discovery cost zero GitHub API calls. When unavailable every method returns empty/None so
downstream sources take over transparently."""
SOURCE_ID = "hermes-index"
@@ -392,8 +290,7 @@ class HermesIndexSource(SkillSource):
def _ensure_loaded(self) -> dict:
if not self._loaded:
self._index = hub()._load_hermes_index()
self._loaded = True
self._index, self._loaded = hub()._load_hermes_index(), True
return self._index or {}
def _skills(self) -> list:
@@ -414,45 +311,31 @@ class HermesIndexSource(SkillSource):
return entry.get("trust_level", "community") if entry else "community"
def search(self, query: str, limit: int = 10) -> List[SkillMeta]:
"""Search the cached index (zero API calls).
Matches name, description, tags, identifier and ``extra.provider`` (so
``nvidia`` finds ``NVIDIA/skills/...`` entries stored as source
"github"). Ranked exact name > name prefix > provider > whole-word >
name substring > other, index order as tiebreaker — a raw
break-at-limit slice buried the most relevant skills.
"""
"""Search the cached index (zero API calls). Matches name, description, tags, identifier and
``extra.provider`` (so ``nvidia`` finds ``NVIDIA/skills/...`` entries stored as source
"github"). Ranked exact name > name prefix > provider > whole-word > name substring > other,
index order as tiebreaker — a raw break-at-limit slice buried the most relevant skills."""
skills = self._skills()
if not skills:
return []
if not query.strip():
return [self._to_meta(s) for s in skills[:limit]] # featured / index order
query_lower = query.lower()
scored: List[Tuple[int, int, dict]] = []
for i, s in enumerate(skills):
name = str(s.get("name", "")).lower()
provider = str((s.get("extra") or {}).get("provider", "")).lower()
haystack = " ".join([
name,
str(s.get("description", "")).lower(),
" ".join(str(t).lower() for t in s.get("tags", [])),
str(s.get("identifier", "")).lower(),
provider,
name, str(s.get("description", "")).lower(), " ".join(str(t).lower() for t in s.get("tags", [])),
str(s.get("identifier", "")).lower(), provider,
])
if query_lower not in haystack:
continue
ranks = (
name == query_lower,
name.startswith(query_lower),
provider == query_lower,
query_lower in name.split() or query_lower in provider.split(),
query_lower in name,
True,
name == query_lower, name.startswith(query_lower), provider == query_lower,
query_lower in name.split() or query_lower in provider.split(), query_lower in name, True,
)
scored.append((ranks.index(True), i, s))
scored.sort(key=lambda x: (x[0], x[1]))
return [self._to_meta(s) for _, _, s in scored[:limit]]
@@ -462,11 +345,8 @@ class HermesIndexSource(SkillSource):
entry = self._find_entry(identifier)
if not entry:
return None
candidates = [entry.get("resolved_github_id")]
repo, path = entry.get("repo", ""), entry.get("path", "")
if repo and path:
candidates.append(f"{repo}/{path}")
candidates = [entry.get("resolved_github_id")] + ([f"{repo}/{path}"] if repo and path else [])
for github_id in filter(None, candidates):
bundle = self._get_github().fetch(github_id)
if bundle:
@@ -484,23 +364,15 @@ class HermesIndexSource(SkillSource):
"""Exact identifier match first, then match with source prefixes stripped."""
skills = self._skills()
normalized = _strip_prefix(identifier, _INDEX_ID_PREFIXES)
return next(
(s for s in skills if s.get("identifier") == identifier), None,
) or next(
(s for s in skills if _strip_prefix(s.get("identifier", ""), _INDEX_ID_PREFIXES) == normalized),
None,
return next((s for s in skills if s.get("identifier") == identifier), None) or next(
(s for s in skills if _strip_prefix(s.get("identifier", ""), _INDEX_ID_PREFIXES) == normalized), None,
)
@staticmethod
def _to_meta(entry: dict) -> SkillMeta:
return SkillMeta(
name=entry.get("name", ""),
description=entry.get("description", ""),
source=entry.get("source", "hermes-index"),
identifier=entry.get("identifier", ""),
trust_level=entry.get("trust_level", "community"),
repo=entry.get("repo"),
path=entry.get("path"),
tags=entry.get("tags", []),
extra=entry.get("extra", {}),
name=entry.get("name", ""), description=entry.get("description", ""),
source=entry.get("source", "hermes-index"), identifier=entry.get("identifier", ""),
trust_level=entry.get("trust_level", "community"), repo=entry.get("repo"), path=entry.get("path"),
tags=entry.get("tags", []), extra=entry.get("extra", {}),
)
+7 -35
View File
@@ -25,7 +25,6 @@ if TYPE_CHECKING: # runtime use resolves through the origin (test patch target)
logger = logging.getLogger("tools.skills_hub")
HERMES_INDEX_URL = "https://hermes-agent.nousresearch.com/docs/api/skills-index.json"
HERMES_INDEX_TTL = 6 * 3600 # 6 hours
@@ -49,16 +48,11 @@ def _load_hermes_index() -> Optional[dict]:
cached = _read_json_if_fresh(cache_file, HERMES_INDEX_TTL)
if cached is not None:
return cached
data = None
for accept_encoding in ("gzip, deflate", "identity"):
try:
resp = httpx.get(
HERMES_INDEX_URL,
timeout=15,
follow_redirects=True,
headers={"Accept-Encoding": accept_encoding},
)
resp = httpx.get(HERMES_INDEX_URL, timeout=15, follow_redirects=True,
headers={"Accept-Encoding": accept_encoding})
if resp.status_code != 200:
logger.debug("Hermes index fetch returned %d", resp.status_code)
return _load_stale_index_cache()
@@ -69,16 +63,13 @@ def _load_hermes_index() -> Optional[dict]:
except (httpx.HTTPError, json.JSONDecodeError) as e:
logger.debug("Hermes index fetch failed: %s", e)
return _load_stale_index_cache()
if not isinstance(data, dict) or "skills" not in data:
return _load_stale_index_cache()
try:
cache_file.parent.mkdir(parents=True, exist_ok=True)
cache_file.write_text(json.dumps(data), encoding="utf-8")
except OSError:
pass
return data
@@ -102,7 +93,6 @@ def create_source_router(auth: Optional[GitHubAuth] = None) -> List[SkillSource]
)
if auth is None:
auth = GitHubAuth()
return [
OptionalSkillSource(auth=auth), # official optional skills (highest priority)
HermesIndexSource(auth=auth), # centralized index (search + resolved install paths)
@@ -116,9 +106,7 @@ def create_source_router(auth: Optional[GitHubAuth] = None) -> List[SkillSource]
]
def _search_one_source(
src: SkillSource, query: str, limit: int
) -> Tuple[str, List[SkillMeta]]:
def _search_one_source(src: SkillSource, query: str, limit: int) -> Tuple[str, List[SkillMeta]]:
"""Search a single source. Runs in a thread for parallelism."""
try:
return src.source_id(), src.search(query, limit=limit)
@@ -137,8 +125,7 @@ def _select_active_sources(sources: List[SkillSource], source_filter: str) -> Li
"""
effective = "all" if source_filter.strip().lower() in _PROVIDER_FILTER_VALUES else source_filter
index_available = effective == "all" and any(
src.source_id() == "hermes-index" and getattr(src, "is_available", False)
for src in sources
src.source_id() == "hermes-index" and getattr(src, "is_available", False) for src in sources
)
active: List[SkillSource] = []
for src in sources:
@@ -152,12 +139,8 @@ def _select_active_sources(sources: List[SkillSource], source_filter: str) -> Li
def parallel_search_sources(
sources: List[SkillSource],
query: str = "",
per_source_limits: Optional[Dict[str, int]] = None,
source_filter: str = "all",
overall_timeout: float = 30,
on_source_done: Optional[Any] = None,
sources: List[SkillSource], query: str = "", per_source_limits: Optional[Dict[str, int]] = None,
source_filter: str = "all", overall_timeout: float = 30, on_source_done: Optional[Any] = None,
) -> Tuple[List[SkillMeta], Dict[str, int], List[str]]:
"""Search all sources in parallel with an overall timeout.
@@ -168,11 +151,9 @@ def parallel_search_sources(
per_source_limits = per_source_limits or {}
active = _select_active_sources(sources, source_filter)
all_results: List[SkillMeta] = []
source_counts: Dict[str, int] = {}
timed_out_ids: List[str] = []
if not active:
return all_results, source_counts, timed_out_ids
@@ -185,7 +166,6 @@ def parallel_search_sources(
pool.submit(_search_one_source, src, query, per_source_limits.get(src.source_id(), 50)): src.source_id()
for src in active
}
try:
for fut in as_completed(futures, timeout=overall_timeout):
try:
@@ -202,7 +182,6 @@ def parallel_search_sources(
logger.debug("Skills browse timed out waiting for: %s", ", ".join(timed_out_ids))
finally:
pool.shutdown(wait=False, cancel_futures=True)
return all_results, source_counts, timed_out_ids
@@ -210,17 +189,10 @@ def unified_search(query: str, sources: List[SkillSource],
source_filter: str = "all", limit: int = 10) -> List[SkillMeta]:
"""Search all sources (in parallel) and merge results."""
from tools.skills_hub import _filter_results_by_provider, parallel_search_sources
all_results, _, _ = parallel_search_sources(
sources,
query=query,
source_filter=source_filter,
overall_timeout=30,
)
all_results, _, _ = parallel_search_sources(sources, query=query, source_filter=source_filter, overall_timeout=30)
# Provider filters target ``extra.provider`` on the merged set, not a source id.
if source_filter.strip().lower() in _PROVIDER_FILTER_VALUES:
all_results = _filter_results_by_provider(all_results, source_filter)
deduped = _dedupe_by_trust(all_results)
# Stable-sort by trust before truncating so the limit cut never drops a
# builtin/official entry because a high-volume community source finished
+48 -128
View File
@@ -32,8 +32,7 @@ class SkillsShSource(SkillSource):
_SITEMAP_HEADERS = {"Accept-Encoding": "gzip"}
_SITEMAP_LOC_RE = re.compile(r"<loc>([^<]+)</loc>", re.IGNORECASE)
_SITEMAP_SKILL_RE = re.compile(
r"^https?://(?:www\.)?skills\.sh/(?P<owner>[^/]+)/(?P<repo>[^/]+)/(?P<skill>[^/]+)/?$",
re.IGNORECASE,
r"^https?://(?:www\.)?skills\.sh/(?P<owner>[^/]+)/(?P<repo>[^/]+)/(?P<skill>[^/]+)/?$", re.IGNORECASE,
)
_SKILL_LINK_RE = re.compile(r'href=["\']/(?P<id>(?!agents/|_next/|api/)[^"\'/]+/[^"\'/]+/[^"\'/]+)["\']')
_INSTALL_CMD_RE = re.compile(
@@ -56,28 +55,20 @@ class SkillsShSource(SkillSource):
_STANDARD_BASE_PATHS = ("skills/", ".agents/skills/", ".claude/skills/")
_strip_html = staticmethod(_strip_html)
SOURCE_ID = "skills-sh"
def __init__(self, auth: GitHubAuth):
self.auth = auth
self.github = GitHubSource(auth=auth)
self.auth, self.github = auth, GitHubSource(auth=auth)
def trust_level_for(self, identifier: str) -> str:
return self.github.trust_level_for(self._normalize_identifier(identifier))
def _meta(self, canonical: str, *, name: str, description: str, path: str,
extra: Optional[Dict[str, Any]] = None) -> SkillMeta:
repo = "/".join(canonical.split("/", 2)[:2])
return SkillMeta(
name=name,
description=description,
source="skills.sh",
identifier=self._wrap_identifier(canonical),
trust_level=self.github.trust_level_for(canonical),
repo=repo,
path=path,
extra=extra if extra is not None else {},
name=name, description=description, source="skills.sh", identifier=self._wrap_identifier(canonical),
trust_level=self.github.trust_level_for(canonical), repo="/".join(canonical.split("/", 2)[:2]),
path=path, extra=extra if extra is not None else {},
)
def _urls_for(self, canonical: str, repo: str) -> Dict[str, str]:
@@ -87,20 +78,17 @@ class SkillsShSource(SkillSource):
if not query.strip():
# Empty query = bulk catalog dump (build_skills_index.py) — walk the sitemap.
return self._sitemap_catalog(limit)
cache_key = f"skills_sh_search_{hashlib.md5(f'{query}|{limit}'.encode()).hexdigest()}"
cached = _cached_metas(cache_key)
if cached is not None:
return cached[:limit]
data = _get_json(self.SEARCH_URL, params={"q": query, "limit": limit})
if data is None:
return []
items = data.get("skills", []) if isinstance(data, dict) else []
if not isinstance(items, list):
return []
results = [m for m in (self._meta_from_search_item(i) for i in items[:limit]) if m]
results = [m for m in map(self._meta_from_search_item, items[:limit]) if m]
_cache_metas(cache_key, results)
return results
@@ -111,8 +99,7 @@ class SkillsShSource(SkillSource):
def _relabel(github_id: Optional[str]) -> Optional[SkillBundle]:
bundle = self.github.fetch(github_id) if github_id else None
if bundle:
bundle.source = "skills.sh"
bundle.identifier = self._wrap_identifier(canonical)
bundle.source, bundle.identifier = "skills.sh", self._wrap_identifier(canonical)
bundle.metadata.update(self._detail_to_metadata(canonical, detail))
return bundle or None
@@ -126,9 +113,7 @@ class SkillsShSource(SkillSource):
canonical = self._normalize_identifier(identifier)
detail = self._fetch_detail_page(canonical)
meta = self._resolve_github_meta(canonical, detail=detail)
if meta:
return self._finalize_inspect_meta(meta, canonical, detail)
return None
return self._finalize_inspect_meta(meta, canonical, detail) if meta else None
def _sitemap_catalog(self, limit: int) -> List[SkillMeta]:
"""Enumerate the full catalog via the sitemap (cached for the index TTL —
@@ -140,37 +125,27 @@ class SkillsShSource(SkillSource):
# Step 1: sitemap index -> per-skill sitemap URLs.
index_xml = _get_text(self.SITEMAP_INDEX_URL, follow_redirects=True, headers=self._SITEMAP_HEADERS)
skill_sitemap_urls = [
m.group(1).strip() for m in self._SITEMAP_LOC_RE.finditer(index_xml or "")
if "sitemap-skills" in m.group(1)
]
skill_sitemap_urls = [m.group(1).strip() for m in self._SITEMAP_LOC_RE.finditer(index_xml or "")
if "sitemap-skills" in m.group(1)]
if not skill_sitemap_urls:
return self._featured_skills(limit)
# Step 2: collect canonical "owner/repo/skill" IDs from each sitemap.
seen: set[str] = set()
results: List[SkillMeta] = []
seen, results = set(), []
for sitemap_url in skill_sitemap_urls:
xml = _get_text(sitemap_url, timeout=30, follow_redirects=True, headers=self._SITEMAP_HEADERS)
for loc_match in self._SITEMAP_LOC_RE.finditer(xml or ""):
m = self._SITEMAP_SKILL_RE.match(loc_match.group(1).strip())
if not m:
continue
repo = f"{m.group('owner')}/{m.group('repo')}"
skill_name = m.group("skill")
canonical = f"{repo}/{skill_name}"
if canonical in seen:
continue
seen.add(canonical)
results.append(self._meta(
canonical, name=skill_name,
description=f"Indexed by skills.sh from {repo}",
path=skill_name, extra=self._urls_for(canonical, repo),
))
repo, skill = f"{m.group('owner')}/{m.group('repo')}", m.group("skill")
canonical = f"{repo}/{skill}"
if canonical not in seen:
seen.add(canonical)
results.append(self._meta(canonical, name=skill, description=f"Indexed by skills.sh from {repo}",
path=skill, extra=self._urls_for(canonical, repo)))
if not results:
return self._featured_skills(limit)
_cache_metas(cache_key, results)
return results[:limit] if limit > 0 else results
@@ -179,57 +154,41 @@ class SkillsShSource(SkillSource):
cached = _cached_metas(cache_key)
if cached is not None:
return cached[:limit]
html = _get_text(self.BASE_URL)
if html is None:
return []
seen: set[str] = set()
results: List[SkillMeta] = []
seen, results = set(), []
for match in self._SKILL_LINK_RE.finditer(html):
canonical = match.group("id")
if canonical in seen:
continue
split = None if canonical in seen else _split_repo_id(canonical)
seen.add(canonical)
split = _split_repo_id(canonical)
if split is None:
continue
repo, skill_path = split
results.append(self._meta(
canonical, name=skill_path.split("/")[-1],
description=f"Featured on skills.sh from {repo}",
path=skill_path,
))
results.append(self._meta(canonical, name=skill_path.split("/")[-1],
description=f"Featured on skills.sh from {repo}", path=skill_path))
if len(results) >= limit:
break
_cache_metas(cache_key, results)
return results
def _meta_from_search_item(self, item: dict) -> Optional[SkillMeta]:
if not isinstance(item, dict):
return None
canonical = item.get("id")
repo = item.get("source")
skill_path = item.get("skillId")
canonical, repo, skill_path = item.get("id"), item.get("source"), item.get("skillId")
if not isinstance(canonical, str) or canonical.count("/") < 2:
if not (isinstance(repo, str) and isinstance(skill_path, str)):
return None
canonical = f"{repo}/{skill_path}"
split = _split_repo_id(canonical)
if split is None:
return None
repo, skill_path = split
installs = item.get("installs")
installs_label = f" · {int(installs):,} installs" if isinstance(installs, int) else ""
return self._meta(
canonical,
name=str(item.get("name") or skill_path.split("/")[-1]),
description=f"Indexed by skills.sh from {repo}{installs_label}",
path=skill_path,
canonical, name=str(item.get("name") or skill_path.split("/")[-1]),
description=f"Indexed by skills.sh from {repo}{installs_label}", path=skill_path,
extra={"installs": installs, **self._urls_for(canonical, repo)},
)
@@ -237,28 +196,21 @@ class SkillsShSource(SkillSource):
def compute():
html = _get_text(f"{self.BASE_URL}/{identifier}")
return None if html is None else self._parse_detail_page(identifier, html) or None
return _memo_json(
f"skills_sh_detail_{hashlib.md5(identifier.encode()).hexdigest()}", compute,
valid=lambda c: isinstance(c, dict),
)
key = f"skills_sh_detail_{hashlib.md5(identifier.encode()).hexdigest()}"
return _memo_json(key, compute, valid=lambda c: isinstance(c, dict))
def _parse_detail_page(self, identifier: str, html: str) -> Optional[dict]:
split = _split_repo_id(identifier)
if split is None:
return None
repo, install_skill = split
install_command = None
install_match = self._INSTALL_CMD_RE.search(html)
install_command, install_match = None, self._INSTALL_CMD_RE.search(html)
if install_match:
install_command = install_match.group(0).strip()
install_skill = (install_match.group("skill") or install_skill).strip()
repo = self._extract_repo_slug((install_match.group("repo") or "").strip()) or repo
return {
"repo": repo,
"install_skill": install_skill,
"repo": repo, "install_skill": install_skill,
"page_title": self._extract_first_match(self._PAGE_H1_RE, html),
"body_title": self._extract_first_match(self._PROSE_H1_RE, html),
"body_summary": self._extract_first_match(self._PROSE_P_RE, html),
@@ -277,32 +229,21 @@ class SkillsShSource(SkillSource):
skill_token = skill_path.split("/")[-1]
tokens = [skill_token]
if isinstance(detail, dict):
tokens.extend([
detail.get("install_skill", ""),
detail.get("page_title", ""),
detail.get("body_title", ""),
])
tokens.extend(detail.get(k, "") for k in ("install_skill", "page_title", "body_title"))
def _match_in(base_path: str) -> Optional[str]:
try:
skills = self.github._list_skills_in_repo(repo, base_path)
except Exception:
return None
for meta in skills:
if self._matches_skill_tokens(meta, tokens):
return meta.identifier
return None
for base_path in self._STANDARD_BASE_PATHS:
found = _match_in(base_path)
if found:
return found
return next((m.identifier for m in skills if self._matches_skill_tokens(m, tokens)), None)
# One recursive tree lookup before brute-forcing every top-level dir
# (avoids request bursts on categorized repos like borghei/claude-skills).
tree_result = self.github._find_skill_in_repo_tree(repo, skill_token)
if tree_result:
return tree_result
found = (next((f for f in map(_match_in, self._STANDARD_BASE_PATHS) if f), None)
or self.github._find_skill_in_repo_tree(repo, skill_token))
if found:
return found
# Fallback: scan repo root for directories that might contain skills.
try:
@@ -322,7 +263,6 @@ class SkillsShSource(SkillSource):
return found
except Exception:
pass
return None
def _resolve_github_meta(self, identifier: str, detail: Optional[dict] = None) -> Optional[SkillMeta]:
@@ -330,21 +270,15 @@ class SkillsShSource(SkillSource):
meta = self.github.inspect(candidate)
if meta:
return meta
resolved = self._discover_identifier(identifier, detail=detail)
if resolved:
return self.github.inspect(resolved)
return None
return self.github.inspect(resolved) if resolved else None
def _finalize_inspect_meta(self, meta: SkillMeta, canonical: str, detail: Optional[dict]) -> SkillMeta:
meta.source = "skills.sh"
meta.identifier = self._wrap_identifier(canonical)
meta.source, meta.identifier = "skills.sh", self._wrap_identifier(canonical)
meta.trust_level = self.trust_level_for(canonical)
meta.extra = {**meta.extra, **self._detail_to_metadata(canonical, detail)}
if isinstance(detail, dict):
body_summary = detail.get("body_summary")
weekly_installs = detail.get("weekly_installs")
body_summary, weekly_installs = detail.get("body_summary"), detail.get("weekly_installs")
if body_summary:
meta.description = body_summary
elif meta.description and weekly_installs:
@@ -353,36 +287,28 @@ class SkillsShSource(SkillSource):
@classmethod
def _matches_skill_tokens(cls, meta: SkillMeta, skill_tokens: List[str]) -> bool:
candidates = set()
candidates.update(cls._token_variants(meta.name))
candidates.update(cls._token_variants(meta.path))
candidates.update(cls._token_variants(meta.identifier.split("/", 2)[-1] if meta.identifier else None))
candidates = (cls._token_variants(meta.name) | cls._token_variants(meta.path)
| cls._token_variants(meta.identifier.split("/", 2)[-1] if meta.identifier else None))
return any(cls._token_variants(token) & candidates for token in skill_tokens)
@staticmethod
def _token_variants(value: Optional[str]) -> set[str]:
if not value:
return set()
plain = _strip_html(str(value)).strip().strip("/").lower()
if not plain:
return set()
base = plain.split("/")[-1]
sanitized = re.sub(r'[^a-z0-9/_-]+', '-', plain).strip('-')
base, sanitized = plain.split("/")[-1], re.sub(r'[^a-z0-9/_-]+', '-', plain).strip('-')
tail = base.lstrip('@')
variants = {
plain, plain.replace("_", "-"), plain.replace("/", "-"),
base, base.replace("_", "-"),
sanitized, sanitized.replace("/", "-"), sanitized.split("/")[-1],
tail, tail.replace("_", "-"),
plain, plain.replace("_", "-"), plain.replace("/", "-"), base, base.replace("_", "-"),
sanitized, sanitized.replace("/", "-"), sanitized.split("/")[-1], tail, tail.replace("_", "-"),
}
return {v for v in variants if v}
@staticmethod
def _extract_repo_slug(repo_value: str) -> Optional[str]:
repo_value = repo_value.strip().removeprefix("https://github.com/")
parts = repo_value.strip("/").split("/")
parts = repo_value.strip().removeprefix("https://github.com/").strip("/").split("/")
return f"{parts[0]}/{parts[1]}" if len(parts) >= 2 else None
@staticmethod
@@ -398,9 +324,8 @@ class SkillsShSource(SkillSource):
metadata["repo_url"] = f"https://github.com/{parts[0]}/{parts[1]}"
if isinstance(detail, dict):
for key in ("weekly_installs", "install_command", "repo_url", "detail_url", "security_audits"):
value = detail.get(key)
if value:
metadata[key] = value
if detail.get(key):
metadata[key] = detail[key]
return metadata
@classmethod
@@ -413,9 +338,7 @@ class SkillsShSource(SkillSource):
audits: Dict[str, str] = {}
for audit in ("agent-trust-hub", "socket", "snyk"):
idx = html.find(f"/security/{audit}")
if idx == -1:
continue
match = re.search(r'(Pass|Warn|Fail)', html[idx:idx + 500], re.IGNORECASE)
match = re.search(r'(Pass|Warn|Fail)', html[idx:idx + 500], re.IGNORECASE) if idx != -1 else None
if match:
audits[audit] = match.group(1).title()
return audits
@@ -430,11 +353,8 @@ class SkillsShSource(SkillSource):
split = _split_repo_id(identifier)
if split is None:
return [identifier]
repo, skill_path = split
skill_path = skill_path.lstrip("/")
return list(dict.fromkeys(
[f"{repo}/{skill_path}"] + [f"{repo}/{base}{skill_path}" for base in cls._STANDARD_BASE_PATHS]
))
repo, path = split[0], split[1].lstrip("/")
return list(dict.fromkeys([f"{repo}/{path}"] + [f"{repo}/{b}{path}" for b in cls._STANDARD_BASE_PATHS]))
@staticmethod
def _wrap_identifier(identifier: str) -> str:
+83 -233
View File
@@ -9,16 +9,14 @@ from urllib.parse import quote, urljoin, urlparse, urlunparse
from tools.skills_hub_models import (
GuardedFetchMixin, SkillBundle, SkillMeta, SkillSource, _first_matching, _get_json, _get_text,
_hermes_tags, _matches_query, _memo_json, _parse_frontmatter, _referenced_support_paths,
_hermes_tags, _memo_json, _parse_frontmatter, _referenced_support_paths,
_validate_bundle_rel_path, _validate_skill_name, hub,
)
logger = logging.getLogger("tools.skills_hub")
# ---------------------------------------------------------------------------
# Well-known Agent Skills endpoint source adapter
# ---------------------------------------------------------------------------
# --- Well-known Agent Skills endpoint source adapter ------------------------
class WellKnownSkillSource(GuardedFetchMixin, SkillSource):
"""Read skills from a domain exposing /.well-known/skills/index.json."""
@@ -28,11 +26,8 @@ class WellKnownSkillSource(GuardedFetchMixin, SkillSource):
def _meta(self, parsed: dict, skill_name: str, description: str, files: Any, **extra) -> SkillMeta:
return SkillMeta(
name=skill_name,
description=description,
source="well-known",
identifier=self._wrap_identifier(parsed["base_url"], skill_name),
trust_level="community",
name=skill_name, description=description, source="well-known",
identifier=self._wrap_identifier(parsed["base_url"], skill_name), trust_level="community",
path=skill_name,
extra={"index_url": parsed["index_url"], "base_url": parsed["base_url"], "files": files, **extra},
)
@@ -42,57 +37,42 @@ class WellKnownSkillSource(GuardedFetchMixin, SkillSource):
parsed = self._parse_index(index_url) if index_url else None
if not parsed:
return []
results: List[SkillMeta] = []
for entry in parsed["skills"][:limit]:
name = entry.get("name")
if not isinstance(name, str) or not name:
continue
files = entry.get("files", ["SKILL.md"])
results.append(self._meta(
parsed, name, str(entry.get("description", "")),
files if isinstance(files, list) else ["SKILL.md"],
))
results.append(self._meta(parsed, name, str(entry.get("description", "")),
files if isinstance(files, list) else ["SKILL.md"]))
return results
def inspect(self, identifier: str) -> Optional[SkillMeta]:
parsed = self._parse_identifier(identifier)
if not parsed:
return None
entry = self._index_entry(parsed["index_url"], parsed["skill_name"])
if not entry:
return None
skill_md = self._fetch_text(f"{parsed['skill_url']}/SKILL.md")
entry = self._index_entry(parsed["index_url"], parsed["skill_name"]) if parsed else None
skill_md = self._fetch_text(f"{parsed['skill_url']}/SKILL.md") if entry else None
if skill_md is None:
return None
fm = _parse_frontmatter(skill_md)
return self._meta(
parsed, str(fm.get("name") or parsed["skill_name"]),
str(fm.get("description") or entry.get("description") or ""),
entry.get("files", ["SKILL.md"]), endpoint=parsed["skill_url"],
)
return self._meta(parsed, str(fm.get("name") or parsed["skill_name"]),
str(fm.get("description") or entry.get("description") or ""),
entry.get("files", ["SKILL.md"]), endpoint=parsed["skill_url"])
def fetch(self, identifier: str) -> Optional[SkillBundle]:
parsed = self._parse_identifier(identifier)
if not parsed:
return None
try:
skill_name = _validate_skill_name(parsed["skill_name"])
except ValueError:
logger.warning("Well-known skill identifier contained unsafe skill name: %s", identifier)
return None
entry = self._index_entry(parsed["index_url"], parsed["skill_name"])
if not entry:
return None
files = entry.get("files", ["SKILL.md"])
if not isinstance(files, list) or not files:
files = ["SKILL.md"]
downloaded: Dict[str, str] = {}
for rel_path in files:
if not isinstance(rel_path, str) or not rel_path:
@@ -100,32 +80,19 @@ class WellKnownSkillSource(GuardedFetchMixin, SkillSource):
try:
safe_rel_path = _validate_bundle_rel_path(rel_path)
except ValueError:
logger.warning(
"Well-known skill %s advertised unsafe file path: %r",
identifier,
rel_path,
)
logger.warning("Well-known skill %s advertised unsafe file path: %r", identifier, rel_path)
return None
text = self._fetch_text(f"{parsed['skill_url']}/{safe_rel_path}")
if text is None:
return None
downloaded[safe_rel_path] = text
if "SKILL.md" not in downloaded:
return None
return SkillBundle(
name=skill_name,
files=downloaded,
source="well-known",
identifier=self._wrap_identifier(parsed["base_url"], skill_name),
trust_level="community",
metadata={
"index_url": parsed["index_url"],
"base_url": parsed["base_url"],
"endpoint": parsed["skill_url"],
"files": files,
},
name=skill_name, files=downloaded, source="well-known",
identifier=self._wrap_identifier(parsed["base_url"], skill_name), trust_level="community",
metadata={"index_url": parsed["index_url"], "base_url": parsed["base_url"],
"endpoint": parsed["skill_url"], "files": files},
)
def _query_to_index_url(self, query: str) -> Optional[str]:
@@ -135,36 +102,27 @@ class WellKnownSkillSource(GuardedFetchMixin, SkillSource):
if query.endswith("/index.json"):
return query
if f"{self.BASE_PATH}/" in query:
base_url = query.split(f"{self.BASE_PATH}/", 1)[0] + self.BASE_PATH
return f"{base_url}/index.json"
return query.split(f"{self.BASE_PATH}/", 1)[0] + f"{self.BASE_PATH}/index.json"
return query.rstrip("/") + f"{self.BASE_PATH}/index.json"
def _parse_identifier(self, identifier: str) -> Optional[dict]:
raw = identifier[len("well-known:"):] if identifier.startswith("well-known:") else identifier
if not raw.startswith(("http://", "https://")):
return None
parsed_url = urlparse(raw)
clean_url = urlunparse(parsed_url._replace(fragment=""))
fragment = parsed_url.fragment
if clean_url.endswith("/index.json"):
if not fragment:
if not parsed_url.fragment:
return None
base_url = clean_url[:-len("/index.json")]
skill_name = fragment
base_url, skill_name = clean_url[:-len("/index.json")], parsed_url.fragment
skill_url = f"{base_url}/{skill_name}"
else:
skill_url = clean_url[:-len("/SKILL.md")] if clean_url.endswith("/SKILL.md") else clean_url.rstrip("/")
if f"{self.BASE_PATH}/" not in skill_url:
return None
base_url, skill_name = skill_url.rsplit("/", 1)
return {
"index_url": f"{base_url}/index.json",
"base_url": base_url,
"skill_name": skill_name,
"skill_url": skill_url,
}
return {"index_url": f"{base_url}/index.json", "base_url": base_url,
"skill_name": skill_name, "skill_url": skill_url}
def _parse_index(self, index_url: str) -> Optional[dict]:
def compute():
@@ -180,42 +138,32 @@ class WellKnownSkillSource(GuardedFetchMixin, SkillSource):
return None
return {"index_url": index_url, "base_url": index_url[:-len("/index.json")], "skills": skills}
return _memo_json(
f"well_known_index_{hashlib.md5(index_url.encode()).hexdigest()}", compute,
valid=lambda c: isinstance(c, dict) and isinstance(c.get("skills"), list),
)
return _memo_json(f"well_known_index_{hashlib.md5(index_url.encode()).hexdigest()}", compute,
valid=lambda c: isinstance(c, dict) and isinstance(c.get("skills"), list))
def _index_entry(self, index_url: str, skill_name: str) -> Optional[dict]:
parsed = self._parse_index(index_url)
if not parsed:
return None
for entry in parsed["skills"]:
if isinstance(entry, dict) and entry.get("name") == skill_name:
return entry
return None
skills = parsed["skills"] if parsed else []
return next((e for e in skills if isinstance(e, dict) and e.get("name") == skill_name), None)
@staticmethod
def _wrap_identifier(base_url: str, skill_name: str) -> str:
return f"well-known:{base_url.rstrip('/')}/{skill_name}"
# ---------------------------------------------------------------------------
# Direct URL source adapter
# ---------------------------------------------------------------------------
# --- Direct URL source adapter ----------------------------------------------
class UrlSource(GuardedFetchMixin, SkillSource):
"""Fetch SKILL.md plus explicitly referenced, allowlisted support files.
The identifier IS the URL (``https://example.com/path/SKILL.md``). Bare URLs
cannot enumerate a repository, so only exact references below
references/templates/scripts/assets are fetched. The skill name comes from
frontmatter ``name:`` (URL-slug fallback); trust is always ``community``.
The identifier IS the URL (``https://example.com/path/SKILL.md``). Bare URLs cannot enumerate a
repository, so only exact references below references/templates/scripts/assets are fetched. The
skill name comes from frontmatter ``name:`` (URL-slug fallback); trust is always ``community``.
"""
SOURCE_ID = "url"
# Skill names must look like identifiers: lowercase letters/digits with
# optional hyphens/underscores. Blocks dangerous (``../evil``) AND useless
# (``SKILL``, ``README``, empty) candidates before they hit the disk.
# Skill names must look like identifiers: lowercase letters/digits with optional hyphens/underscores.
# Blocks dangerous (``../evil``) AND useless (``SKILL``, ``README``, empty) candidates before they hit the disk.
_VALID_NAME_RE = re.compile(r"^[a-z][a-z0-9_-]*$")
def search(self, query: str, limit: int = 10) -> List[SkillMeta]:
@@ -227,15 +175,13 @@ class UrlSource(GuardedFetchMixin, SkillSource):
if not isinstance(identifier, str):
return False
ident = identifier.strip()
if not ident.lower().startswith(("http://", "https://")):
return False
if "/.well-known/skills/" in ident or ident.rstrip("/").endswith("/index.json"):
if (not ident.lower().startswith(("http://", "https://")) or "/.well-known/skills/" in ident
or ident.rstrip("/").endswith("/index.json")):
return False
try:
path = urlparse(ident).path
return urlparse(ident).path.lower().endswith(".md")
except ValueError:
return False
return path.lower().endswith(".md")
def _load(self, identifier: str):
"""``(url, text, frontmatter, resolved name)`` for a claimed identifier, else None."""
@@ -255,12 +201,8 @@ class UrlSource(GuardedFetchMixin, SkillSource):
url, _text, fm, name = loaded
raw_tags = _hermes_tags(fm)
return SkillMeta(
name=name or "",
description=str(fm.get("description") or ""),
source="url",
identifier=url,
trust_level="community",
path=name or "",
name=name or "", description=str(fm.get("description") or ""), source="url", identifier=url,
trust_level="community", path=name or "",
tags=[str(t) for t in raw_tags] if isinstance(raw_tags, list) else [],
extra={"url": url, "awaiting_name": name is None},
)
@@ -280,19 +222,13 @@ class UrlSource(GuardedFetchMixin, SkillSource):
if urlparse(support_url).netloc != urlparse(url).netloc:
return None
content = self._fetch_bytes(support_url)
if content is None:
# A 404ing support file shouldn't sink the whole install.
logger.warning(
"URL skill %s: referenced support file %r could not be "
"fetched from %s; skipping it",
url, rel_path, support_url,
)
if content is None: # A 404ing support file shouldn't sink the whole install.
logger.warning("URL skill %s: referenced support file %r could not be fetched from %s; skipping it",
url, rel_path, support_url)
continue
files[rel_path] = content
# When no name resolves, return the bundle with an empty name and
# ``awaiting_name=True``: ``do_install`` prompts on a TTY or refuses
# non-interactively, without re-downloading after the user picks a name.
# When no name resolves, return the bundle with an empty name and ``awaiting_name=True``: ``do_install``
# prompts on a TTY or refuses non-interactively, without re-downloading after the user picks a name.
skill_name = ""
if name is not None:
try:
@@ -300,37 +236,25 @@ class UrlSource(GuardedFetchMixin, SkillSource):
except ValueError:
logger.warning("URL skill %s produced unsafe skill name: %r", url, name)
return None
return SkillBundle(
name=skill_name,
files=files,
source="url",
identifier=url,
trust_level="community",
metadata={"url": url, "source_url": url, "awaiting_name": not skill_name},
)
return SkillBundle(name=skill_name, files=files, source="url", identifier=url, trust_level="community",
metadata={"url": url, "source_url": url, "awaiting_name": not skill_name})
@classmethod
def _is_valid_skill_name(cls, name: Optional[str]) -> bool:
if not isinstance(name, str):
return False
candidate = name.strip().lower()
if not candidate or candidate in {"skill", "readme", "index", "unnamed-skill"}:
return False
return bool(cls._VALID_NAME_RE.match(candidate))
return bool(candidate) and candidate not in {"skill", "readme", "index", "unnamed-skill"} and bool(
cls._VALID_NAME_RE.match(candidate))
@classmethod
def _resolve_skill_name(cls, fm: dict, url: str) -> Optional[str]:
"""Frontmatter ``name:`` when valid, else a URL-slug candidate
(``.../<name>/SKILL.md`` -> ``<name>``, ``.../<name>.md`` -> ``<name>``).
None when nothing usable — the CLI then prompts or refuses rather than
auto-naming something like ``SKILL``.
"""
"""Frontmatter ``name:`` when valid, else a URL-slug candidate (``.../<name>/SKILL.md`` -> ``<name>``,
``.../<name>.md`` -> ``<name>``). None when nothing usable — the CLI then prompts or refuses rather
than auto-naming something like ``SKILL``."""
fm_name = fm.get("name") if isinstance(fm, dict) else None
if isinstance(fm_name, str) and cls._is_valid_skill_name(fm_name):
return fm_name.strip()
try:
path = urlparse(url).path
except ValueError:
@@ -338,16 +262,11 @@ class UrlSource(GuardedFetchMixin, SkillSource):
parts = [p for p in path.split("/") if p]
if len(parts) >= 2 and parts[-1].lower() == "skill.md" and cls._is_valid_skill_name(parts[-2]):
return parts[-2]
if parts:
candidate = re.sub(r"\.md$", "", parts[-1], flags=re.IGNORECASE)
if cls._is_valid_skill_name(candidate):
return candidate
return None
candidate = re.sub(r"\.md$", "", parts[-1], flags=re.IGNORECASE) if parts else ""
return candidate if cls._is_valid_skill_name(candidate) else None
# ---------------------------------------------------------------------------
# LobeHub source adapter
# ---------------------------------------------------------------------------
# --- LobeHub source adapter -------------------------------------------------
class LobeHubSource(SkillSource):
"""LobeHub agent marketplace (14,500+ system-prompt agents, converted to
@@ -358,9 +277,7 @@ class LobeHubSource(SkillSource):
def _agents(self) -> Optional[list]:
index = self._fetch_index()
if not index:
return None
agents = index.get("agents", index) if isinstance(index, dict) else index
agents = (index.get("agents", index) if isinstance(index, dict) else index) if index else None
return agents if isinstance(agents, list) else None
@staticmethod
@@ -370,14 +287,8 @@ class LobeHubSource(SkillSource):
@staticmethod
def _agent_meta(agent: dict, name: str, description: str) -> SkillMeta:
tags = agent.get("meta", agent).get("tags", [])
return SkillMeta(
name=name,
description=description,
source="lobehub",
identifier=f"lobehub/{name}",
trust_level="community",
tags=tags if isinstance(tags, list) else [],
)
return SkillMeta(name=name, description=description, source="lobehub", identifier=f"lobehub/{name}",
trust_level="community", tags=tags if isinstance(tags, list) else [])
def search(self, query: str, limit: int = 10) -> List[SkillMeta]:
agents = self._agents()
@@ -403,28 +314,18 @@ class LobeHubSource(SkillSource):
agent_data = self._fetch_agent(agent_id)
if not agent_data:
return None
return SkillBundle(
name=agent_id,
files={"SKILL.md": self._convert_to_skill_md(agent_data)},
source="lobehub",
identifier=f"lobehub/{agent_id}",
trust_level="community",
)
return SkillBundle(name=agent_id, files={"SKILL.md": self._convert_to_skill_md(agent_data)}, source="lobehub",
identifier=f"lobehub/{agent_id}", trust_level="community")
def inspect(self, identifier: str) -> Optional[SkillMeta]:
agent_id = self._agent_id(identifier)
for agent in self._agents() or []:
if agent.get("identifier") == agent_id:
return self._agent_meta(agent, agent_id, agent.get("meta", agent).get("description", ""))
return None
agent = next((a for a in self._agents() or [] if a.get("identifier") == agent_id), None)
return self._agent_meta(agent, agent_id, agent.get("meta", agent).get("description", "")) if agent else None
def _fetch_index(self) -> Optional[Any]:
"""Fetch the LobeHub agent index (cached for 1 hour)."""
return _memo_json("lobehub_index", lambda: _get_json(self.INDEX_URL, timeout=30))
def _fetch_agent(self, agent_id: str) -> Optional[dict]:
"""Fetch a single agent's JSON file."""
return _get_json(f"https://chat-agents.lobehub.com/{agent_id}.json", timeout=15)
@staticmethod
@@ -435,44 +336,22 @@ class LobeHubSource(SkillSource):
title = meta.get("title", identifier)
description = meta.get("description", "")
tags = meta.get("tags", [])
system_role = agent_data.get("config", {}).get("systemRole", "")
tag_list = tags if isinstance(tags, list) else []
fm_lines = [
"---",
f"name: {identifier}",
f"description: {description[:500]}",
"metadata:",
" hermes:",
f" tags: [{', '.join(str(t) for t in tag_list)}]",
" lobehub:",
" source: lobehub",
"---",
]
body_lines = [
f"# {title}",
"",
description,
"",
"## Instructions",
"",
system_role if system_role else "(No system role defined)",
]
system_role = agent_data.get("config", {}).get("systemRole", "")
fm_lines = ["---", f"name: {identifier}", f"description: {description[:500]}", "metadata:", " hermes:",
f" tags: [{', '.join(str(t) for t in tag_list)}]", " lobehub:", " source: lobehub", "---"]
body_lines = [f"# {title}", "", description, "", "## Instructions", "",
system_role if system_role else "(No system role defined)"]
return "\n".join(fm_lines) + "\n\n" + "\n".join(body_lines) + "\n"
# ---------------------------------------------------------------------------
# browse.sh source adapter
# ---------------------------------------------------------------------------
# --- browse.sh source adapter -----------------------------------------------
class BrowseShSource(SkillSource):
"""Browserbase's browse.sh catalog of site-specific browser-automation SKILL.md files.
The catalog is ``/api/skills``; content comes from ``/api/skills/{slug}``'s
``skillMdUrl`` (CDN blob). The catalog's ``sourceUrl`` is a GitHub HTML URL
whose repo is not always public, so it is not used for content.
The catalog is ``/api/skills``; content comes from ``/api/skills/{slug}``'s ``skillMdUrl`` (CDN blob).
The catalog's ``sourceUrl`` is a GitHub HTML URL whose repo is not always public, so it is not used for content.
"""
SOURCE_ID = "browse-sh"
@@ -491,28 +370,17 @@ class BrowseShSource(SkillSource):
def _item_to_meta(self, item: Dict) -> Optional[SkillMeta]:
slug = item.get("slug", "")
name = item.get("name", "")
title = item.get("title", name)
description = item.get("description", title)
description = item.get("description", item.get("title", name))
if not slug or not name:
return None
if len(description) > 1024:
description = description[:1021] + "..."
return SkillMeta(
name=name,
description=description,
source="browse-sh",
identifier=f"browse-sh/{slug}",
trust_level="community",
tags=item.get("tags", []),
extra={
"slug": slug,
"hostname": item.get("hostname", ""),
"category": item.get("category", ""),
"source_url": item.get("sourceUrl", ""),
"recommended_method": item.get("recommendedMethod", ""),
"proxies": item.get("proxies", False),
"install_count": item.get("installCount", 0),
},
name=name, description=description, source="browse-sh", identifier=f"browse-sh/{slug}",
trust_level="community", tags=item.get("tags", []),
extra={"slug": slug, "hostname": item.get("hostname", ""), "category": item.get("category", ""),
"source_url": item.get("sourceUrl", ""), "recommended_method": item.get("recommendedMethod", ""),
"proxies": item.get("proxies", False), "install_count": item.get("installCount", 0)},
)
def search(self, query: str, limit: int = 10) -> List[SkillMeta]:
@@ -524,9 +392,7 @@ class BrowseShSource(SkillSource):
def _catalog_item(self, identifier: str) -> Optional[Dict]:
slug = self._slug_from_identifier(identifier)
if not slug:
return None
return next((i for i in self._fetch_catalog() if i.get("slug") == slug), None)
return next((i for i in self._fetch_catalog() if i.get("slug") == slug), None) if slug else None
def inspect(self, identifier: str) -> Optional[SkillMeta]:
item = self._catalog_item(identifier)
@@ -537,44 +403,28 @@ class BrowseShSource(SkillSource):
if not item:
return None
slug = item["slug"]
md_url = self._resolve_skill_md_url(slug, item)
content = _get_text(md_url, follow_redirects=True) if md_url else None
if content is None:
return None
meta = self._item_to_meta(item)
return SkillBundle(
name=meta.name if meta else slug.split("/")[-1],
files={"SKILL.md": content},
source="browse-sh",
identifier=identifier,
trust_level="community",
metadata={
"slug": slug,
"hostname": item.get("hostname", ""),
"source_url": item.get("sourceUrl", ""),
"skill_md_url": md_url,
},
name=meta.name if meta else slug.split("/")[-1], files={"SKILL.md": content}, source="browse-sh",
identifier=identifier, trust_level="community",
metadata={"slug": slug, "hostname": item.get("hostname", ""), "source_url": item.get("sourceUrl", ""),
"skill_md_url": md_url},
)
def _resolve_skill_md_url(self, slug: str, item: Dict) -> Optional[str]:
"""``skillMdUrl`` from ``/api/skills/{slug}``; fallback to a
``raw.githubusercontent.com`` catalog ``sourceUrl`` when present."""
"""``skillMdUrl`` from ``/api/skills/{slug}``; fallback to a ``raw.githubusercontent.com`` ``sourceUrl``."""
data = _get_json(self.SKILL_DETAIL_URL.format(slug=slug), follow_redirects=True)
if isinstance(data, dict):
md_url = data.get("skillMdUrl")
if isinstance(md_url, str) and md_url.startswith("http"):
return md_url
md_url = data.get("skillMdUrl") if isinstance(data, dict) else None
if isinstance(md_url, str) and md_url.startswith("http"):
return md_url
source_url = item.get("sourceUrl", "") if isinstance(item, dict) else ""
from utils import base_url_host_matches
if source_url and base_url_host_matches(source_url, "raw.githubusercontent.com"):
return source_url
return None
return source_url if source_url and base_url_host_matches(source_url, "raw.githubusercontent.com") else None
def _slug_from_identifier(self, identifier: str) -> str:
"""'browse-sh/airbnb.com/search-listings-abc' -> 'airbnb.com/search-listings-abc'."""
if identifier.startswith("browse-sh/"):
return identifier[len("browse-sh/"):]
return identifier
return identifier[len("browse-sh/"):] if identifier.startswith("browse-sh/") else identifier