diff --git a/tools/skills_hub_clawhub.py b/tools/skills_hub_clawhub.py index 3f63ac38db..1c46cabbf6 100644 --- a/tools/skills_hub_clawhub.py +++ b/tools/skills_hub_clawhub.py @@ -33,28 +33,17 @@ def _search_score(query: str, meta: SkillMeta) -> int: query_norm = query.strip().lower() if not query_norm: return 1 - - identifier = (meta.identifier or "").lower() - name = (meta.name or "").lower() + identifier, name = (meta.identifier or "").lower(), (meta.name or "").lower() description = (meta.description or "").lower() - query_terms = _query_terms(query_norm) - identifier_terms = _query_terms(identifier) - name_terms = _query_terms(name) - normalized_identifier = " ".join(identifier_terms) - normalized_name = " ".join(name_terms) - + query_terms, identifier_terms, name_terms = _query_terms(query_norm), _query_terms(identifier), _query_terms(name) + normalized_identifier, normalized_name = " ".join(identifier_terms), " ".join(name_terms) checks = ( - (140, query_norm == identifier), - (130, query_norm == name), - (125, normalized_identifier == query_norm), - (120, normalized_name == query_norm), - (95, normalized_identifier.startswith(query_norm)), - (90, normalized_name.startswith(query_norm)), + (140, query_norm == identifier), (130, query_norm == name), + (125, normalized_identifier == query_norm), (120, normalized_name == query_norm), + (95, normalized_identifier.startswith(query_norm)), (90, normalized_name.startswith(query_norm)), (70, bool(query_terms) and identifier_terms[: len(query_terms)] == query_terms), (65, bool(query_terms) and name_terms[: len(query_terms)] == query_terms), - (40, query_norm in identifier), - (35, query_norm in name), - (10, query_norm in description), + (40, query_norm in identifier), (35, query_norm in name), (10, query_norm in description), ) score = sum(points for points, hit in checks if hit) for term in query_terms: @@ -68,24 +57,20 @@ def _first_str(*values: Any) -> Optional[str]: class ClawHubSource(GuardedFetchMixin, SkillSource): - """ClawHub (clawhub.ai) HTTP API. Every skill is community trust — the - ClawHavoc incident (341 malicious skills, Feb 2026) showed their vetting - is insufficient.""" + """ClawHub (clawhub.ai) HTTP API. Every skill is community trust — the ClawHavoc + incident (341 malicious skills, Feb 2026) showed their vetting is insufficient.""" SOURCE_ID = "clawhub" BASE_URL = "https://clawhub.ai/api/v1" - # Wall-clock budget for a full catalog walk: 50k+ skills, sequential # (~250 requests each under timeout=30), so unbounded it blocks for minutes. CATALOG_WALK_BUDGET_SECONDS = 12 - _SLUG_RE = re.compile(r"[A-Za-z0-9][A-Za-z0-9._-]*$") _query_terms = staticmethod(_query_terms) _dedupe_results = staticmethod(_dedupe_results) _search_score = staticmethod(_search_score) - - # -- payload helpers --------------------------------------------------- + _get_json = staticmethod(_get_json) @staticmethod def _normalize_tags(tags: Any) -> List[str]: @@ -113,9 +98,7 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): @staticmethod def _owner_from_payload(data: Optional[Dict[str, Any]]) -> Optional[str]: - if not isinstance(data, dict): - return None - owner = data.get("owner") + owner = data.get("owner") if isinstance(data, dict) else None if isinstance(owner, dict): owner = owner.get("handle") return owner.strip() if isinstance(owner, str) and owner.strip() else None @@ -137,11 +120,8 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): return SkillMeta( name=item.get("displayName") or item.get("name") or slug, description=item.get("summary") or item.get("description") or "", - source="clawhub", - identifier=slug, - trust_level="community", - tags=cls._normalize_tags(item.get("tags", [])), - extra={"owner": owner} if owner else {}, + source="clawhub", identifier=slug, trust_level="community", + tags=cls._normalize_tags(item.get("tags", [])), extra={"owner": owner} if owner else {}, ) def _skill_detail(self, identifier: str) -> Optional[Tuple[str, Dict[str, Any]]]: @@ -156,64 +136,38 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): return None return slug, data - # -- search / ranking -------------------------------------------------- - def _exact_slug_meta(self, query: str) -> Optional[SkillMeta]: query = query.strip() - parsed = self._parse_identifier(query) - query_terms = _query_terms(query) - candidates: List[str] = [] - - if parsed: - candidates.append(parsed[0]) - elif "/" not in query and self._SLUG_RE.fullmatch(query): - candidates.append(query) - + parsed, query_terms = self._parse_identifier(query), _query_terms(query) + candidates = [parsed[0]] if parsed else [query] if "/" not in query and self._SLUG_RE.fullmatch(query) else [] if query_terms: base_slug = "-".join(query_terms) if len(query_terms) >= 2: - candidates.extend( - f"{base_slug}-{suffix}" for suffix in ("agent", "skill", "tool", "assistant", "playbook") - ) + candidates.extend(f"{base_slug}-{s}" for s in ("agent", "skill", "tool", "assistant", "playbook")) candidates.append(base_slug) - - for candidate in dict.fromkeys(candidates): - meta = self.inspect(candidate) - if meta: - return meta - return None + return next((m for m in map(self.inspect, dict.fromkeys(candidates)) if m), None) def _finalize_search_results(self, query: str, results: List[SkillMeta], limit: int) -> List[SkillMeta]: query_norm = query.strip() if not query_norm: return _dedupe_results(results)[:limit] - filtered = [meta for meta in results if _search_score(query_norm, meta) > 0] filtered.sort(key=lambda meta: (-_search_score(query_norm, meta), meta.name.lower(), meta.identifier.lower())) filtered = _dedupe_results(filtered) - exact = self._exact_slug_meta(query_norm) if exact: - filtered = [meta for meta in filtered if _search_score(query_norm, meta) >= 20] - filtered = _dedupe_results([exact] + filtered) - - if filtered: + filtered = _dedupe_results([exact] + [m for m in filtered if _search_score(query_norm, m) >= 20]) + if filtered or re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._/-]*", query_norm): return filtered[:limit] - - if re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._/-]*", query_norm): - return [] - return _dedupe_results(results)[:limit] def search(self, query: str, limit: int = 10) -> List[SkillMeta]: query = query.strip() - if query: if len(_query_terms(query)) >= 2: direct = self._exact_slug_meta(query) if direct: return [direct] - results = self._search_catalog(query, limit=limit) if results: return results @@ -222,7 +176,7 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): # is returned whole (caller paginates); on a cold cache the walk is # bounded to `limit` so browse renders page one without walking # 50k+ skills (max_items=0 = unbounded, offline index builder only). - catalog = self._load_catalog_index(max_items=limit if limit > 0 else 0) + catalog = self._load_catalog_index(max_items=max(limit, 0)) if catalog: deduped = _dedupe_results(catalog) return deduped[:limit] if limit > 0 else deduped @@ -232,16 +186,11 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): cached = _cached_metas(cache_key) if cached is not None: return self._finalize_search_results(query, cached, limit) - - data = self._get_json(f"{self.BASE_URL}/skills", timeout=15, - params={"search": query, "limit": limit}) - if data is None: - return [] + data = self._get_json(f"{self.BASE_URL}/skills", timeout=15, params={"search": query, "limit": limit}) skills_data = data.get("items", data) if isinstance(data, dict) else data if not isinstance(skills_data, list): return [] - - results = [m for m in (self._item_to_meta(item) for item in skills_data[:limit]) if m] + results = [m for m in map(self._item_to_meta, skills_data[:limit]) if m] final_results = self._finalize_search_results(query, results, limit) _cache_metas(cache_key, final_results) return final_results @@ -259,31 +208,22 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): if not raw: return None had_at = raw.startswith("@") - ident = raw[1:] if had_at else raw - if ident.startswith("clawhub/"): - ident = ident[len("clawhub/"):] - parts = [part for part in ident.split("/") if part] - owner = slug = None + parts = [part for part in raw.removeprefix("@").removeprefix("clawhub/").split("/") if part] if len(parts) == 1: - slug = parts[0] - elif len(parts) == 2 and had_at: - owner, slug = parts - elif len(parts) == 3 and parts[1].lower() == "skills": - owner, _, slug = parts + owner, slug = None, parts[0] + elif (len(parts) == 2 and had_at) or (len(parts) == 3 and parts[1].lower() == "skills"): + owner, slug = parts[0], parts[-1] else: return None if not cls._SLUG_RE.fullmatch(slug) or (owner is not None and not cls._SLUG_RE.fullmatch(owner)): return None return slug, owner - # -- fetch / inspect --------------------------------------------------- - def fetch(self, identifier: str) -> Optional[SkillBundle]: detail = self._skill_detail(identifier) if detail is None: return None slug, skill_data = detail - latest_version = self._resolve_latest_version(slug, skill_data) if not latest_version: logger.warning("ClawHub fetch failed for %s: could not resolve latest version", slug) @@ -296,26 +236,15 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): version_data = self._get_json(f"{self.BASE_URL}/skills/{slug}/versions/{latest_version}") if isinstance(version_data, dict): files = self._extract_files(version_data) or files - if "SKILL.md" not in files: - nested = version_data.get("version", {}) - if isinstance(nested, dict): - files = self._extract_files(nested) or files - + nested = version_data.get("version", {}) + if "SKILL.md" not in files and isinstance(nested, dict): + files = self._extract_files(nested) or files if "SKILL.md" not in files: logger.warning( - "ClawHub fetch for %s resolved version %s but could not retrieve file content", - slug, - latest_version, + "ClawHub fetch for %s resolved version %s but could not retrieve file content", slug, latest_version, ) return None - - return SkillBundle( - name=slug, - files=files, - source="clawhub", - identifier=slug, - trust_level="community", - ) + return SkillBundle(name=slug, files=files, source="clawhub", identifier=slug, trust_level="community") def inspect(self, identifier: str) -> Optional[SkillMeta]: detail = self._skill_detail(identifier) @@ -329,11 +258,9 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): cached = _cached_metas(cache_key) if cached is not None: return cached[:limit] - catalog = self._load_catalog_index() if not catalog: return [] - results = self._finalize_search_results(query, catalog, limit) _cache_metas(cache_key, results) return results @@ -352,7 +279,6 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): cached = _cached_metas(cache_key) if cached is not None: return cached - cursor: Optional[str] = None results: List[SkillMeta] = [] seen: set[str] = set() @@ -362,57 +288,42 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): # (max_items=0) must walk everything or it trips the deploy health floor. deadline = time.monotonic() + self.CATALOG_WALK_BUDGET_SECONDS if max_items > 0 else None partial = False - for _ in range(750): if deadline is not None and time.monotonic() > deadline: partial = True break - params: Dict[str, Any] = {"limit": 200} - if cursor: - params["cursor"] = cursor - + params: Dict[str, Any] = {"limit": 200, "cursor": cursor} if cursor else {"limit": 200} data = self._get_json(f"{self.BASE_URL}/skills", timeout=30, params=params) items = data.get("items", []) if isinstance(data, dict) else [] if not isinstance(items, list) or not items: break - for item in items: slug = item.get("slug") - if not isinstance(slug, str) or not slug or slug in seen: - continue - seen.add(slug) - meta = self._item_to_meta(item) - if meta: - results.append(meta) - + if isinstance(slug, str) and slug and slug not in seen: + seen.add(slug) + meta = self._item_to_meta(item) + if meta: + results.append(meta) cursor = data.get("nextCursor") if isinstance(data, dict) else None if not isinstance(cursor, str) or not cursor: break - if max_items > 0 and len(results) >= max_items: partial = True break - if not partial: _cache_metas(cache_key, results) return results - _get_json = staticmethod(_get_json) - def _resolve_latest_version(self, slug: str, skill_data: Dict[str, Any]) -> Optional[str]: - latest = skill_data.get("latestVersion") - tags = skill_data.get("tags") + latest, tags = skill_data.get("latestVersion"), skill_data.get("tags") version = _first_str( latest.get("version") if isinstance(latest, dict) else None, tags.get("latest") if isinstance(tags, dict) else None, ) if version: return version - - versions_data = self._get_json(f"{self.BASE_URL}/skills/{slug}/versions") - if isinstance(versions_data, list) and versions_data and isinstance(versions_data[0], dict): - return _first_str(versions_data[0].get("version")) - return None + vd = self._get_json(f"{self.BASE_URL}/skills/{slug}/versions") + return _first_str(vd[0].get("version")) if isinstance(vd, list) and vd and isinstance(vd[0], dict) else None def _fetch_owner_handle(self, slug: str) -> Optional[str]: """Owner handle from the detail API (the listing API lacks it), or None. @@ -423,7 +334,6 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): """ url = f"{self.BASE_URL}/skills/{slug}" max_attempts = 3 - for attempt in range(max_attempts): delay = 2.0 * (2 ** attempt) try: @@ -438,9 +348,8 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): return None return self._owner_from_payload(self._coerce_skill_payload(raw)) if resp.status_code == 429: - retry_after_raw = resp.headers.get("Retry-After") try: - delay = float(retry_after_raw) if retry_after_raw else delay + delay = float(resp.headers.get("Retry-After") or delay) except (TypeError, ValueError): pass reason = "HTTP 429" @@ -450,12 +359,9 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): return None # 4xx (non-429): doesn't exist / bad request if attempt >= max_attempts - 1: return None - logger.debug( - "_fetch_owner_handle(%s): %s on attempt %d/%d, retrying in %.1fs", - slug, reason, attempt + 1, max_attempts, delay, - ) + logger.debug("_fetch_owner_handle(%s): %s on attempt %d/%d, retrying in %.1fs", + slug, reason, attempt + 1, max_attempts, delay) time.sleep(delay) - return None def enrich_owners(self, skills: List[SkillMeta], max_workers: int = 30) -> int: @@ -466,13 +372,9 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): Safety rails: aborts after 50 consecutive failures (systemic outage), per-request 429 backoff, progress log every 1000 skills. """ - needs_enrichment = [ - s for s in skills - if s.source == "clawhub" and not (s.extra or {}).get("owner") - ] + needs_enrichment = [s for s in skills if s.source == "clawhub" and not (s.extra or {}).get("owner")] if not needs_enrichment: return 0 - enriched = consecutive_failures = processed = 0 max_consecutive_failures = 50 @@ -486,60 +388,44 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): handle = future.result() except Exception: handle = None + consecutive_failures = 0 if handle else consecutive_failures + 1 if handle: - if not meta.extra: - meta.extra = {} + meta.extra = meta.extra or {} meta.extra["owner"] = handle enriched += 1 - consecutive_failures = 0 - else: - consecutive_failures += 1 - if processed % 1000 == 0: - logger.info( - "ClawHub owner enrichment: %d/%d processed, %d enriched", - processed, len(needs_enrichment), enriched, - ) - + logger.info("ClawHub owner enrichment: %d/%d processed, %d enriched", + processed, len(needs_enrichment), enriched) if consecutive_failures >= max_consecutive_failures: logger.warning( "ClawHub owner enrichment: %d consecutive failures — " "aborting early (%d/%d processed, %d enriched). " "The ClawHub API may be down or rate-limited.", - max_consecutive_failures, processed, - len(needs_enrichment), enriched, + max_consecutive_failures, processed, len(needs_enrichment), enriched, ) for f in futures: f.cancel() break - return enriched def _extract_files(self, version_data: Dict[str, Any]) -> Dict[str, str]: files: Dict[str, str] = {} file_list = version_data.get("files") - if isinstance(file_list, dict): return {k: v for k, v in file_list.items() if isinstance(v, str)} - if not isinstance(file_list, list): - return files - - for file_meta in file_list: + for file_meta in file_list if isinstance(file_list, list) else []: if not isinstance(file_meta, dict): continue - fname = file_meta.get("path") or file_meta.get("name") + fname, inline = file_meta.get("path") or file_meta.get("name"), file_meta.get("content") + raw_url = file_meta.get("rawUrl") or file_meta.get("downloadUrl") or file_meta.get("url") if not fname or not isinstance(fname, str): continue - inline_content = file_meta.get("content") - if isinstance(inline_content, str): - files[fname] = inline_content - continue - raw_url = file_meta.get("rawUrl") or file_meta.get("downloadUrl") or file_meta.get("url") - if isinstance(raw_url, str) and raw_url.startswith("http"): + if isinstance(inline, str): + files[fname] = inline + elif isinstance(raw_url, str) and raw_url.startswith("http"): content = self._fetch_text(raw_url) if content is not None: files[fname] = content - return files def _download_zip(self, slug: str, version: str) -> Dict[str, str]: @@ -551,28 +437,20 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): max_retries = 3 for attempt in range(max_retries): try: - resp = httpx.get( - f"{self.BASE_URL}/download", - params={"slug": slug, "version": version}, - timeout=30, - follow_redirects=True, - ) + resp = httpx.get(f"{self.BASE_URL}/download", params={"slug": slug, "version": version}, + timeout=30, follow_redirects=True) if resp.status_code == 429: try: - retry_after = int(resp.headers.get("retry-after", "5")) + retry_after = min(int(resp.headers.get("retry-after", "5")), 15) # Cap wait time except (ValueError, TypeError): retry_after = 5 - retry_after = min(retry_after, 15) # Cap wait time - logger.debug( - "ClawHub download rate-limited for %s, retrying in %ds (attempt %d/%d)", - slug, retry_after, attempt + 1, max_retries, - ) + logger.debug("ClawHub download rate-limited for %s, retrying in %ds (attempt %d/%d)", + slug, retry_after, attempt + 1, max_retries) time.sleep(retry_after) continue if resp.status_code != 200: logger.debug("ClawHub ZIP download for %s v%s returned %s", slug, version, resp.status_code) return files - with zipfile.ZipFile(io.BytesIO(resp.content)) as zf: for info in zf.infolist(): if info.is_dir(): @@ -590,13 +468,11 @@ class ClawHubSource(GuardedFetchMixin, SkillSource): except (UnicodeDecodeError, KeyError): logger.debug("Skipping non-text file in ZIP: %s", name) return files - except zipfile.BadZipFile: logger.warning("ClawHub returned invalid ZIP for %s v%s", slug, version) return files except httpx.HTTPError as exc: logger.debug("ClawHub ZIP download failed for %s v%s: %s", slug, version, exc) return files - logger.debug("ClawHub ZIP download exhausted retries for %s v%s", slug, version) return files diff --git a/tools/skills_hub_github.py b/tools/skills_hub_github.py index d2fd4db9b0..8283a8be9f 100644 --- a/tools/skills_hub_github.py +++ b/tools/skills_hub_github.py @@ -20,19 +20,13 @@ from tools.skills_hub_models import ( logger = logging.getLogger("tools.skills_hub") - -# Maps a GitHub tap repo (owner/repo) to the provider label used by the -# docs-site catalog (website/scripts/extract-skills.py::GITHUB_TAP_LABELS). -# The runtime index collapses every tap into source="github"; the label in -# ``extra.provider`` keeps per-tap identity searchable/filterable without -# disturbing the dedup / floor / index-skip logic keyed on the bare source id. +# GitHub tap repo (owner/repo) -> provider label used by the docs-site catalog +# (website/scripts/extract-skills.py::GITHUB_TAP_LABELS). The runtime index collapses every tap into +# source="github"; ``extra.provider`` keeps per-tap identity searchable/filterable without disturbing +# dedup / floor / index-skip logic keyed on the bare source id. GITHUB_TAP_PROVIDERS = { - "openai/skills": "OpenAI", - "anthropics/skills": "Anthropic", - "huggingface/skills": "HuggingFace", - "nvidia/skills": "NVIDIA", - "voltagent/awesome-agent-skills": "VoltAgent", - "garrytan/gstack": "gstack", + "openai/skills": "OpenAI", "anthropics/skills": "Anthropic", "huggingface/skills": "HuggingFace", + "nvidia/skills": "NVIDIA", "voltagent/awesome-agent-skills": "VoltAgent", "garrytan/gstack": "gstack", "minimax-ai/cli": "MiniMax", } @@ -40,20 +34,19 @@ GITHUB_TAP_PROVIDERS = { # they narrow merged results to GitHub-tap skills carrying that ``extra.provider``. _PROVIDER_FILTER_VALUES = frozenset(v.lower() for v in GITHUB_TAP_PROVIDERS.values()) +_API = "https://api.github.com/repos" +_ACCEPT_JSON = "application/vnd.github.v3+json" + def github_provider_for(repo: str) -> Optional[str]: """Provider label for an ``owner/repo`` tap (case-insensitive), or None.""" - if not repo: - return None - return GITHUB_TAP_PROVIDERS.get(repo.strip().lower()) + return GITHUB_TAP_PROVIDERS.get(repo.strip().lower()) if repo else None def _filter_results_by_provider(results: List[SkillMeta], provider: str) -> List[SkillMeta]: - """Keep only results whose ``extra.provider`` matches ``provider``. - - An explicit provider filter (``--source nvidia``) narrows to exactly that - provider — the official catalog is NOT injected the way unfiltered browse does. - """ + """Keep only results whose ``extra.provider`` matches ``provider``. An explicit provider filter + (``--source nvidia``) narrows to exactly that provider — the official catalog is NOT injected the + way unfiltered browse does.""" want = provider.strip().lower() return [r for r in results if str((r.extra or {}).get("provider", "")).lower() == want] @@ -65,17 +58,9 @@ def _is_rate_limit_response(resp: httpx.Response) -> bool: ) -# --------------------------------------------------------------------------- -# GitHub Authentication -# --------------------------------------------------------------------------- - class GitHubAuth: - """GitHub API authentication, tried in priority order: - 1. GITHUB_TOKEN / GH_TOKEN (PAT) - 2. `gh auth token` (gh CLI) - 3. GitHub App JWT + installation token - 4. Unauthenticated (60 req/hr, public repos only) - """ + """GitHub API authentication, tried in priority order: GITHUB_TOKEN / GH_TOKEN (PAT), `gh auth token` + (gh CLI), GitHub App JWT + installation token, then unauthenticated (60 req/hr, public repos only).""" def __init__(self): self._cached_token: Optional[str] = None @@ -84,10 +69,7 @@ class GitHubAuth: def get_headers(self) -> Dict[str, str]: token = self._resolve_token() - headers = {"Accept": "application/vnd.github.v3+json"} - if token: - headers["Authorization"] = f"token {token}" - return headers + return {"Accept": _ACCEPT_JSON, **({"Authorization": f"token {token}"} if token else {})} def is_authenticated(self) -> bool: return self._resolve_token() is not None @@ -98,11 +80,8 @@ class GitHubAuth: return self._cached_method or "anonymous" def _resolve_token(self) -> Optional[str]: - if self._cached_token and ( - self._cached_method != "github-app" or time.time() < self._app_token_expiry - ): + if self._cached_token and (self._cached_method != "github-app" or time.time() < self._app_token_expiry): return self._cached_token - for method, resolve in ( ("pat", self._try_pat), ("gh-cli", self._try_gh_cli), ("github-app", self._try_github_app), ): @@ -112,7 +91,6 @@ class GitHubAuth: if method == "github-app": self._app_token_expiry = time.time() + 3500 # ~58 min (tokens last 1 hour) return token - self._cached_method = "anonymous" return None @@ -125,10 +103,8 @@ class GitHubAuth: def _try_gh_cli(self) -> Optional[str]: try: result = subprocess.run( - ["gh", "auth", "token"], - capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5, - stdin=subprocess.DEVNULL, - creationflags=windows_hide_flags(), + ["gh", "auth", "token"], capture_output=True, text=True, encoding='utf-8', errors='replace', + timeout=5, stdin=subprocess.DEVNULL, creationflags=windows_hide_flags(), ) if result.returncode == 0 and result.stdout.strip(): return result.stdout.strip() @@ -138,19 +114,15 @@ class GitHubAuth: def _try_github_app(self) -> Optional[str]: from agent.secret_scope import get_secret - app_id = get_secret("GITHUB_APP_ID") - key_path = get_secret("GITHUB_APP_PRIVATE_KEY_PATH") + app_id, key_path = get_secret("GITHUB_APP_ID"), get_secret("GITHUB_APP_PRIVATE_KEY_PATH") installation_id = get_secret("GITHUB_APP_INSTALLATION_ID") - if not all([app_id, key_path, installation_id]): return None - try: import jwt # PyJWT except ImportError: logger.debug("PyJWT not installed, skipping GitHub App auth") return None - try: key_file = Path(key_path) if not key_file.exists(): @@ -162,27 +134,19 @@ class GitHubAuth: ) resp = httpx.post( f"https://api.github.com/app/installations/{installation_id}/access_tokens", - headers={"Authorization": f"Bearer {encoded_jwt}", "Accept": "application/vnd.github.v3+json"}, - timeout=10, + headers={"Authorization": f"Bearer {encoded_jwt}", "Accept": _ACCEPT_JSON}, timeout=10, ) if resp.status_code == 201: return resp.json().get("token") except Exception as e: logger.debug("GitHub App auth failed: %s", e) - return None -# --------------------------------------------------------------------------- -# GitHub source adapter -# --------------------------------------------------------------------------- - def _split_repo_id(identifier: str) -> Optional[Tuple[str, str]]: """``owner/repo/path/to/skill`` -> ``(owner/repo, path/to/skill)``; None when too short.""" parts = identifier.split("/", 2) - if len(parts) < 3: - return None - return f"{parts[0]}/{parts[1]}", parts[2] + return (f"{parts[0]}/{parts[1]}", parts[2]) if len(parts) >= 3 else None def _skip_bundle_file(rel_path: str) -> bool: @@ -192,32 +156,27 @@ def _skip_bundle_file(rel_path: str) -> bool: def _tree_members(entries: List[dict], prefix: str): - """``(rel_path, item_path, is_regular_blob)`` for every git-tree entry under ``prefix``. - - Symlinks (mode 120000) and non-blobs report ``is_regular_blob=False`` so callers - can reject a SKILL.md-linked symlink instead of silently following it. - """ + """``(rel_path, item_path, is_regular_blob)`` for every git-tree entry under ``prefix``. Symlinks + (mode 120000) and non-blobs report ``is_regular_blob=False`` so callers can reject a SKILL.md-linked + symlink instead of silently following it.""" for item in entries: item_path = item.get("path", "") if item_path.startswith(prefix): - regular = item.get("type") == "blob" and item.get("mode") != "120000" - yield item_path[len(prefix):], item_path, regular + yield item_path[len(prefix):], item_path, item.get("type") == "blob" and item.get("mode") != "120000" class GitHubSource(SkillSource): """Fetch skills from GitHub repos via the Contents API.""" DEFAULT_TAPS = [ - # openai/skills keeps its content under skills/.curated/ and - # skills/.system/; _list_skills_in_repo skips "."/"_" directories, - # so both entries point at the inner paths directly. + # openai/skills keeps content under skills/.curated/ + skills/.system/; _list_skills_in_repo + # skips "."/"_" directories, so both entries point at the inner paths. {"repo": "openai/skills", "path": "skills/.curated/"}, {"repo": "openai/skills", "path": "skills/.system/"}, {"repo": "anthropics/skills", "path": "skills/"}, {"repo": "huggingface/skills", "path": "skills/"}, - # NVIDIA-verified skills (CUDA-X, NeMo, cuOpt, ...), each with a - # signed skill.oms.sig + governance card; `trusted` via - # tools/skills_guard.py::TRUSTED_REPOS. + # NVIDIA-verified skills (CUDA-X, NeMo, cuOpt, ...), each with a signed skill.oms.sig + # + governance card; `trusted` via tools/skills_guard.py::TRUSTED_REPOS. {"repo": "NVIDIA/skills", "path": "skills/"}, {"repo": "garrytan/gstack", "path": ""}, ] @@ -227,9 +186,7 @@ class GitHubSource(SkillSource): def __init__(self, auth: GitHubAuth, extra_taps: Optional[List[Dict]] = None): self.auth = auth - self.taps = list(self.DEFAULT_TAPS) - if extra_taps: - self.taps.extend(extra_taps) + self.taps = list(self.DEFAULT_TAPS) + list(extra_taps or []) # Per-instance repo -> (default_branch, tree_entries); lives for one # search/install flow so repeated tree lookups cost no API calls. self._tree_cache: Dict[str, Tuple[str, List[dict]]] = {} @@ -239,22 +196,18 @@ class GitHubSource(SkillSource): self._rate_limited: bool = False @property - def is_rate_limited(self) -> bool: - """Whether GitHub API rate limit was hit during operations.""" + def is_rate_limited(self) -> bool: # whether the GitHub API rate limit was hit during operations return self._rate_limited def trust_level_for(self, identifier: str) -> str: # identifier format: "owner/repo/path/to/skill" parts = identifier.split("/", 2) - if len(parts) >= 2 and f"{parts[0]}/{parts[1]}" in TRUSTED_REPOS: - return "trusted" - return "community" + return "trusted" if len(parts) >= 2 and f"{parts[0]}/{parts[1]}" in TRUSTED_REPOS else "community" def search(self, query: str, limit: int = 10) -> List[SkillMeta]: """Substring-match all taps; dedupe by identifier preferring higher trust.""" results: List[SkillMeta] = [] query_lower = query.lower() - for tap in self.taps: try: for skill in self._list_skills_in_repo(tap["repo"], tap.get("path", "")): @@ -262,22 +215,17 @@ class GitHubSource(SkillSource): results.append(skill) except Exception as e: logger.debug("Failed to search %s: %s", tap['repo'], e) - continue - return _dedupe_by_trust(results)[:limit] def fetch(self, identifier: str) -> Optional[SkillBundle]: """Download a skill; identifier format: "owner/repo/path/to/skill-dir".""" - split = _split_repo_id(identifier) - if split is None: + if (split := _split_repo_id(identifier)) is None: return None repo, skill_path = split skill_dir = skill_path.rstrip("/") - - # Resolve the tree FIRST so every byte fetch — SKILL.md included — is - # pinned to the same revision; an unpinned /contents fetch floats to - # HEAD and can serve bytes newer than the tree the paths were - # validated against (TOCTOU). Idempotent + cached. + # Resolve the tree FIRST so every byte fetch — SKILL.md included — is pinned to the + # same revision; an unpinned /contents fetch floats to HEAD and can serve bytes newer + # than the tree the paths were validated against (TOCTOU). Idempotent + cached. tree = self._get_repo_tree(repo) pinned_ref = self._tree_revisions.get(repo) skill_md = self._fetch_file_content(repo, f"{skill_dir}/SKILL.md", ref=pinned_ref) @@ -286,57 +234,39 @@ class GitHubSource(SkillSource): referenced = _referenced_support_paths(skill_md) if referenced is None: return None - files: Dict[str, Union[str, bytes]] = {"SKILL.md": skill_md} if tree is not None: - branch, entries = tree - if not self._collect_tree_files(repo, skill_dir, entries, pinned_ref, referenced, files): + if not self._collect_tree_files(repo, skill_dir, tree[1], pinned_ref, referenced, files): return None - revision = pinned_ref or branch + revision = pinned_ref or tree[0] else: for rel_path in referenced: - content = self._fetch_file_bytes(repo, f"{skill_dir}/{rel_path}") - if content is None: - logger.warning("Failed to fetch referenced skill support " - "file; continuing without it: %s", rel_path) - continue - files[rel_path] = content + self._add_support_file(repo, f"{skill_dir}/{rel_path}", rel_path, files, rel_path) revision = "" - + url = f"https://github.com/{repo}/" + (f"tree/{revision}/{skill_path}" if revision else skill_path) return SkillBundle( - name=skill_dir.split("/")[-1], - files=files, - source="github", - identifier=identifier, - trust_level=self.trust_level_for(identifier), - metadata={ - "source_url": ( - f"https://github.com/{repo}/tree/{revision}/{skill_path}" - if revision else f"https://github.com/{repo}/{skill_path}" - ), - "source_revision": revision, - }, + name=skill_dir.split("/")[-1], files=files, source="github", identifier=identifier, + trust_level=self.trust_level_for(identifier), metadata={"source_url": url, "source_revision": revision}, ) + def _add_support_file(self, repo: str, item_path: str, rel_path: str, files: dict, shown: str, **kw) -> None: + """Fetch one support file into ``files``; a failed fetch warns (naming ``shown``) and is skipped.""" + content = self._fetch_file_bytes(repo, item_path, **kw) + if content is None: + logger.warning("Failed to fetch referenced skill support file; continuing without it: %s", shown) + else: + files[rel_path] = content + def _collect_tree_files( - self, - repo: str, - skill_path: str, - entries: List[dict], - ref: Optional[str], - referenced: set, + self, repo: str, skill_path: str, entries: List[dict], ref: Optional[str], referenced: set, files: Dict[str, Union[str, bytes]], ) -> bool: - """Download the FULL skill directory from the pinned tree into ``files``. - - Link-driven fetching silently dropped support files under non-canonical - dirs (``reference/``, ``agents/``, root LICENSE); everything still goes - through quarantine + scan, and the scanner sees MORE this way. - Returns False (bundle rejected) on an unsafe path or a SKILL.md-linked - path that exists in the tree as a symlink/non-blob — that shape is an - escape attempt. A linked path that is simply absent is a dangling link - (repo-only dev tool, prose over-match): warn and install without it. - """ + """Download the FULL skill directory from the pinned tree into ``files``. Link-driven fetching + silently dropped support files under non-canonical dirs (``reference/``, ``agents/``, root + LICENSE); everything still goes through quarantine + scan, and the scanner sees MORE this way. + Returns False (bundle rejected) on an unsafe path or a SKILL.md-linked path that exists in the + tree as a symlink/non-blob — that shape is an escape attempt. A linked path that is simply absent + is a dangling link (repo-only dev tool, prose over-match): warn and install without it.""" prefix = f"{skill_path}/" symlinked: set = set() for rel_path, item_path, regular in _tree_members(entries, prefix): @@ -350,55 +280,31 @@ class GitHubSource(SkillSource): except ValueError: logger.warning("Rejected unsafe file path in skill bundle: %s", item_path) return False - content = self._fetch_file_bytes(repo, item_path, ref=ref) - if content is None: - logger.warning("Failed to fetch referenced skill support " - "file; continuing without it: %s", item_path) - continue - files[rel_path] = content + self._add_support_file(repo, item_path, rel_path, files, item_path, ref=ref) for rel_path in sorted(referenced): if rel_path in symlinked: - logger.warning( - "Rejected non-regular referenced file in skill " - "bundle: %s%s", prefix, rel_path, - ) + logger.warning("Rejected non-regular referenced file in skill bundle: %s%s", prefix, rel_path) return False if rel_path not in files: logger.warning( - "Referenced skill support file is missing; " - "continuing without it: %s%s", - prefix, rel_path, - ) + "Referenced skill support file is missing; continuing without it: %s%s", prefix, rel_path) return True def inspect(self, identifier: str) -> Optional[SkillMeta]: """Fetch just the SKILL.md metadata for preview.""" - split = _split_repo_id(identifier) - if split is None: + if (split := _split_repo_id(identifier)) is None: return None - repo, skill_path = split - skill_path = skill_path.rstrip("/") - + repo, skill_path = split[0], split[1].rstrip("/") content = self._fetch_file_content(repo, f"{skill_path}/SKILL.md") if not content: return None - fm = _parse_frontmatter(content) - tags = _hermes_tags(fm) - if not tags: - raw_tags = fm.get("tags", []) - tags = raw_tags if isinstance(raw_tags, list) else [] - + tags = _hermes_tags(fm) or (fm["tags"] if isinstance(fm.get("tags"), list) else []) provider = github_provider_for(repo) return SkillMeta( - name=fm.get("name", skill_path.split("/")[-1]), - description=str(fm.get("description", "")), - source="github", - identifier=identifier, - trust_level=self.trust_level_for(identifier), - repo=repo, - path=skill_path, - tags=[str(t) for t in tags], + name=fm.get("name", skill_path.split("/")[-1]), description=str(fm.get("description", "")), + source="github", identifier=identifier, trust_level=self.trust_level_for(identifier), + repo=repo, path=skill_path, tags=[str(t) for t in tags], extra={"provider": provider} if provider else {}, ) @@ -410,105 +316,69 @@ class GitHubSource(SkillSource): cached = _cached_metas(cache_key) if cached is not None: return cached - - url = f"https://api.github.com/repos/{repo}/contents/{path.rstrip('/')}" - resp = self._github_get(url) + resp = self._github_get(f"{_API}/{repo}/contents/{path.rstrip('/')}") if resp is None or resp.status_code != 200: return [] - entries = resp.json() if not isinstance(entries, list): return [] - skills: List[SkillMeta] = [] groupings = self._get_skillsh_groupings(repo) prefix = path.rstrip("/") for entry in entries: - if entry.get("type") != "dir": + if entry.get("type") != "dir" or entry["name"].startswith((".", "_")): continue dir_name = entry["name"] - if dir_name.startswith((".", "_")): - continue meta = self.inspect(f"{repo}/{prefix}/{dir_name}" if prefix else f"{repo}/{dir_name}") if meta: - if groupings: - category = groupings.get(meta.name) or groupings.get(dir_name) - if category: - meta.extra["category"] = category + category = groupings and (groupings.get(meta.name) or groupings.get(dir_name)) + if category: + meta.extra["category"] = category skills.append(meta) - _cache_metas(cache_key, skills) return skills def _get_repo_tree(self, repo: str) -> Optional[Tuple[str, List[dict]]]: - """Cached ``(default_branch, tree_entries)`` for a repo, or None. - - One install may need the tree several times; caching saves the - ``GET /repos/{repo}`` + ``GET .../git/trees/{branch}`` pair each time - (~12 of the 60/hr unauthenticated budget before). - """ + """Cached ``(default_branch, tree_entries)`` for a repo, or None. One install may need the tree + several times; caching saves the ``GET /repos/{repo}`` + ``GET .../git/trees/{branch}`` pair each + time (~12 of the 60/hr unauthenticated budget before).""" if repo in self._tree_cache: return self._tree_cache[repo] - - repo_data = self._github_json(f"https://api.github.com/repos/{repo}") + repo_data = self._github_json(f"{_API}/{repo}") if repo_data is None: return None default_branch = repo_data.get("default_branch", "main") tree_data = self._github_json( - f"https://api.github.com/repos/{repo}/git/trees/{default_branch}", - params={"recursive": "1"}, timeout=30.0, + f"{_API}/{repo}/git/trees/{default_branch}", params={"recursive": "1"}, timeout=30.0, ) if tree_data is None: return None if tree_data.get("truncated"): logger.debug("Git tree truncated for %s, cannot cache", repo) return None - - entries = tree_data.get("tree", []) - revision = tree_data.get("sha") - if isinstance(revision, str) and revision: - self._tree_revisions[repo] = revision - self._tree_cache[repo] = (default_branch, entries) - return (default_branch, entries) + if isinstance(tree_data.get("sha"), str) and tree_data["sha"]: + self._tree_revisions[repo] = tree_data["sha"] + self._tree_cache[repo] = tree = (default_branch, tree_data.get("tree", [])) + return tree def _github_json(self, url: str, **kwargs) -> Optional[dict]: """Decoded JSON body of a 200 ``_github_get`` (which flags rate-limit exhaustion), else None.""" resp = self._github_get(url, **kwargs) - if resp is None or resp.status_code != 200: - return None try: - return resp.json() + return resp.json() if resp is not None and resp.status_code == 200 else None except ValueError: return None - def _check_rate_limit_response(self, resp: httpx.Response) -> None: - """Flag the instance as rate-limited when GitHub returns 403 + exhausted quota.""" - if _is_rate_limit_response(resp): - self._rate_limited = True - logger.warning( - "GitHub API rate limit exhausted (unauthenticated: 60 req/hr). " - "Set GITHUB_TOKEN or install the gh CLI to raise the limit to 5,000/hr." - ) - def _github_get( - self, - url: str, - *, - params: Optional[Dict] = None, - headers: Optional[Dict] = None, - timeout: float = 15.0, - max_retries: int = 3, + self, url: str, *, params: Optional[Dict] = None, headers: Optional[Dict] = None, + timeout: float = 15.0, max_retries: int = 3, ) -> Optional[httpx.Response]: - """GET against the GitHub API with retry/backoff on transient failures. - - Returns the final response (caller inspects status) or None when every - attempt raised a transport error. Retries rate-limit 403/429 (waiting - until ``Retry-After``/``X-RateLimit-Reset`` when present, capped 60s — - one shared limit zeroes every GitHub tap at once during an index build), - 5xx, and transport errors with exponential backoff. Terminal rate-limit - exhaustion flags the instance so an index build fails loud instead of - silently shipping zero GitHub skills. - """ + """GET against the GitHub API with retry/backoff on transient failures. Returns the final + response (caller inspects status) or None when every attempt raised a transport error. + Retries rate-limit 403/429 (waiting until ``Retry-After`` / ``X-RateLimit-Reset`` when present, + capped 60s — one shared limit zeroes every GitHub tap at once during an index build), 5xx, and + transport errors with exponential backoff. Terminal rate-limit exhaustion flags the instance so + an index build fails loud instead of silently shipping zero GitHub skills.""" hdrs = headers if headers is not None else self.auth.get_headers() backoff = 1.0 last_resp: Optional[httpx.Response] = None @@ -516,13 +386,9 @@ class GitHubSource(SkillSource): last_attempt = attempt >= max_retries - 1 wait = backoff try: - resp = httpx.get( - url, params=params, headers=hdrs, - timeout=timeout, follow_redirects=True, - ) + resp = httpx.get(url, params=params, headers=hdrs, timeout=timeout, follow_redirects=True) except httpx.HTTPError as e: - logger.debug("GitHub GET %s failed (attempt %d/%d): %s", - url, attempt + 1, max_retries, e) + logger.debug("GitHub GET %s failed (attempt %d/%d): %s", url, attempt + 1, max_retries, e) if last_attempt: return None else: @@ -530,8 +396,12 @@ class GitHubSource(SkillSource): if resp.status_code == 200: return resp if resp.status_code in (403, 429): - if not _is_rate_limit_response(resp) or last_attempt: - self._check_rate_limit_response(resp) + limited = _is_rate_limit_response(resp) + if not limited or last_attempt: + if limited: # terminal exhaustion: flag the instance so callers fail loud + self._rate_limited = True + logger.warning("GitHub API rate limit exhausted (unauthenticated: 60 req/hr). " + "Set GITHUB_TOKEN or install the gh CLI to raise the limit to 5,000/hr.") return resp reset = resp.headers.get("X-RateLimit-Reset", "") retry_after = resp.headers.get("Retry-After", "") @@ -541,35 +411,23 @@ class GitHubSource(SkillSource): delta = float(reset) - time.time() if 0 < delta <= 60.0: wait = delta - logger.debug( - "GitHub rate limited on %s, waiting %.1fs (attempt %d/%d)", - url, wait, attempt + 1, max_retries, - ) + logger.debug("GitHub rate limited on %s, waiting %.1fs (attempt %d/%d)", + url, wait, attempt + 1, max_retries) elif not (500 <= resp.status_code < 600) or last_attempt: return resp time.sleep(wait) backoff = min(backoff * 2, 30.0) - return last_resp def _find_skill_in_repo_tree(self, repo: str, skill_name: str) -> Optional[str]: - """Locate ``/SKILL.md`` anywhere in the repo tree (one API call). - - Returns the full identifier (``repo/path/to/skill``) or None. - """ - cached = self._get_repo_tree(repo) - if cached is None: + """Locate ``/SKILL.md`` anywhere in the repo tree (one API call); full identifier or None.""" + if (cached := self._get_repo_tree(repo)) is None: return None - _default_branch, tree_entries = cached - skill_md_suffix = f"/{skill_name}/SKILL.md" - for entry in tree_entries: - if entry.get("type") != "blob": - continue + for entry in cached[1]: path = entry.get("path", "") - if path.endswith(skill_md_suffix) or path == f"{skill_name}/SKILL.md": + if entry.get("type") == "blob" and (path.endswith(skill_md_suffix) or path == skill_md_suffix[1:]): return f"{repo}/{path[: -len('/SKILL.md')]}" - return None def _fetch_file_content(self, repo: str, path: str, ref: Optional[str] = None) -> Optional[str]: @@ -583,32 +441,20 @@ class GitHubSource(SkillSource): def _fetch_file_bytes(self, repo: str, path: str, ref: Optional[str] = None) -> Optional[bytes]: """Fetch exact file bytes. ``ref`` pins to a tree SHA (see ``fetch`` on the TOCTOU); None keeps the legacy unpinned behavior.""" - encoded_path = quote(path, safe="/") - url = f"https://api.github.com/repos/{repo}/contents/{encoded_path}" resp = self._github_get( - url, - params={"ref": ref} if ref else None, + f"{_API}/{repo}/contents/{quote(path, safe='/')}", params={"ref": ref} if ref else None, headers={**self.auth.get_headers(), "Accept": "application/vnd.github.v3.raw"}, ) - if resp is not None and resp.status_code == 200: - return resp.content - return None + return resp.content if resp is not None and resp.status_code == 200 else None def _get_skillsh_groupings(self, repo: str) -> Optional[Dict[str, str]]: - """Repo-root ``skills.sh.json`` groupings flattened to ``{skill_name: title}``. - - ``skills.sh.json`` is a cross-ecosystem standard - (``$schema: https://skills.sh/schemas/skills.sh.schema.json``); any tap - shipping it gets category pills for free. None when absent/unparsable. - Cached per repo on the instance. - """ - if repo in self._skillsh_groupings: - return self._skillsh_groupings[repo] - - content = self._fetch_file_content(repo, "skills.sh.json") - groupings = self._parse_skillsh_groupings(content) if content else None - self._skillsh_groupings[repo] = groupings - return groupings + """Repo-root ``skills.sh.json`` groupings flattened to ``{skill_name: title}``. ``skills.sh.json`` + is a cross-ecosystem standard (``$schema: https://skills.sh/schemas/skills.sh.schema.json``); any + tap shipping it gets category pills for free. None when absent/unparsable; cached per repo.""" + if repo not in self._skillsh_groupings: + content = self._fetch_file_content(repo, "skills.sh.json") + self._skillsh_groupings[repo] = self._parse_skillsh_groupings(content) if content else None + return self._skillsh_groupings[repo] @staticmethod def _parse_skillsh_groupings(content: str) -> Optional[Dict[str, str]]: @@ -617,22 +463,17 @@ class GitHubSource(SkillSource): data = json.loads(content) except (json.JSONDecodeError, TypeError): return None - if not isinstance(data, dict): - return None - groupings = data.get("groupings") + groupings = data.get("groupings") if isinstance(data, dict) else None if not isinstance(groupings, list): return None - mapping: Dict[str, str] = {} for group in groupings: if not isinstance(group, dict): continue - title = group.get("title") - members = group.get("skills") + title, members = group.get("title"), group.get("skills") if not isinstance(title, str) or not isinstance(members, list): continue for member in members: if isinstance(member, str) and member: mapping.setdefault(member, title) # first grouping wins return mapping - diff --git a/tools/skills_hub_install.py b/tools/skills_hub_install.py index 351788a6f7..70b62fbb03 100644 --- a/tools/skills_hub_install.py +++ b/tools/skills_hub_install.py @@ -17,12 +17,8 @@ from agent.skill_utils import is_excluded_skill_path from tools.skills_guard import ScanResult, content_hash from tools.skills_hub_github import GitHubAuth from tools.skills_hub_models import ( - SkillBundle, - SkillSource, - _normalize_lock_install_path, - _validate_bundle_rel_path, - _validate_install_parent_path, - _validate_skill_name, + SkillBundle, SkillSource, _normalize_lock_install_path, _validate_bundle_rel_path, + _validate_install_parent_path, _validate_skill_name, ) if TYPE_CHECKING: # origin class; runtime use is via the lazy origin import @@ -49,15 +45,12 @@ def _resolve_lock_install_path(install_path: str, skill_name: str) -> Path: """ from tools.skills_hub import _skills_dir normalized = _normalize_lock_install_path(install_path, skill_name) - skills_dir = _skills_dir() + target = skills_dir = _skills_dir() skills_root = skills_dir.resolve() - - target = skills_dir for part in normalized.split("/"): target = target / part if _is_path_redirect(target): raise ValueError(f"Unsafe install path: {install_path}") - target = target.resolve() if target == skills_root or not target.is_relative_to(skills_root): raise ValueError(f"Unsafe install path: {install_path}") @@ -70,16 +63,11 @@ def quarantine_bundle(bundle: SkillBundle) -> Path: ensure_hub_dirs() skill_name = _validate_skill_name(bundle.name) # Validate every path before touching disk so a bad member aborts cleanly. - validated_files = [ - (_validate_bundle_rel_path(rel_path), file_content) - for rel_path, file_content in bundle.files.items() - ] - + validated_files = [(_validate_bundle_rel_path(rel_path), content) for rel_path, content in bundle.files.items()] dest = _quarantine_dir() / skill_name if dest.exists(): shutil.rmtree(dest) dest.mkdir(parents=True) - for rel_path, file_content in validated_files: file_dest = dest.joinpath(*rel_path.split("/")) file_dest.parent.mkdir(parents=True, exist_ok=True) @@ -87,7 +75,6 @@ def quarantine_bundle(bundle: SkillBundle) -> Path: file_dest.write_bytes(file_content) else: file_dest.write_text(file_content, encoding="utf-8") - return dest @@ -100,16 +87,13 @@ def _category_skill_dirs(directory: Path) -> List[str]: ``references/pkg/SKILL.md`` does not make the directory a category. Shared with ``hermes_cli.skills_hub._existing_categories``. """ - skill_dirs: List[str] = [] - for entry in directory.iterdir(): - if not entry.is_dir() or entry.name.startswith("."): - continue - if any( + return [ + entry.name for entry in directory.iterdir() + if entry.is_dir() and not entry.name.startswith(".") and any( not is_excluded_skill_path(skill_md.relative_to(directory), root=directory) for skill_md in entry.rglob("SKILL.md") - ): - skill_dirs.append(entry.name) - return skill_dirs + ) + ] def _check_install_target(install_dir: Path) -> None: @@ -127,38 +111,24 @@ def _check_install_target(install_dir: Path) -> None: ancestor = install_dir.parent while ancestor != skills_root and ancestor.is_relative_to(skills_root): if (ancestor / "SKILL.md").is_file(): - raise ValueError( - f"Refusing to install into '{ancestor.name}': it is an " - f"existing skill directory, not a category. Choose a " - f"different category." - ) + raise ValueError(f"Refusing to install into '{ancestor.name}': it is an " + f"existing skill directory, not a category. Choose a different category.") ancestor = ancestor.parent - if not install_dir.exists(): return if not install_dir.is_dir(): - raise ValueError( - f"Refusing to install: '{install_dir.name}' already exists " - f"and is not a directory. Remove it or choose a different " - f"skill name." - ) + raise ValueError(f"Refusing to install: '{install_dir.name}' already exists " + f"and is not a directory. Remove it or choose a different skill name.") if not (install_dir / "SKILL.md").exists(): skill_dirs_in = _category_skill_dirs(install_dir) if skill_dirs_in: - raise ValueError( - f"Refusing to overwrite category directory '{install_dir}' " - f"which contains {len(skill_dirs_in)} skill(s): " - f"{', '.join(sorted(skill_dirs_in))}. " - f"Use a different --name or install into a subcategory." - ) + raise ValueError(f"Refusing to overwrite category directory '{install_dir}' " + f"which contains {len(skill_dirs_in)} skill(s): {', '.join(sorted(skill_dirs_in))}. " + f"Use a different --name or install into a subcategory.") def install_from_quarantine( - quarantine_path: Path, - skill_name: str, - category: str, - bundle: SkillBundle, - scan_result: ScanResult, + quarantine_path: Path, skill_name: str, category: str, bundle: SkillBundle, scan_result: ScanResult, scan_provenance: Optional[Dict[str, Any]] = None, ) -> Path: """Move a scanned skill from quarantine into the skills directory.""" @@ -186,8 +156,7 @@ def install_from_quarantine( "Skill '%s' has a large SKILL.md (%s chars). " "Large skills consume significant context when loaded. " "Consider asking the author to split it into smaller files.", - safe_skill_name, - f"{skill_size:,}", + safe_skill_name, f"{skill_size:,}", ) # A symlink in the bundle would copy its target into skills/ and leak it @@ -202,33 +171,21 @@ def install_from_quarantine( install_dir.parent.mkdir(parents=True, exist_ok=True) shutil.move(str(quarantine_path), str(install_dir)) - installed_hash = content_hash(install_dir) HubLockFile().record_install( - name=safe_skill_name, - source=bundle.source, - identifier=bundle.identifier, - trust_level=bundle.trust_level, - scan_verdict=scan_result.verdict, - skill_hash=installed_hash, + name=safe_skill_name, source=bundle.source, identifier=bundle.identifier, trust_level=bundle.trust_level, + scan_verdict=scan_result.verdict, skill_hash=installed_hash, install_path=install_dir.resolve().relative_to(_skills_dir().resolve()).as_posix(), - files=list(bundle.files.keys()), - metadata=bundle.metadata, + files=list(bundle.files.keys()), metadata=bundle.metadata, scan_provenance=scan_provenance or getattr(scan_result, "scan_provenance", None), ) - - append_audit_log( - "INSTALL", safe_skill_name, bundle.source, - bundle.trust_level, scan_result.verdict, - installed_hash, - ) - + append_audit_log("INSTALL", safe_skill_name, bundle.source, bundle.trust_level, scan_result.verdict, + installed_hash) try: from tools.skill_usage import record_installed record_installed(safe_skill_name) except Exception: logger.debug("Unable to record skill install lifecycle for %s", safe_skill_name, exc_info=True) - return install_dir @@ -239,20 +196,16 @@ def uninstall_skill(skill_name: str) -> Tuple[bool, str]: entry = lock.get_installed(skill_name) if not entry: return False, f"'{skill_name}' is not a hub-installed skill (may be a builtin)" - # The destructive boundary: whatever reaches rmtree MUST be inside # SKILLS_DIR and MUST NOT be SKILLS_DIR itself (see _resolve_lock_install_path). try: install_path = _resolve_lock_install_path(entry.get("install_path", ""), skill_name) except ValueError as exc: return False, f"Refusing to uninstall '{skill_name}': {exc}" - if install_path.exists(): shutil.rmtree(install_path) - lock.record_uninstall(skill_name) append_audit_log("UNINSTALL", skill_name, entry["source"], entry["trust_level"], "n/a", "user_request") - return True, f"Uninstalled '{skill_name}' from {entry['install_path']}" @@ -266,10 +219,7 @@ def bundle_content_hash(bundle: SkillBundle) -> str: path is hashed too so swapping contents between two files changes the hash. """ h = hashlib.sha256() - normalized = { - rel_path.replace("\\", "/"): content - for rel_path, content in bundle.files.items() - } + normalized = {rel_path.replace("\\", "/"): content for rel_path, content in bundle.files.items()} for rel_path in sorted(normalized): h.update(rel_path.encode("utf-8")) h.update(b"\x00") @@ -286,11 +236,8 @@ def _source_matches(source: SkillSource, source_name: str) -> bool: def check_for_skill_updates( - name: Optional[str] = None, - *, - lock: Optional[HubLockFile] = None, - sources: Optional[List[SkillSource]] = None, - auth: Optional[GitHubAuth] = None, + name: Optional[str] = None, *, lock: Optional[HubLockFile] = None, + sources: Optional[List[SkillSource]] = None, auth: Optional[GitHubAuth] = None, ) -> List[dict]: """Check installed hub skills for upstream changes. @@ -304,16 +251,13 @@ def check_for_skill_updates( installed = lock.list_installed() if name: installed = [entry for entry in installed if entry.get("name") == name] - if sources is None: sources = create_source_router(auth=auth) results: List[dict] = [] for entry in installed: - identifier = entry.get("identifier", "") - source_name = entry.get("source", "") + identifier, source_name = entry.get("identifier", ""), entry.get("source", "") row = {"name": entry.get("name", ""), "identifier": identifier, "source": source_name} - bundle = None for src in filter(lambda s: _source_matches(s, source_name), sources): try: @@ -322,19 +266,12 @@ def check_for_skill_updates( bundle = None if bundle: break - if not bundle: results.append({**row, "status": "unavailable"}) continue - - current_hash = entry.get("content_hash", "") - latest_hash = bundle_content_hash(bundle) + current_hash, latest_hash = entry.get("content_hash", ""), bundle_content_hash(bundle) results.append({ - **row, - "status": "up_to_date" if current_hash == latest_hash else "update_available", - "current_hash": current_hash, - "latest_hash": latest_hash, - "bundle": bundle, + **row, "status": "up_to_date" if current_hash == latest_hash else "update_available", + "current_hash": current_hash, "latest_hash": latest_hash, "bundle": bundle, }) - return results diff --git a/tools/skills_hub_models.py b/tools/skills_hub_models.py index 59bad0fc44..24eb07bed2 100644 --- a/tools/skills_hub_models.py +++ b/tools/skills_hub_models.py @@ -1,8 +1,5 @@ -"""Skills Hub data models, path validators, and SKILL.md helpers. - -Leaf module (no imports from tools.skills_hub) so every source adapter module -can import it at top level without cycles. -""" +"""Skills Hub data models, path validators, and SKILL.md helpers. Leaf module (no imports from +tools.skills_hub) so every source adapter module can import it at top level without cycles.""" import json import logging @@ -20,12 +17,8 @@ logger = logging.getLogger("tools.skills_hub") def hub(): - """``tools.skills_hub`` resolved at call time. - - Its cache / HTTP / index helpers are the test-patch targets - (``patch("tools.skills_hub._read_index_cache")`` ...), so adapters look - them up through the module on every call instead of binding at import. - """ + """``tools.skills_hub`` resolved at call time: its cache / HTTP / index helpers are the test-patch targets + (``patch("tools.skills_hub._read_index_cache")`` ...), so adapters look them up on every call, not at import.""" import tools.skills_hub as mod return mod @@ -56,7 +49,6 @@ class SkillBundle: def _skill_meta_to_dict(meta: SkillMeta) -> dict: - """Convert a SkillMeta to a dict for caching.""" return dict(vars(meta)) @@ -71,8 +63,8 @@ def _cache_metas(key: str, metas: List[SkillMeta]) -> None: def _memo_json(key: str, compute: Callable[[], Any], valid: Callable[[Any], bool] = lambda c: c is not None) -> Any: - """Shared-index-cache memo: a cached value passing ``valid`` is returned as-is; - otherwise ``compute()`` runs and a non-None result is written back.""" + """Shared-index-cache memo: a cached value passing ``valid`` is returned as-is; otherwise ``compute()`` + runs and a non-None result is written back.""" cached = hub()._read_index_cache(key) if valid(cached): return cached @@ -86,9 +78,7 @@ def _get_json(url: str, *, timeout: int = 20, **kwargs) -> Optional[Any]: """Plain (unguarded) GET + JSON decode; None on non-200 or transport/decode error.""" try: resp = httpx.get(url, timeout=timeout, **kwargs) - if resp.status_code != 200: - return None - return resp.json() + return resp.json() if resp.status_code == 200 else None except (httpx.HTTPError, json.JSONDecodeError): return None @@ -103,8 +93,7 @@ def _get_text(url: str, *, timeout: int = 20, **kwargs) -> Optional[str]: def _matches_query(query_lower: str, *fields: Any) -> bool: - """Case-insensitive substring match of ``query_lower`` against joined fields - (lists are space-joined; an empty query matches everything).""" + """Case-insensitive substring match against joined fields (lists space-joined; empty query matches all).""" parts = [" ".join(str(t) for t in f) if isinstance(f, list) else str(f) for f in fields] return query_lower in " ".join(parts).lower() @@ -114,10 +103,8 @@ def _first_matching(query_lower: str, items: Iterable[Any], fields_of: Callable[ """Substring-search ``items`` in order, converting hits with ``to_meta`` until ``limit``.""" results: List[SkillMeta] = [] for item in items: - if _matches_query(query_lower, *fields_of(item)): - meta = to_meta(item) - if meta: - results.append(meta) + if _matches_query(query_lower, *fields_of(item)) and (meta := to_meta(item)): + results.append(meta) if len(results) >= limit: break return results @@ -127,11 +114,8 @@ TRUST_RANK = {"builtin": 2, "trusted": 1, "community": 0} def _dedupe_by_trust(results: Iterable[SkillMeta]) -> List[SkillMeta]: - """Dedupe by identifier, keeping the higher-trust copy (first wins on ties). - - identifier is unique per skill; name is not — two taps can publish - same-named skills, and browse-sh reuses task names across sites. - """ + """Dedupe by identifier, keeping the higher-trust copy (first wins on ties). identifier is unique per + skill; name is not — two taps can publish same-named skills, and browse-sh reuses task names across sites.""" seen: Dict[str, SkillMeta] = {} for r in results: kept = seen.get(r.identifier) @@ -141,11 +125,8 @@ def _dedupe_by_trust(results: Iterable[SkillMeta]) -> List[SkillMeta]: class SkillSource(ABC): - """Abstract base for all skill registry adapters. - - ``SOURCE_ID`` is the unique source id (e.g. 'github', 'clawhub'); ``TRUST_LEVEL`` - the trust every identifier gets unless ``trust_level_for`` is overridden. - """ + """Abstract base for all skill registry adapters. ``SOURCE_ID`` is the unique source id (e.g. 'github', + 'clawhub'); ``TRUST_LEVEL`` the trust every identifier gets unless ``trust_level_for`` is overridden.""" SOURCE_ID: str = "" TRUST_LEVEL: str = "community" @@ -183,21 +164,15 @@ class GuardedFetchMixin: return resp.content if resp is not None and resp.status_code == 200 else None -# --------------------------------------------------------------------------- -# SKILL.md frontmatter -# --------------------------------------------------------------------------- - +# --- SKILL.md frontmatter --------------------------------------------------- def _parse_frontmatter(content: str) -> dict: """Parse YAML frontmatter from SKILL.md content ({} when absent/invalid).""" content = content.lstrip("\ufeff") # tolerate UTF-8 BOM (Windows editors) - if not content.startswith("---"): - return {} - match = re.search(r'\n---\s*\n', content[3:]) + match = re.search(r'\n---\s*\n', content[3:]) if content.startswith("---") else None if not match: return {} - yaml_text = content[3:match.start() + 3] try: - parsed = yaml.safe_load(yaml_text) + parsed = yaml.safe_load(content[3:match.start() + 3]) return parsed if isinstance(parsed, dict) else {} except yaml.YAMLError: return {} @@ -206,11 +181,8 @@ def _parse_frontmatter(content: str) -> dict: def _hermes_tags(fm: dict) -> Any: """``metadata.hermes.tags`` from parsed frontmatter, or ``[]`` (unvalidated type).""" metadata = fm.get("metadata", {}) - if isinstance(metadata, dict): - hermes_meta = metadata.get("hermes", {}) - if isinstance(hermes_meta, dict): - return hermes_meta.get("tags", []) - return [] + hermes_meta = metadata.get("hermes", {}) if isinstance(metadata, dict) else None + return hermes_meta.get("tags", []) if isinstance(hermes_meta, dict) else [] def source_url_for_bundle(bundle: SkillBundle) -> str: @@ -226,34 +198,22 @@ def source_url_for_bundle(bundle: SkillBundle) -> str: return bundle.identifier -# --------------------------------------------------------------------------- -# Bundle path validation -# --------------------------------------------------------------------------- - +# --- Bundle path validation ------------------------------------------------- def _normalize_bundle_path(path_value: str, *, field_name: str, allow_nested: bool) -> str: """Normalize and validate bundle-controlled paths before touching disk.""" if not isinstance(path_value, str): raise ValueError(f"Unsafe {field_name}: expected a string") - raw = path_value.strip() if not raw: raise ValueError(f"Unsafe {field_name}: empty path") - normalized = raw.replace("\\", "/") path = PurePosixPath(normalized) parts = [part for part in path.parts if part not in {"", "."}] - - # A colon in any component is rejected: on Windows it marks a drive - # (``C:foo``) or an NTFS Alternate Data Stream (``file.py:payload`` writes - # scanner-invisible bytes); ``/`` is the only legal separator once normalized. - if ( - normalized.startswith("/") or path.is_absolute() - or not parts or any(part == ".." for part in parts) - or any(":" in part for part in parts) - or (not allow_nested and len(parts) != 1) - ): + # A colon in any component is rejected: on Windows it marks a drive (``C:foo``) or an NTFS Alternate Data + # Stream (``file.py:payload`` writes scanner-invisible bytes); ``/`` is the only legal separator once normalized. + if (normalized.startswith("/") or path.is_absolute() or not parts or any(part == ".." for part in parts) + or any(":" in part for part in parts) or (not allow_nested and len(parts) != 1)): raise ValueError(f"Unsafe {field_name}: {path_value}") - return "/".join(parts) @@ -272,10 +232,9 @@ def _validate_bundle_rel_path(rel_path: str) -> str: def _normalize_lock_install_path(install_path: str, skill_name: str) -> str: """Validate a lock-file ``install_path`` (the ``uninstall_skill`` rmtree target). - Must be relative, traversal-free, and end with ```` — nested - official skills legitimately live at ``mlops/training/``; an - empty/``"."``/absolute/mismatched entry could point rmtree at the whole - ``skills/`` tree or outside it. + Must be relative, traversal-free, and end with ```` — nested official skills legitimately + live at ``mlops/training/``; an empty/``"."``/absolute/mismatched entry could point rmtree + at the whole ``skills/`` tree or outside it. """ safe_skill_name = _validate_skill_name(skill_name) normalized = _normalize_bundle_path(install_path, field_name="install path", allow_nested=True) @@ -284,23 +243,16 @@ def _normalize_lock_install_path(install_path: str, skill_name: str) -> str: return normalized -# --------------------------------------------------------------------------- -# Referenced support-file extraction from SKILL.md -# --------------------------------------------------------------------------- - +# --- Referenced support-file extraction from SKILL.md ----------------------- _ALLOWED_SUPPORT_DIRS = frozenset({"references", "templates", "scripts", "assets", "examples"}) _LOCAL_LINK_RE = re.compile( - r"(?:\]\(|`|(?:^|[\s\"']))((?:references|templates|scripts|assets|examples)/[^\s)`\"'<>]+)", - re.MULTILINE, -) + r"(?:\]\(|`|(?:^|[\s\"']))((?:references|templates|scripts|assets|examples)/[^\s)`\"'<>]+)", re.MULTILINE) _SUSPICIOUS_LOCAL_REF_RE = re.compile( - r"(?:references|templates|scripts|assets|examples)/(?:[^\s)`\"'<>]*/)?\.\.(?:/|$)" -) + r"(?:references|templates|scripts|assets|examples)/(?:[^\s)`\"'<>]*/)?\.\.(?:/|$)") _VALUELESS_QUERY_FLAG_RE = re.compile(r"(?:[A-Za-z0-9_~-]|%[0-9A-Fa-f]{2})+\Z") -# Same-directory links (``](./FILE.ext)`` / ``](FILE.ext)``): siblings of SKILL.md -# the document links explicitly (e.g. ./CONTEXT-FORMAT.md). Dropping them made the -# install "succeed" with unresolved links. The extension requirement keeps prose -# words out; support-dir links stay on _LOCAL_LINK_RE. +# Same-directory links (``](./FILE.ext)`` / ``](FILE.ext)``): siblings of SKILL.md the document links +# explicitly (e.g. ./CONTEXT-FORMAT.md). Dropping them made the install "succeed" with unresolved links. +# The extension requirement keeps prose words out; support-dir links stay on _LOCAL_LINK_RE. _SAMEDIR_LINK_RE = re.compile(r"\]\(([^)\s\"'<>]+)") _SAMEDIR_NAME_RE = re.compile(r"^(?:\./)?[A-Za-z0-9][A-Za-z0-9._-]*$") @@ -308,16 +260,12 @@ _SAMEDIR_NAME_RE = re.compile(r"^(?:\./)?[A-Za-z0-9][A-Za-z0-9._-]*$") def _query_is_concrete(query: str) -> bool: """Whether a URL query is real URL syntax rather than glob prose. - A non-empty ``key=value`` part is always concrete. Valueless flags are - accepted only when RFC 3986 unreserved-token shaped (percent escapes ok); - ``.``, brackets and extra ``?`` are excluded because ``?x.md`` / ``?.md`` - is indistinguishable from a single-char glob finishing a filename in prose. + A non-empty ``key=value`` part is always concrete. Valueless flags are accepted only when RFC 3986 + unreserved-token shaped (percent escapes ok); ``.``, brackets and extra ``?`` are excluded because + ``?x.md`` / ``?.md`` is indistinguishable from a single-char glob finishing a filename in prose. """ - return all( - ("=" in part and bool(part.split("=", 1)[0])) - or bool(_VALUELESS_QUERY_FLAG_RE.fullmatch(part)) - for part in query.split("&") - ) + return all(("=" in part and bool(part.split("=", 1)[0])) or bool(_VALUELESS_QUERY_FLAG_RE.fullmatch(part)) + for part in query.split("&")) def _referenced_support_paths(skill_md: str) -> Optional[set[str]]: @@ -328,45 +276,37 @@ def _referenced_support_paths(skill_md: str) -> Optional[set[str]]: paths: set[str] = set() for match in _LOCAL_LINK_RE.finditer(normalized): candidate = match.group(1).rstrip(".,;:") - if candidate.endswith("?"): - continue parsed = urlsplit(candidate) raw = unquote(parsed.path) - if any(char in raw for char in "*?[]"): - continue - if parsed.query and not _query_is_concrete(parsed.query): + if (candidate.endswith("?") or any(char in raw for char in "*?[]") + or (parsed.query and not _query_is_concrete(parsed.query))): continue try: safe = _validate_bundle_rel_path(raw) except ValueError: return None if safe.split("/", 1)[0] in _ALLOWED_SUPPORT_DIRS: - # Prose placeholders (``references/type-.md``, truncated at - # ``<`` to ``references/type-``) are instructions, not files: a - # basename ending in a separator is skipped. No extension - # requirement — ``references/LICENSE`` is legitimate. + # Prose placeholders (``references/type-.md``, truncated at ``<`` to + # ``references/type-``) are instructions, not files: a basename ending in a + # separator is skipped. No extension requirement — ``references/LICENSE`` is legitimate. base = safe.rsplit("/", 1)[-1] if re.search(r"[*?<>]", safe) or not re.search(r"[A-Za-z0-9]$", base): continue paths.add(safe) for match in _SAMEDIR_LINK_RE.finditer(normalized): raw = match.group(1).rstrip(".,;:") - # Canonicalize like the support-dir branch (drop query/fragment, - # percent-decode), then strip a leading ``./``. + # Canonicalize like the support-dir branch (drop query/fragment, percent-decode), strip leading ``./``. name = unquote(urlsplit(raw).path) name = name[2:] if name.startswith("./") else name - # External URLs, anchors, mailto and site-absolute targets are not - # same-directory file links. + # External URLs, anchors, mailto and site-absolute targets are not same-directory file links. if not name or "://" in raw or raw.startswith(("mailto:", "#", "/")): continue if name.startswith(".."): return None - # Only unambiguous file links: an extension, no internal slash, never - # SKILL.md itself (casefolded — a ``skill.md`` entry would collide with - # the bundle root on macOS/Windows; skipped, not merged). - if "/" in name or name.casefold() == "skill.md" or "." not in name.lstrip("."): - continue - if not _SAMEDIR_NAME_RE.match(name): + # Only unambiguous file links: an extension, no internal slash, never SKILL.md itself (casefolded — + # a ``skill.md`` entry would collide with the bundle root on macOS/Windows; skipped, not merged). + if ("/" in name or name.casefold() == "skill.md" or "." not in name.lstrip(".") + or not _SAMEDIR_NAME_RE.match(name)): continue try: safe = _validate_bundle_rel_path(name) @@ -375,12 +315,10 @@ def _referenced_support_paths(skill_md: str) -> Optional[set[str]]: paths.add(safe) # Case-folded collisions among accepted same-dir names (``A.md`` + ``a.md``) # would collide on install — drop the pair rather than guess. - folded: dict[str, str] = {} + folded: dict[str, list[str]] = {} for p in sorted(paths): - key = p.casefold() - if key in folded: - paths.discard(folded[key]) - paths.discard(p) - else: - folded[key] = p + folded.setdefault(p.casefold(), []).append(p) + for group in folded.values(): + if len(group) > 1: + paths.difference_update(group) return paths diff --git a/tools/skills_hub_official.py b/tools/skills_hub_official.py index e90993b4d6..55ac5a1577 100644 --- a/tools/skills_hub_official.py +++ b/tools/skills_hub_official.py @@ -17,31 +17,19 @@ _INDEX_ID_PREFIXES = ("skills-sh/", "skills.sh/", "official/", "github/", "clawh def _strip_prefix(value: str, prefixes) -> str: - for prefix in prefixes: - if value.startswith(prefix): - return value[len(prefix):] - return value + return next((value[len(p):] for p in prefixes if value.startswith(p)), value) def _clean_rel_parts(path: str) -> Optional[List[str]]: """Split a relative path, dropping ``.``/empty parts; None on traversal or empty.""" parts = [p for p in path.split("/") if p not in ("", ".")] - if not parts or any(p == ".." for p in parts): - return None - return parts + return None if not parts or ".." in parts else parts -# --------------------------------------------------------------------------- -# Official optional skills source adapter -# --------------------------------------------------------------------------- - class OptionalSkillSource(SkillSource): - """Skills from the repo's ``optional-skills/`` directory. - - Official (Nous-maintained) but not activated by default — absent from the - system prompt and not copied to ~/.hermes/skills/ at setup. Discoverable - via the Skills Hub as source "official" with "builtin" trust. - """ + """Skills from the repo's ``optional-skills/`` directory: official (Nous-maintained) but not + activated by default — absent from the system prompt and not copied to ~/.hermes/skills/ at + setup. Discoverable via the Skills Hub as source "official" with "builtin" trust.""" SOURCE_ID = "official" TRUST_LEVEL = "builtin" @@ -53,12 +41,9 @@ class OptionalSkillSource(SkillSource): def __init__(self, auth: Optional[GitHubAuth] = None): from hermes_constants import get_optional_skills_dir - self._optional_dir = get_optional_skills_dir( - Path(__file__).parent.parent / "optional-skills" - ) + self._optional_dir = get_optional_skills_dir(Path(__file__).parent.parent / "optional-skills") self._auth = auth - # GitHubSource for the live-repo fallback, created only when a skill is - # missing from the local checkout. + # GitHubSource for the live-repo fallback, created only when a skill is missing locally. self._github: Optional[GitHubSource] = None # "category/skill" -> True from the live repo tree; None = not fetched yet. self._remote_dirs: Optional[Dict[str, bool]] = None @@ -69,37 +54,25 @@ class OptionalSkillSource(SkillSource): def _meta(self, rel_dir: str, name: str, description: str, tags: list) -> SkillMeta: return SkillMeta( - name=name, - description=description, - source="official", - identifier=f"official/{rel_dir}", - trust_level="builtin", - repo=self.OFFICIAL_REPO, + name=name, description=description, source="official", identifier=f"official/{rel_dir}", + trust_level="builtin", repo=self.OFFICIAL_REPO, # The centralized skills index consumes repo-root-relative paths. - path=f"{self.OPTIONAL_SKILLS_PREFIX}/{rel_dir}", - tags=tags, + path=f"{self.OPTIONAL_SKILLS_PREFIX}/{rel_dir}", tags=tags, ) def _remote_meta(self, rel_dir: str) -> SkillMeta: """Placeholder meta for a skill that exists on live main but not locally.""" - return self._meta( - rel_dir, rel_dir.rsplit("/", 1)[-1], - "Official optional skill (from live repo; run install to fetch)", [], - ) + desc = "Official optional skill (from live repo; run install to fetch)" + return self._meta(rel_dir, rel_dir.rsplit("/", 1)[-1], desc, []) @staticmethod def _bundle(rel_id: str, files: Dict[str, Union[str, bytes]], **kwargs) -> SkillBundle: - return SkillBundle( - name=rel_id.rsplit("/", 1)[-1], files=files, source="official", - identifier=f"official/{rel_id}", trust_level="builtin", **kwargs, - ) - - # -- search ----------------------------------------------------------- + return SkillBundle(name=rel_id.rsplit("/", 1)[-1], files=files, source="official", + identifier=f"official/{rel_id}", trust_level="builtin", **kwargs) def search(self, query: str, limit: int = 10) -> List[SkillMeta]: results: List[SkillMeta] = [] query_lower = query.lower() - local_rels: set = set() for meta in self._scan_all(): local_rels.add(meta.identifier.split("/", 1)[-1] if meta.identifier else "") @@ -107,7 +80,6 @@ class OptionalSkillSource(SkillSource): results.append(meta) if len(results) >= limit: break - # Also surface skills that landed on live main after this install was cut. if len(results) < limit: for rel_dir in sorted(self._list_remote_skill_dirs()): @@ -116,42 +88,36 @@ class OptionalSkillSource(SkillSource): results.append(self._remote_meta(rel_dir)) if len(results) >= limit: break - return results - # -- fetch ------------------------------------------------------------ - def fetch(self, identifier: str) -> Optional[SkillBundle]: # identifier format: "official/category/skill" or "official/skill" rel = self._rel(identifier) - skill_dir = self._optional_dir / rel - # Guard against path traversal (e.g. "official/../../etc") try: - resolved = skill_dir.resolve() + resolved = (self._optional_dir / rel).resolve() optional_root = self._optional_dir.resolve() if not resolved.is_relative_to(optional_root): return None except (OSError, ValueError): return None - if resolved.is_dir(): - skill_dir = resolved - else: - # Try by bare skill name; if still absent, the skill may have - # landed on main after this install was cut — use the live repo. - skill_dir = self._find_skill_dir(rel.rsplit("/", 1)[-1]) - if not skill_dir: - return self._fetch_from_live_repo(rel) - + # Else try by bare skill name; if still absent, the skill may have landed on main + # after this install was cut — use the live repo. + skill_dir = resolved if resolved.is_dir() else self._find_skill_dir(rel.rsplit("/", 1)[-1]) + if not skill_dir: + return self._fetch_from_live_repo(rel) rel_id = skill_dir.resolve().relative_to(optional_root).as_posix() # Catalog stubs point at the real skill in an upstream-maintained repo # (metadata.hermes.upstream); install pulls the live content from there. - upstream = self._upstream_pointer(skill_dir) + try: + skill_md = (skill_dir / "SKILL.md").read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError): + skill_md = None + upstream = None if skill_md is None else self._upstream_pointer_from_content(skill_md) if upstream is not None: return self._fetch_from_upstream(upstream, rel_id) - files: Dict[str, Union[str, bytes]] = {} for f in skill_dir.rglob("*"): if f.is_file() and not _skip_bundle_file(f.relative_to(skill_dir).as_posix()): @@ -159,33 +125,21 @@ class OptionalSkillSource(SkillSource): files[str(f.relative_to(skill_dir))] = f.read_bytes() except OSError: continue - return self._bundle(rel_id, files) if files else None - # -- inspect ---------------------------------------------------------- - def inspect(self, identifier: str) -> Optional[SkillMeta]: skill_name = self._rel(identifier).rsplit("/", 1)[-1] - for meta in self._scan_all(): if meta.name == skill_name: return meta - - # Not in the local checkout — check live main. - matches = self._remote_matches(skill_name) - if len(matches) == 1: - return self._remote_meta(matches[0]) - return None - - # -- catalog ---------------------------------------------------------- + matches = self._remote_matches(skill_name) # not in the local checkout — check live main + return self._remote_meta(matches[0]) if len(matches) == 1 else None def list_local(self) -> List[SkillMeta]: """Every optional skill in the local checkout, with frontmatter metadata (backs the dashboard/desktop "built-in optional skills" catalog).""" return self._scan_all() - # -- internal helpers ------------------------------------------------- - def _get_github(self) -> GitHubSource: if self._github is None: self._github = GitHubSource(auth=self._auth or GitHubAuth()) @@ -195,17 +149,13 @@ class OptionalSkillSource(SkillSource): return [d for d in self._list_remote_skill_dirs() if d.rsplit("/", 1)[-1] == name] def _fetch_from_live_repo(self, rel: str) -> Optional[SkillBundle]: - """Fetch an optional skill straight from the live default branch. - - Local installs lag ``main``; rather than demanding ``hermes update`` - first, resolve against the live repo. ``rel`` is ``category/skill`` - (used verbatim) or a bare skill name (located via the repo tree). - """ + """Fetch an optional skill straight from the live default branch. Local installs lag + ``main``; rather than demanding ``hermes update`` first, resolve against the live repo. + ``rel`` is ``category/skill`` (used verbatim) or a bare skill name (located via the repo tree).""" parts = _clean_rel_parts(rel.strip("/")) if parts is None: return None rel = "/".join(parts) - github = self._get_github() if rel not in self._list_remote_skill_dirs(): # Bare name (or stale category) — locate by final path segment. @@ -213,16 +163,14 @@ class OptionalSkillSource(SkillSource): if len(matches) != 1: return None rel = matches[0] - repo_path = f"{self.OPTIONAL_SKILLS_PREFIX}/{rel}" - # Download the FULL directory byte-exact (root-level install scripts, - # LICENSE, tests/). GitHubSource.fetch() would only pull SKILL.md + - # referenced support dirs. + # Download the FULL directory byte-exact (root-level install scripts, LICENSE, tests/). + # GitHubSource.fetch() would only pull SKILL.md + referenced support dirs. tree = github._get_repo_tree(self.OFFICIAL_REPO) if tree is None: return None files: Dict[str, Union[str, bytes]] = {} - for rel_file, item_path, regular in _tree_members(tree[1], f"{repo_path}/"): + for rel_file, item_path, regular in _tree_members(tree[1], f"{self.OPTIONAL_SKILLS_PREFIX}/{rel}/"): if not regular or _skip_bundle_file(rel_file): continue content = github._fetch_file_bytes(self.OFFICIAL_REPO, item_path) @@ -230,35 +178,27 @@ class OptionalSkillSource(SkillSource): logger.warning("Live-repo optional skill fetch failed for %s", item_path) return None files[rel_file] = content - if "SKILL.md" not in files: return None - # Live-fetched catalog stubs redirect the same way local ones do. upstream = self._upstream_pointer_from_content(files["SKILL.md"]) if upstream is not None: return self._fetch_from_upstream(upstream, rel) - logger.info("Optional skill '%s' fetched from live repo (not in local checkout)", rel) return self._bundle(rel, files) def _list_remote_skill_dirs(self) -> Dict[str, bool]: - """``category/skill`` dirs under optional-skills/ on live main. - - One repo-tree call (cached per-process by GitHubSource + the on-disk - index cache). {} when the network/API is unavailable — callers degrade - to local-only. - """ + """``category/skill`` dirs under optional-skills/ on live main. One repo-tree call (cached + per-process by GitHubSource + the on-disk index cache). {} when the network/API is + unavailable — callers degrade to local-only.""" if self._remote_dirs is not None: return self._remote_dirs def compute(): dirs: Dict[str, bool] = {} - tree = self._get_github()._get_repo_tree(self.OFFICIAL_REPO) - if tree is None: + if (tree := self._get_github()._get_repo_tree(self.OFFICIAL_REPO)) is None: return None - prefix = f"{self.OPTIONAL_SKILLS_PREFIX}/" - suffix = "/SKILL.md" + prefix, suffix = f"{self.OPTIONAL_SKILLS_PREFIX}/", "/SKILL.md" for item in tree[1]: path = item.get("path", "") if item.get("type") == "blob" and path.startswith(prefix) and path.endswith(suffix): @@ -267,25 +207,13 @@ class OptionalSkillSource(SkillSource): dirs[rel_dir] = True return dirs or None - self._remote_dirs = _memo_json( - "official_optional_dirs", compute, valid=lambda c: isinstance(c, dict) and bool(c), - ) or {} + self._remote_dirs = _memo_json("official_optional_dirs", compute, + valid=lambda c: isinstance(c, dict) and bool(c)) or {} return self._remote_dirs - def _upstream_pointer(self, skill_dir: Path) -> Optional[Dict[str, str]]: - """Upstream pointer for a catalog-stub skill dir, or None for vendored skills. - - A stub declares ``metadata.hermes.upstream: {repo: owner/name, path: ...}`` - in its SKILL.md frontmatter. - """ - try: - content = (skill_dir / "SKILL.md").read_text(encoding="utf-8") - except (OSError, UnicodeDecodeError): - return None - return self._upstream_pointer_from_content(content) - def _upstream_pointer_from_content(self, content: Union[str, bytes]) -> Optional[Dict[str, str]]: - """Parse ``metadata.hermes.upstream`` out of SKILL.md content.""" + """Parse ``metadata.hermes.upstream: {repo: owner/name, path: ...}`` out of SKILL.md content + (a catalog stub); None for vendored skills.""" if isinstance(content, bytes): try: content = content.decode("utf-8") @@ -302,51 +230,33 @@ class OptionalSkillSource(SkillSource): if not repo or repo.count("/") != 1 or not path: return None parts = _clean_rel_parts(path) - if parts is None: - return None - return {"repo": repo, "path": "/".join(parts)} + return None if parts is None else {"repo": repo, "path": "/".join(parts)} def _fetch_from_upstream(self, upstream: Dict[str, str], rel_id: str) -> Optional[SkillBundle]: - """Fetch an upstream-maintained optional skill via GitHubSource.fetch() - (full-tree download, symlink/unsafe-path rejection, quarantine + scan - downstream) and re-label it as an official catalog entry.""" + """Fetch an upstream-maintained optional skill via GitHubSource.fetch() (full-tree download, + symlink/unsafe-path rejection, quarantine + scan downstream) and re-label it as an official + catalog entry.""" bundle = self._get_github().fetch(f"{upstream['repo']}/{upstream['path']}") if bundle is None: - logger.warning( - "Upstream fetch failed for optional skill %s (%s:%s)", - rel_id, upstream["repo"], upstream["path"], - ) + logger.warning("Upstream fetch failed for optional skill %s (%s:%s)", + rel_id, upstream["repo"], upstream["path"]) return None return SkillBundle( - name=bundle.name, - files=bundle.files, - source="official", - identifier=f"official/{rel_id}", + name=bundle.name, files=bundle.files, source="official", identifier=f"official/{rel_id}", # Curated endorsement, but the content is live third-party: # "trusted", not "builtin", so a dangerous scan verdict still blocks. trust_level="trusted", - metadata={ - **bundle.metadata, - "upstream_repo": upstream["repo"], - "upstream_path": upstream["path"], - }, + metadata={**bundle.metadata, "upstream_repo": upstream["repo"], "upstream_path": upstream["path"]}, ) def _local_skill_mds(self): - if not self._optional_dir.is_dir(): - return - for skill_md in sorted(self._optional_dir.rglob("SKILL.md")): - if not is_excluded_skill_path( - skill_md.relative_to(self._optional_dir), root=self._optional_dir - ): - yield skill_md + root = self._optional_dir + return (md for md in (sorted(root.rglob("SKILL.md")) if root.is_dir() else []) + if not is_excluded_skill_path(md.relative_to(root), root=root)) def _find_skill_dir(self, name: str) -> Optional[Path]: """Find a skill directory by name anywhere in optional-skills/.""" - for skill_md in self._local_skill_mds(): - if skill_md.parent.name == name: - return skill_md.parent - return None + return next((md.parent for md in self._local_skill_mds() if md.parent.name == name), None) def _scan_all(self) -> List[SkillMeta]: """Enumerate all optional skills with metadata.""" @@ -357,30 +267,18 @@ class OptionalSkillSource(SkillSource): content = skill_md.read_text(encoding="utf-8") except (OSError, UnicodeDecodeError): continue - fm = _parse_frontmatter(content) tags = _hermes_tags(fm) - results.append(self._meta( - parent.relative_to(self._optional_dir).as_posix(), - fm.get("name", parent.name), fm.get("description", "")[:200], - tags if isinstance(tags, list) else [], - )) - + results.append(self._meta(parent.relative_to(self._optional_dir).as_posix(), fm.get("name", parent.name), + fm.get("description", "")[:200], tags if isinstance(tags, list) else [])) return results -# --------------------------------------------------------------------------- -# Hermes centralized index source -# --------------------------------------------------------------------------- - class HermesIndexSource(SkillSource): - """Skill source backed by the centralized Hermes Skills Index. - - A JSON catalog on the docs site, rebuilt daily by CI, with metadata + - resolved GitHub paths for every skill — search and path discovery cost - zero GitHub API calls. When unavailable every method returns empty/None - so downstream sources take over transparently. - """ + """Skill source backed by the centralized Hermes Skills Index: a JSON catalog on the docs site, + rebuilt daily by CI, with metadata + resolved GitHub paths for every skill — search and path + discovery cost zero GitHub API calls. When unavailable every method returns empty/None so + downstream sources take over transparently.""" SOURCE_ID = "hermes-index" @@ -392,8 +290,7 @@ class HermesIndexSource(SkillSource): def _ensure_loaded(self) -> dict: if not self._loaded: - self._index = hub()._load_hermes_index() - self._loaded = True + self._index, self._loaded = hub()._load_hermes_index(), True return self._index or {} def _skills(self) -> list: @@ -414,45 +311,31 @@ class HermesIndexSource(SkillSource): return entry.get("trust_level", "community") if entry else "community" def search(self, query: str, limit: int = 10) -> List[SkillMeta]: - """Search the cached index (zero API calls). - - Matches name, description, tags, identifier and ``extra.provider`` (so - ``nvidia`` finds ``NVIDIA/skills/...`` entries stored as source - "github"). Ranked exact name > name prefix > provider > whole-word > - name substring > other, index order as tiebreaker — a raw - break-at-limit slice buried the most relevant skills. - """ + """Search the cached index (zero API calls). Matches name, description, tags, identifier and + ``extra.provider`` (so ``nvidia`` finds ``NVIDIA/skills/...`` entries stored as source + "github"). Ranked exact name > name prefix > provider > whole-word > name substring > other, + index order as tiebreaker — a raw break-at-limit slice buried the most relevant skills.""" skills = self._skills() if not skills: return [] - if not query.strip(): return [self._to_meta(s) for s in skills[:limit]] # featured / index order - query_lower = query.lower() scored: List[Tuple[int, int, dict]] = [] for i, s in enumerate(skills): name = str(s.get("name", "")).lower() provider = str((s.get("extra") or {}).get("provider", "")).lower() haystack = " ".join([ - name, - str(s.get("description", "")).lower(), - " ".join(str(t).lower() for t in s.get("tags", [])), - str(s.get("identifier", "")).lower(), - provider, + name, str(s.get("description", "")).lower(), " ".join(str(t).lower() for t in s.get("tags", [])), + str(s.get("identifier", "")).lower(), provider, ]) if query_lower not in haystack: continue ranks = ( - name == query_lower, - name.startswith(query_lower), - provider == query_lower, - query_lower in name.split() or query_lower in provider.split(), - query_lower in name, - True, + name == query_lower, name.startswith(query_lower), provider == query_lower, + query_lower in name.split() or query_lower in provider.split(), query_lower in name, True, ) scored.append((ranks.index(True), i, s)) - scored.sort(key=lambda x: (x[0], x[1])) return [self._to_meta(s) for _, _, s in scored[:limit]] @@ -462,11 +345,8 @@ class HermesIndexSource(SkillSource): entry = self._find_entry(identifier) if not entry: return None - - candidates = [entry.get("resolved_github_id")] repo, path = entry.get("repo", ""), entry.get("path", "") - if repo and path: - candidates.append(f"{repo}/{path}") + candidates = [entry.get("resolved_github_id")] + ([f"{repo}/{path}"] if repo and path else []) for github_id in filter(None, candidates): bundle = self._get_github().fetch(github_id) if bundle: @@ -484,23 +364,15 @@ class HermesIndexSource(SkillSource): """Exact identifier match first, then match with source prefixes stripped.""" skills = self._skills() normalized = _strip_prefix(identifier, _INDEX_ID_PREFIXES) - return next( - (s for s in skills if s.get("identifier") == identifier), None, - ) or next( - (s for s in skills if _strip_prefix(s.get("identifier", ""), _INDEX_ID_PREFIXES) == normalized), - None, + return next((s for s in skills if s.get("identifier") == identifier), None) or next( + (s for s in skills if _strip_prefix(s.get("identifier", ""), _INDEX_ID_PREFIXES) == normalized), None, ) @staticmethod def _to_meta(entry: dict) -> SkillMeta: return SkillMeta( - name=entry.get("name", ""), - description=entry.get("description", ""), - source=entry.get("source", "hermes-index"), - identifier=entry.get("identifier", ""), - trust_level=entry.get("trust_level", "community"), - repo=entry.get("repo"), - path=entry.get("path"), - tags=entry.get("tags", []), - extra=entry.get("extra", {}), + name=entry.get("name", ""), description=entry.get("description", ""), + source=entry.get("source", "hermes-index"), identifier=entry.get("identifier", ""), + trust_level=entry.get("trust_level", "community"), repo=entry.get("repo"), path=entry.get("path"), + tags=entry.get("tags", []), extra=entry.get("extra", {}), ) diff --git a/tools/skills_hub_search.py b/tools/skills_hub_search.py index ab4020e6ef..505cb4cea5 100644 --- a/tools/skills_hub_search.py +++ b/tools/skills_hub_search.py @@ -25,7 +25,6 @@ if TYPE_CHECKING: # runtime use resolves through the origin (test patch target) logger = logging.getLogger("tools.skills_hub") HERMES_INDEX_URL = "https://hermes-agent.nousresearch.com/docs/api/skills-index.json" - HERMES_INDEX_TTL = 6 * 3600 # 6 hours @@ -49,16 +48,11 @@ def _load_hermes_index() -> Optional[dict]: cached = _read_json_if_fresh(cache_file, HERMES_INDEX_TTL) if cached is not None: return cached - data = None for accept_encoding in ("gzip, deflate", "identity"): try: - resp = httpx.get( - HERMES_INDEX_URL, - timeout=15, - follow_redirects=True, - headers={"Accept-Encoding": accept_encoding}, - ) + resp = httpx.get(HERMES_INDEX_URL, timeout=15, follow_redirects=True, + headers={"Accept-Encoding": accept_encoding}) if resp.status_code != 200: logger.debug("Hermes index fetch returned %d", resp.status_code) return _load_stale_index_cache() @@ -69,16 +63,13 @@ def _load_hermes_index() -> Optional[dict]: except (httpx.HTTPError, json.JSONDecodeError) as e: logger.debug("Hermes index fetch failed: %s", e) return _load_stale_index_cache() - if not isinstance(data, dict) or "skills" not in data: return _load_stale_index_cache() - try: cache_file.parent.mkdir(parents=True, exist_ok=True) cache_file.write_text(json.dumps(data), encoding="utf-8") except OSError: pass - return data @@ -102,7 +93,6 @@ def create_source_router(auth: Optional[GitHubAuth] = None) -> List[SkillSource] ) if auth is None: auth = GitHubAuth() - return [ OptionalSkillSource(auth=auth), # official optional skills (highest priority) HermesIndexSource(auth=auth), # centralized index (search + resolved install paths) @@ -116,9 +106,7 @@ def create_source_router(auth: Optional[GitHubAuth] = None) -> List[SkillSource] ] -def _search_one_source( - src: SkillSource, query: str, limit: int -) -> Tuple[str, List[SkillMeta]]: +def _search_one_source(src: SkillSource, query: str, limit: int) -> Tuple[str, List[SkillMeta]]: """Search a single source. Runs in a thread for parallelism.""" try: return src.source_id(), src.search(query, limit=limit) @@ -137,8 +125,7 @@ def _select_active_sources(sources: List[SkillSource], source_filter: str) -> Li """ effective = "all" if source_filter.strip().lower() in _PROVIDER_FILTER_VALUES else source_filter index_available = effective == "all" and any( - src.source_id() == "hermes-index" and getattr(src, "is_available", False) - for src in sources + src.source_id() == "hermes-index" and getattr(src, "is_available", False) for src in sources ) active: List[SkillSource] = [] for src in sources: @@ -152,12 +139,8 @@ def _select_active_sources(sources: List[SkillSource], source_filter: str) -> Li def parallel_search_sources( - sources: List[SkillSource], - query: str = "", - per_source_limits: Optional[Dict[str, int]] = None, - source_filter: str = "all", - overall_timeout: float = 30, - on_source_done: Optional[Any] = None, + sources: List[SkillSource], query: str = "", per_source_limits: Optional[Dict[str, int]] = None, + source_filter: str = "all", overall_timeout: float = 30, on_source_done: Optional[Any] = None, ) -> Tuple[List[SkillMeta], Dict[str, int], List[str]]: """Search all sources in parallel with an overall timeout. @@ -168,11 +151,9 @@ def parallel_search_sources( per_source_limits = per_source_limits or {} active = _select_active_sources(sources, source_filter) - all_results: List[SkillMeta] = [] source_counts: Dict[str, int] = {} timed_out_ids: List[str] = [] - if not active: return all_results, source_counts, timed_out_ids @@ -185,7 +166,6 @@ def parallel_search_sources( pool.submit(_search_one_source, src, query, per_source_limits.get(src.source_id(), 50)): src.source_id() for src in active } - try: for fut in as_completed(futures, timeout=overall_timeout): try: @@ -202,7 +182,6 @@ def parallel_search_sources( logger.debug("Skills browse timed out waiting for: %s", ", ".join(timed_out_ids)) finally: pool.shutdown(wait=False, cancel_futures=True) - return all_results, source_counts, timed_out_ids @@ -210,17 +189,10 @@ def unified_search(query: str, sources: List[SkillSource], source_filter: str = "all", limit: int = 10) -> List[SkillMeta]: """Search all sources (in parallel) and merge results.""" from tools.skills_hub import _filter_results_by_provider, parallel_search_sources - all_results, _, _ = parallel_search_sources( - sources, - query=query, - source_filter=source_filter, - overall_timeout=30, - ) - + all_results, _, _ = parallel_search_sources(sources, query=query, source_filter=source_filter, overall_timeout=30) # Provider filters target ``extra.provider`` on the merged set, not a source id. if source_filter.strip().lower() in _PROVIDER_FILTER_VALUES: all_results = _filter_results_by_provider(all_results, source_filter) - deduped = _dedupe_by_trust(all_results) # Stable-sort by trust before truncating so the limit cut never drops a # builtin/official entry because a high-volume community source finished diff --git a/tools/skills_hub_skillssh.py b/tools/skills_hub_skillssh.py index 436fc892ef..606aa701b9 100644 --- a/tools/skills_hub_skillssh.py +++ b/tools/skills_hub_skillssh.py @@ -32,8 +32,7 @@ class SkillsShSource(SkillSource): _SITEMAP_HEADERS = {"Accept-Encoding": "gzip"} _SITEMAP_LOC_RE = re.compile(r"([^<]+)", re.IGNORECASE) _SITEMAP_SKILL_RE = re.compile( - r"^https?://(?:www\.)?skills\.sh/(?P[^/]+)/(?P[^/]+)/(?P[^/]+)/?$", - re.IGNORECASE, + r"^https?://(?:www\.)?skills\.sh/(?P[^/]+)/(?P[^/]+)/(?P[^/]+)/?$", re.IGNORECASE, ) _SKILL_LINK_RE = re.compile(r'href=["\']/(?P(?!agents/|_next/|api/)[^"\'/]+/[^"\'/]+/[^"\'/]+)["\']') _INSTALL_CMD_RE = re.compile( @@ -56,28 +55,20 @@ class SkillsShSource(SkillSource): _STANDARD_BASE_PATHS = ("skills/", ".agents/skills/", ".claude/skills/") _strip_html = staticmethod(_strip_html) - SOURCE_ID = "skills-sh" def __init__(self, auth: GitHubAuth): - self.auth = auth - self.github = GitHubSource(auth=auth) + self.auth, self.github = auth, GitHubSource(auth=auth) def trust_level_for(self, identifier: str) -> str: return self.github.trust_level_for(self._normalize_identifier(identifier)) def _meta(self, canonical: str, *, name: str, description: str, path: str, extra: Optional[Dict[str, Any]] = None) -> SkillMeta: - repo = "/".join(canonical.split("/", 2)[:2]) return SkillMeta( - name=name, - description=description, - source="skills.sh", - identifier=self._wrap_identifier(canonical), - trust_level=self.github.trust_level_for(canonical), - repo=repo, - path=path, - extra=extra if extra is not None else {}, + name=name, description=description, source="skills.sh", identifier=self._wrap_identifier(canonical), + trust_level=self.github.trust_level_for(canonical), repo="/".join(canonical.split("/", 2)[:2]), + path=path, extra=extra if extra is not None else {}, ) def _urls_for(self, canonical: str, repo: str) -> Dict[str, str]: @@ -87,20 +78,17 @@ class SkillsShSource(SkillSource): if not query.strip(): # Empty query = bulk catalog dump (build_skills_index.py) — walk the sitemap. return self._sitemap_catalog(limit) - cache_key = f"skills_sh_search_{hashlib.md5(f'{query}|{limit}'.encode()).hexdigest()}" cached = _cached_metas(cache_key) if cached is not None: return cached[:limit] - data = _get_json(self.SEARCH_URL, params={"q": query, "limit": limit}) if data is None: return [] items = data.get("skills", []) if isinstance(data, dict) else [] if not isinstance(items, list): return [] - - results = [m for m in (self._meta_from_search_item(i) for i in items[:limit]) if m] + results = [m for m in map(self._meta_from_search_item, items[:limit]) if m] _cache_metas(cache_key, results) return results @@ -111,8 +99,7 @@ class SkillsShSource(SkillSource): def _relabel(github_id: Optional[str]) -> Optional[SkillBundle]: bundle = self.github.fetch(github_id) if github_id else None if bundle: - bundle.source = "skills.sh" - bundle.identifier = self._wrap_identifier(canonical) + bundle.source, bundle.identifier = "skills.sh", self._wrap_identifier(canonical) bundle.metadata.update(self._detail_to_metadata(canonical, detail)) return bundle or None @@ -126,9 +113,7 @@ class SkillsShSource(SkillSource): canonical = self._normalize_identifier(identifier) detail = self._fetch_detail_page(canonical) meta = self._resolve_github_meta(canonical, detail=detail) - if meta: - return self._finalize_inspect_meta(meta, canonical, detail) - return None + return self._finalize_inspect_meta(meta, canonical, detail) if meta else None def _sitemap_catalog(self, limit: int) -> List[SkillMeta]: """Enumerate the full catalog via the sitemap (cached for the index TTL — @@ -140,37 +125,27 @@ class SkillsShSource(SkillSource): # Step 1: sitemap index -> per-skill sitemap URLs. index_xml = _get_text(self.SITEMAP_INDEX_URL, follow_redirects=True, headers=self._SITEMAP_HEADERS) - skill_sitemap_urls = [ - m.group(1).strip() for m in self._SITEMAP_LOC_RE.finditer(index_xml or "") - if "sitemap-skills" in m.group(1) - ] + skill_sitemap_urls = [m.group(1).strip() for m in self._SITEMAP_LOC_RE.finditer(index_xml or "") + if "sitemap-skills" in m.group(1)] if not skill_sitemap_urls: return self._featured_skills(limit) # Step 2: collect canonical "owner/repo/skill" IDs from each sitemap. - seen: set[str] = set() - results: List[SkillMeta] = [] + seen, results = set(), [] for sitemap_url in skill_sitemap_urls: xml = _get_text(sitemap_url, timeout=30, follow_redirects=True, headers=self._SITEMAP_HEADERS) for loc_match in self._SITEMAP_LOC_RE.finditer(xml or ""): m = self._SITEMAP_SKILL_RE.match(loc_match.group(1).strip()) if not m: continue - repo = f"{m.group('owner')}/{m.group('repo')}" - skill_name = m.group("skill") - canonical = f"{repo}/{skill_name}" - if canonical in seen: - continue - seen.add(canonical) - results.append(self._meta( - canonical, name=skill_name, - description=f"Indexed by skills.sh from {repo}", - path=skill_name, extra=self._urls_for(canonical, repo), - )) - + repo, skill = f"{m.group('owner')}/{m.group('repo')}", m.group("skill") + canonical = f"{repo}/{skill}" + if canonical not in seen: + seen.add(canonical) + results.append(self._meta(canonical, name=skill, description=f"Indexed by skills.sh from {repo}", + path=skill, extra=self._urls_for(canonical, repo))) if not results: return self._featured_skills(limit) - _cache_metas(cache_key, results) return results[:limit] if limit > 0 else results @@ -179,57 +154,41 @@ class SkillsShSource(SkillSource): cached = _cached_metas(cache_key) if cached is not None: return cached[:limit] - html = _get_text(self.BASE_URL) if html is None: return [] - - seen: set[str] = set() - results: List[SkillMeta] = [] + seen, results = set(), [] for match in self._SKILL_LINK_RE.finditer(html): canonical = match.group("id") - if canonical in seen: - continue + split = None if canonical in seen else _split_repo_id(canonical) seen.add(canonical) - split = _split_repo_id(canonical) if split is None: continue repo, skill_path = split - results.append(self._meta( - canonical, name=skill_path.split("/")[-1], - description=f"Featured on skills.sh from {repo}", - path=skill_path, - )) + results.append(self._meta(canonical, name=skill_path.split("/")[-1], + description=f"Featured on skills.sh from {repo}", path=skill_path)) if len(results) >= limit: break - _cache_metas(cache_key, results) return results def _meta_from_search_item(self, item: dict) -> Optional[SkillMeta]: if not isinstance(item, dict): return None - - canonical = item.get("id") - repo = item.get("source") - skill_path = item.get("skillId") + canonical, repo, skill_path = item.get("id"), item.get("source"), item.get("skillId") if not isinstance(canonical, str) or canonical.count("/") < 2: if not (isinstance(repo, str) and isinstance(skill_path, str)): return None canonical = f"{repo}/{skill_path}" - split = _split_repo_id(canonical) if split is None: return None repo, skill_path = split installs = item.get("installs") installs_label = f" · {int(installs):,} installs" if isinstance(installs, int) else "" - return self._meta( - canonical, - name=str(item.get("name") or skill_path.split("/")[-1]), - description=f"Indexed by skills.sh from {repo}{installs_label}", - path=skill_path, + canonical, name=str(item.get("name") or skill_path.split("/")[-1]), + description=f"Indexed by skills.sh from {repo}{installs_label}", path=skill_path, extra={"installs": installs, **self._urls_for(canonical, repo)}, ) @@ -237,28 +196,21 @@ class SkillsShSource(SkillSource): def compute(): html = _get_text(f"{self.BASE_URL}/{identifier}") return None if html is None else self._parse_detail_page(identifier, html) or None - - return _memo_json( - f"skills_sh_detail_{hashlib.md5(identifier.encode()).hexdigest()}", compute, - valid=lambda c: isinstance(c, dict), - ) + key = f"skills_sh_detail_{hashlib.md5(identifier.encode()).hexdigest()}" + return _memo_json(key, compute, valid=lambda c: isinstance(c, dict)) def _parse_detail_page(self, identifier: str, html: str) -> Optional[dict]: split = _split_repo_id(identifier) if split is None: return None repo, install_skill = split - - install_command = None - install_match = self._INSTALL_CMD_RE.search(html) + install_command, install_match = None, self._INSTALL_CMD_RE.search(html) if install_match: install_command = install_match.group(0).strip() install_skill = (install_match.group("skill") or install_skill).strip() repo = self._extract_repo_slug((install_match.group("repo") or "").strip()) or repo - return { - "repo": repo, - "install_skill": install_skill, + "repo": repo, "install_skill": install_skill, "page_title": self._extract_first_match(self._PAGE_H1_RE, html), "body_title": self._extract_first_match(self._PROSE_H1_RE, html), "body_summary": self._extract_first_match(self._PROSE_P_RE, html), @@ -277,32 +229,21 @@ class SkillsShSource(SkillSource): skill_token = skill_path.split("/")[-1] tokens = [skill_token] if isinstance(detail, dict): - tokens.extend([ - detail.get("install_skill", ""), - detail.get("page_title", ""), - detail.get("body_title", ""), - ]) + tokens.extend(detail.get(k, "") for k in ("install_skill", "page_title", "body_title")) def _match_in(base_path: str) -> Optional[str]: try: skills = self.github._list_skills_in_repo(repo, base_path) except Exception: return None - for meta in skills: - if self._matches_skill_tokens(meta, tokens): - return meta.identifier - return None - - for base_path in self._STANDARD_BASE_PATHS: - found = _match_in(base_path) - if found: - return found + return next((m.identifier for m in skills if self._matches_skill_tokens(m, tokens)), None) # One recursive tree lookup before brute-forcing every top-level dir # (avoids request bursts on categorized repos like borghei/claude-skills). - tree_result = self.github._find_skill_in_repo_tree(repo, skill_token) - if tree_result: - return tree_result + found = (next((f for f in map(_match_in, self._STANDARD_BASE_PATHS) if f), None) + or self.github._find_skill_in_repo_tree(repo, skill_token)) + if found: + return found # Fallback: scan repo root for directories that might contain skills. try: @@ -322,7 +263,6 @@ class SkillsShSource(SkillSource): return found except Exception: pass - return None def _resolve_github_meta(self, identifier: str, detail: Optional[dict] = None) -> Optional[SkillMeta]: @@ -330,21 +270,15 @@ class SkillsShSource(SkillSource): meta = self.github.inspect(candidate) if meta: return meta - resolved = self._discover_identifier(identifier, detail=detail) - if resolved: - return self.github.inspect(resolved) - return None + return self.github.inspect(resolved) if resolved else None def _finalize_inspect_meta(self, meta: SkillMeta, canonical: str, detail: Optional[dict]) -> SkillMeta: - meta.source = "skills.sh" - meta.identifier = self._wrap_identifier(canonical) + meta.source, meta.identifier = "skills.sh", self._wrap_identifier(canonical) meta.trust_level = self.trust_level_for(canonical) meta.extra = {**meta.extra, **self._detail_to_metadata(canonical, detail)} - if isinstance(detail, dict): - body_summary = detail.get("body_summary") - weekly_installs = detail.get("weekly_installs") + body_summary, weekly_installs = detail.get("body_summary"), detail.get("weekly_installs") if body_summary: meta.description = body_summary elif meta.description and weekly_installs: @@ -353,36 +287,28 @@ class SkillsShSource(SkillSource): @classmethod def _matches_skill_tokens(cls, meta: SkillMeta, skill_tokens: List[str]) -> bool: - candidates = set() - candidates.update(cls._token_variants(meta.name)) - candidates.update(cls._token_variants(meta.path)) - candidates.update(cls._token_variants(meta.identifier.split("/", 2)[-1] if meta.identifier else None)) + candidates = (cls._token_variants(meta.name) | cls._token_variants(meta.path) + | cls._token_variants(meta.identifier.split("/", 2)[-1] if meta.identifier else None)) return any(cls._token_variants(token) & candidates for token in skill_tokens) @staticmethod def _token_variants(value: Optional[str]) -> set[str]: if not value: return set() - plain = _strip_html(str(value)).strip().strip("/").lower() if not plain: return set() - - base = plain.split("/")[-1] - sanitized = re.sub(r'[^a-z0-9/_-]+', '-', plain).strip('-') + base, sanitized = plain.split("/")[-1], re.sub(r'[^a-z0-9/_-]+', '-', plain).strip('-') tail = base.lstrip('@') variants = { - plain, plain.replace("_", "-"), plain.replace("/", "-"), - base, base.replace("_", "-"), - sanitized, sanitized.replace("/", "-"), sanitized.split("/")[-1], - tail, tail.replace("_", "-"), + plain, plain.replace("_", "-"), plain.replace("/", "-"), base, base.replace("_", "-"), + sanitized, sanitized.replace("/", "-"), sanitized.split("/")[-1], tail, tail.replace("_", "-"), } return {v for v in variants if v} @staticmethod def _extract_repo_slug(repo_value: str) -> Optional[str]: - repo_value = repo_value.strip().removeprefix("https://github.com/") - parts = repo_value.strip("/").split("/") + parts = repo_value.strip().removeprefix("https://github.com/").strip("/").split("/") return f"{parts[0]}/{parts[1]}" if len(parts) >= 2 else None @staticmethod @@ -398,9 +324,8 @@ class SkillsShSource(SkillSource): metadata["repo_url"] = f"https://github.com/{parts[0]}/{parts[1]}" if isinstance(detail, dict): for key in ("weekly_installs", "install_command", "repo_url", "detail_url", "security_audits"): - value = detail.get(key) - if value: - metadata[key] = value + if detail.get(key): + metadata[key] = detail[key] return metadata @classmethod @@ -413,9 +338,7 @@ class SkillsShSource(SkillSource): audits: Dict[str, str] = {} for audit in ("agent-trust-hub", "socket", "snyk"): idx = html.find(f"/security/{audit}") - if idx == -1: - continue - match = re.search(r'(Pass|Warn|Fail)', html[idx:idx + 500], re.IGNORECASE) + match = re.search(r'(Pass|Warn|Fail)', html[idx:idx + 500], re.IGNORECASE) if idx != -1 else None if match: audits[audit] = match.group(1).title() return audits @@ -430,11 +353,8 @@ class SkillsShSource(SkillSource): split = _split_repo_id(identifier) if split is None: return [identifier] - repo, skill_path = split - skill_path = skill_path.lstrip("/") - return list(dict.fromkeys( - [f"{repo}/{skill_path}"] + [f"{repo}/{base}{skill_path}" for base in cls._STANDARD_BASE_PATHS] - )) + repo, path = split[0], split[1].lstrip("/") + return list(dict.fromkeys([f"{repo}/{path}"] + [f"{repo}/{b}{path}" for b in cls._STANDARD_BASE_PATHS])) @staticmethod def _wrap_identifier(identifier: str) -> str: diff --git a/tools/skills_hub_sources.py b/tools/skills_hub_sources.py index 6e61d443fe..4de7b5d132 100644 --- a/tools/skills_hub_sources.py +++ b/tools/skills_hub_sources.py @@ -9,16 +9,14 @@ from urllib.parse import quote, urljoin, urlparse, urlunparse from tools.skills_hub_models import ( GuardedFetchMixin, SkillBundle, SkillMeta, SkillSource, _first_matching, _get_json, _get_text, - _hermes_tags, _matches_query, _memo_json, _parse_frontmatter, _referenced_support_paths, + _hermes_tags, _memo_json, _parse_frontmatter, _referenced_support_paths, _validate_bundle_rel_path, _validate_skill_name, hub, ) logger = logging.getLogger("tools.skills_hub") -# --------------------------------------------------------------------------- -# Well-known Agent Skills endpoint source adapter -# --------------------------------------------------------------------------- +# --- Well-known Agent Skills endpoint source adapter ------------------------ class WellKnownSkillSource(GuardedFetchMixin, SkillSource): """Read skills from a domain exposing /.well-known/skills/index.json.""" @@ -28,11 +26,8 @@ class WellKnownSkillSource(GuardedFetchMixin, SkillSource): def _meta(self, parsed: dict, skill_name: str, description: str, files: Any, **extra) -> SkillMeta: return SkillMeta( - name=skill_name, - description=description, - source="well-known", - identifier=self._wrap_identifier(parsed["base_url"], skill_name), - trust_level="community", + name=skill_name, description=description, source="well-known", + identifier=self._wrap_identifier(parsed["base_url"], skill_name), trust_level="community", path=skill_name, extra={"index_url": parsed["index_url"], "base_url": parsed["base_url"], "files": files, **extra}, ) @@ -42,57 +37,42 @@ class WellKnownSkillSource(GuardedFetchMixin, SkillSource): parsed = self._parse_index(index_url) if index_url else None if not parsed: return [] - results: List[SkillMeta] = [] for entry in parsed["skills"][:limit]: name = entry.get("name") if not isinstance(name, str) or not name: continue files = entry.get("files", ["SKILL.md"]) - results.append(self._meta( - parsed, name, str(entry.get("description", "")), - files if isinstance(files, list) else ["SKILL.md"], - )) + results.append(self._meta(parsed, name, str(entry.get("description", "")), + files if isinstance(files, list) else ["SKILL.md"])) return results def inspect(self, identifier: str) -> Optional[SkillMeta]: parsed = self._parse_identifier(identifier) - if not parsed: - return None - entry = self._index_entry(parsed["index_url"], parsed["skill_name"]) - if not entry: - return None - - skill_md = self._fetch_text(f"{parsed['skill_url']}/SKILL.md") + entry = self._index_entry(parsed["index_url"], parsed["skill_name"]) if parsed else None + skill_md = self._fetch_text(f"{parsed['skill_url']}/SKILL.md") if entry else None if skill_md is None: return None - fm = _parse_frontmatter(skill_md) - return self._meta( - parsed, str(fm.get("name") or parsed["skill_name"]), - str(fm.get("description") or entry.get("description") or ""), - entry.get("files", ["SKILL.md"]), endpoint=parsed["skill_url"], - ) + return self._meta(parsed, str(fm.get("name") or parsed["skill_name"]), + str(fm.get("description") or entry.get("description") or ""), + entry.get("files", ["SKILL.md"]), endpoint=parsed["skill_url"]) def fetch(self, identifier: str) -> Optional[SkillBundle]: parsed = self._parse_identifier(identifier) if not parsed: return None - try: skill_name = _validate_skill_name(parsed["skill_name"]) except ValueError: logger.warning("Well-known skill identifier contained unsafe skill name: %s", identifier) return None - entry = self._index_entry(parsed["index_url"], parsed["skill_name"]) if not entry: return None - files = entry.get("files", ["SKILL.md"]) if not isinstance(files, list) or not files: files = ["SKILL.md"] - downloaded: Dict[str, str] = {} for rel_path in files: if not isinstance(rel_path, str) or not rel_path: @@ -100,32 +80,19 @@ class WellKnownSkillSource(GuardedFetchMixin, SkillSource): try: safe_rel_path = _validate_bundle_rel_path(rel_path) except ValueError: - logger.warning( - "Well-known skill %s advertised unsafe file path: %r", - identifier, - rel_path, - ) + logger.warning("Well-known skill %s advertised unsafe file path: %r", identifier, rel_path) return None text = self._fetch_text(f"{parsed['skill_url']}/{safe_rel_path}") if text is None: return None downloaded[safe_rel_path] = text - if "SKILL.md" not in downloaded: return None - return SkillBundle( - name=skill_name, - files=downloaded, - source="well-known", - identifier=self._wrap_identifier(parsed["base_url"], skill_name), - trust_level="community", - metadata={ - "index_url": parsed["index_url"], - "base_url": parsed["base_url"], - "endpoint": parsed["skill_url"], - "files": files, - }, + name=skill_name, files=downloaded, source="well-known", + identifier=self._wrap_identifier(parsed["base_url"], skill_name), trust_level="community", + metadata={"index_url": parsed["index_url"], "base_url": parsed["base_url"], + "endpoint": parsed["skill_url"], "files": files}, ) def _query_to_index_url(self, query: str) -> Optional[str]: @@ -135,36 +102,27 @@ class WellKnownSkillSource(GuardedFetchMixin, SkillSource): if query.endswith("/index.json"): return query if f"{self.BASE_PATH}/" in query: - base_url = query.split(f"{self.BASE_PATH}/", 1)[0] + self.BASE_PATH - return f"{base_url}/index.json" + return query.split(f"{self.BASE_PATH}/", 1)[0] + f"{self.BASE_PATH}/index.json" return query.rstrip("/") + f"{self.BASE_PATH}/index.json" def _parse_identifier(self, identifier: str) -> Optional[dict]: raw = identifier[len("well-known:"):] if identifier.startswith("well-known:") else identifier if not raw.startswith(("http://", "https://")): return None - parsed_url = urlparse(raw) clean_url = urlunparse(parsed_url._replace(fragment="")) - fragment = parsed_url.fragment - if clean_url.endswith("/index.json"): - if not fragment: + if not parsed_url.fragment: return None - base_url = clean_url[:-len("/index.json")] - skill_name = fragment + base_url, skill_name = clean_url[:-len("/index.json")], parsed_url.fragment skill_url = f"{base_url}/{skill_name}" else: skill_url = clean_url[:-len("/SKILL.md")] if clean_url.endswith("/SKILL.md") else clean_url.rstrip("/") if f"{self.BASE_PATH}/" not in skill_url: return None base_url, skill_name = skill_url.rsplit("/", 1) - return { - "index_url": f"{base_url}/index.json", - "base_url": base_url, - "skill_name": skill_name, - "skill_url": skill_url, - } + return {"index_url": f"{base_url}/index.json", "base_url": base_url, + "skill_name": skill_name, "skill_url": skill_url} def _parse_index(self, index_url: str) -> Optional[dict]: def compute(): @@ -180,42 +138,32 @@ class WellKnownSkillSource(GuardedFetchMixin, SkillSource): return None return {"index_url": index_url, "base_url": index_url[:-len("/index.json")], "skills": skills} - return _memo_json( - f"well_known_index_{hashlib.md5(index_url.encode()).hexdigest()}", compute, - valid=lambda c: isinstance(c, dict) and isinstance(c.get("skills"), list), - ) + return _memo_json(f"well_known_index_{hashlib.md5(index_url.encode()).hexdigest()}", compute, + valid=lambda c: isinstance(c, dict) and isinstance(c.get("skills"), list)) def _index_entry(self, index_url: str, skill_name: str) -> Optional[dict]: parsed = self._parse_index(index_url) - if not parsed: - return None - for entry in parsed["skills"]: - if isinstance(entry, dict) and entry.get("name") == skill_name: - return entry - return None + skills = parsed["skills"] if parsed else [] + return next((e for e in skills if isinstance(e, dict) and e.get("name") == skill_name), None) @staticmethod def _wrap_identifier(base_url: str, skill_name: str) -> str: return f"well-known:{base_url.rstrip('/')}/{skill_name}" -# --------------------------------------------------------------------------- -# Direct URL source adapter -# --------------------------------------------------------------------------- +# --- Direct URL source adapter ---------------------------------------------- class UrlSource(GuardedFetchMixin, SkillSource): """Fetch SKILL.md plus explicitly referenced, allowlisted support files. - The identifier IS the URL (``https://example.com/path/SKILL.md``). Bare URLs - cannot enumerate a repository, so only exact references below - references/templates/scripts/assets are fetched. The skill name comes from - frontmatter ``name:`` (URL-slug fallback); trust is always ``community``. + The identifier IS the URL (``https://example.com/path/SKILL.md``). Bare URLs cannot enumerate a + repository, so only exact references below references/templates/scripts/assets are fetched. The + skill name comes from frontmatter ``name:`` (URL-slug fallback); trust is always ``community``. """ SOURCE_ID = "url" - # Skill names must look like identifiers: lowercase letters/digits with - # optional hyphens/underscores. Blocks dangerous (``../evil``) AND useless - # (``SKILL``, ``README``, empty) candidates before they hit the disk. + # Skill names must look like identifiers: lowercase letters/digits with optional hyphens/underscores. + # Blocks dangerous (``../evil``) AND useless (``SKILL``, ``README``, empty) candidates before they hit the disk. _VALID_NAME_RE = re.compile(r"^[a-z][a-z0-9_-]*$") def search(self, query: str, limit: int = 10) -> List[SkillMeta]: @@ -227,15 +175,13 @@ class UrlSource(GuardedFetchMixin, SkillSource): if not isinstance(identifier, str): return False ident = identifier.strip() - if not ident.lower().startswith(("http://", "https://")): - return False - if "/.well-known/skills/" in ident or ident.rstrip("/").endswith("/index.json"): + if (not ident.lower().startswith(("http://", "https://")) or "/.well-known/skills/" in ident + or ident.rstrip("/").endswith("/index.json")): return False try: - path = urlparse(ident).path + return urlparse(ident).path.lower().endswith(".md") except ValueError: return False - return path.lower().endswith(".md") def _load(self, identifier: str): """``(url, text, frontmatter, resolved name)`` for a claimed identifier, else None.""" @@ -255,12 +201,8 @@ class UrlSource(GuardedFetchMixin, SkillSource): url, _text, fm, name = loaded raw_tags = _hermes_tags(fm) return SkillMeta( - name=name or "", - description=str(fm.get("description") or ""), - source="url", - identifier=url, - trust_level="community", - path=name or "", + name=name or "", description=str(fm.get("description") or ""), source="url", identifier=url, + trust_level="community", path=name or "", tags=[str(t) for t in raw_tags] if isinstance(raw_tags, list) else [], extra={"url": url, "awaiting_name": name is None}, ) @@ -280,19 +222,13 @@ class UrlSource(GuardedFetchMixin, SkillSource): if urlparse(support_url).netloc != urlparse(url).netloc: return None content = self._fetch_bytes(support_url) - if content is None: - # A 404ing support file shouldn't sink the whole install. - logger.warning( - "URL skill %s: referenced support file %r could not be " - "fetched from %s; skipping it", - url, rel_path, support_url, - ) + if content is None: # A 404ing support file shouldn't sink the whole install. + logger.warning("URL skill %s: referenced support file %r could not be fetched from %s; skipping it", + url, rel_path, support_url) continue files[rel_path] = content - - # When no name resolves, return the bundle with an empty name and - # ``awaiting_name=True``: ``do_install`` prompts on a TTY or refuses - # non-interactively, without re-downloading after the user picks a name. + # When no name resolves, return the bundle with an empty name and ``awaiting_name=True``: ``do_install`` + # prompts on a TTY or refuses non-interactively, without re-downloading after the user picks a name. skill_name = "" if name is not None: try: @@ -300,37 +236,25 @@ class UrlSource(GuardedFetchMixin, SkillSource): except ValueError: logger.warning("URL skill %s produced unsafe skill name: %r", url, name) return None - - return SkillBundle( - name=skill_name, - files=files, - source="url", - identifier=url, - trust_level="community", - metadata={"url": url, "source_url": url, "awaiting_name": not skill_name}, - ) + return SkillBundle(name=skill_name, files=files, source="url", identifier=url, trust_level="community", + metadata={"url": url, "source_url": url, "awaiting_name": not skill_name}) @classmethod def _is_valid_skill_name(cls, name: Optional[str]) -> bool: if not isinstance(name, str): return False candidate = name.strip().lower() - if not candidate or candidate in {"skill", "readme", "index", "unnamed-skill"}: - return False - return bool(cls._VALID_NAME_RE.match(candidate)) + return bool(candidate) and candidate not in {"skill", "readme", "index", "unnamed-skill"} and bool( + cls._VALID_NAME_RE.match(candidate)) @classmethod def _resolve_skill_name(cls, fm: dict, url: str) -> Optional[str]: - """Frontmatter ``name:`` when valid, else a URL-slug candidate - (``...//SKILL.md`` -> ````, ``.../.md`` -> ````). - - None when nothing usable — the CLI then prompts or refuses rather than - auto-naming something like ``SKILL``. - """ + """Frontmatter ``name:`` when valid, else a URL-slug candidate (``...//SKILL.md`` -> ````, + ``.../.md`` -> ````). None when nothing usable — the CLI then prompts or refuses rather + than auto-naming something like ``SKILL``.""" fm_name = fm.get("name") if isinstance(fm, dict) else None if isinstance(fm_name, str) and cls._is_valid_skill_name(fm_name): return fm_name.strip() - try: path = urlparse(url).path except ValueError: @@ -338,16 +262,11 @@ class UrlSource(GuardedFetchMixin, SkillSource): parts = [p for p in path.split("/") if p] if len(parts) >= 2 and parts[-1].lower() == "skill.md" and cls._is_valid_skill_name(parts[-2]): return parts[-2] - if parts: - candidate = re.sub(r"\.md$", "", parts[-1], flags=re.IGNORECASE) - if cls._is_valid_skill_name(candidate): - return candidate - return None + candidate = re.sub(r"\.md$", "", parts[-1], flags=re.IGNORECASE) if parts else "" + return candidate if cls._is_valid_skill_name(candidate) else None -# --------------------------------------------------------------------------- -# LobeHub source adapter -# --------------------------------------------------------------------------- +# --- LobeHub source adapter ------------------------------------------------- class LobeHubSource(SkillSource): """LobeHub agent marketplace (14,500+ system-prompt agents, converted to @@ -358,9 +277,7 @@ class LobeHubSource(SkillSource): def _agents(self) -> Optional[list]: index = self._fetch_index() - if not index: - return None - agents = index.get("agents", index) if isinstance(index, dict) else index + agents = (index.get("agents", index) if isinstance(index, dict) else index) if index else None return agents if isinstance(agents, list) else None @staticmethod @@ -370,14 +287,8 @@ class LobeHubSource(SkillSource): @staticmethod def _agent_meta(agent: dict, name: str, description: str) -> SkillMeta: tags = agent.get("meta", agent).get("tags", []) - return SkillMeta( - name=name, - description=description, - source="lobehub", - identifier=f"lobehub/{name}", - trust_level="community", - tags=tags if isinstance(tags, list) else [], - ) + return SkillMeta(name=name, description=description, source="lobehub", identifier=f"lobehub/{name}", + trust_level="community", tags=tags if isinstance(tags, list) else []) def search(self, query: str, limit: int = 10) -> List[SkillMeta]: agents = self._agents() @@ -403,28 +314,18 @@ class LobeHubSource(SkillSource): agent_data = self._fetch_agent(agent_id) if not agent_data: return None - - return SkillBundle( - name=agent_id, - files={"SKILL.md": self._convert_to_skill_md(agent_data)}, - source="lobehub", - identifier=f"lobehub/{agent_id}", - trust_level="community", - ) + return SkillBundle(name=agent_id, files={"SKILL.md": self._convert_to_skill_md(agent_data)}, source="lobehub", + identifier=f"lobehub/{agent_id}", trust_level="community") def inspect(self, identifier: str) -> Optional[SkillMeta]: agent_id = self._agent_id(identifier) - for agent in self._agents() or []: - if agent.get("identifier") == agent_id: - return self._agent_meta(agent, agent_id, agent.get("meta", agent).get("description", "")) - return None + agent = next((a for a in self._agents() or [] if a.get("identifier") == agent_id), None) + return self._agent_meta(agent, agent_id, agent.get("meta", agent).get("description", "")) if agent else None def _fetch_index(self) -> Optional[Any]: - """Fetch the LobeHub agent index (cached for 1 hour).""" return _memo_json("lobehub_index", lambda: _get_json(self.INDEX_URL, timeout=30)) def _fetch_agent(self, agent_id: str) -> Optional[dict]: - """Fetch a single agent's JSON file.""" return _get_json(f"https://chat-agents.lobehub.com/{agent_id}.json", timeout=15) @staticmethod @@ -435,44 +336,22 @@ class LobeHubSource(SkillSource): title = meta.get("title", identifier) description = meta.get("description", "") tags = meta.get("tags", []) - system_role = agent_data.get("config", {}).get("systemRole", "") - tag_list = tags if isinstance(tags, list) else [] - fm_lines = [ - "---", - f"name: {identifier}", - f"description: {description[:500]}", - "metadata:", - " hermes:", - f" tags: [{', '.join(str(t) for t in tag_list)}]", - " lobehub:", - " source: lobehub", - "---", - ] - - body_lines = [ - f"# {title}", - "", - description, - "", - "## Instructions", - "", - system_role if system_role else "(No system role defined)", - ] - + system_role = agent_data.get("config", {}).get("systemRole", "") + fm_lines = ["---", f"name: {identifier}", f"description: {description[:500]}", "metadata:", " hermes:", + f" tags: [{', '.join(str(t) for t in tag_list)}]", " lobehub:", " source: lobehub", "---"] + body_lines = [f"# {title}", "", description, "", "## Instructions", "", + system_role if system_role else "(No system role defined)"] return "\n".join(fm_lines) + "\n\n" + "\n".join(body_lines) + "\n" -# --------------------------------------------------------------------------- -# browse.sh source adapter -# --------------------------------------------------------------------------- +# --- browse.sh source adapter ----------------------------------------------- class BrowseShSource(SkillSource): """Browserbase's browse.sh catalog of site-specific browser-automation SKILL.md files. - The catalog is ``/api/skills``; content comes from ``/api/skills/{slug}``'s - ``skillMdUrl`` (CDN blob). The catalog's ``sourceUrl`` is a GitHub HTML URL - whose repo is not always public, so it is not used for content. + The catalog is ``/api/skills``; content comes from ``/api/skills/{slug}``'s ``skillMdUrl`` (CDN blob). + The catalog's ``sourceUrl`` is a GitHub HTML URL whose repo is not always public, so it is not used for content. """ SOURCE_ID = "browse-sh" @@ -491,28 +370,17 @@ class BrowseShSource(SkillSource): def _item_to_meta(self, item: Dict) -> Optional[SkillMeta]: slug = item.get("slug", "") name = item.get("name", "") - title = item.get("title", name) - description = item.get("description", title) + description = item.get("description", item.get("title", name)) if not slug or not name: return None if len(description) > 1024: description = description[:1021] + "..." return SkillMeta( - name=name, - description=description, - source="browse-sh", - identifier=f"browse-sh/{slug}", - trust_level="community", - tags=item.get("tags", []), - extra={ - "slug": slug, - "hostname": item.get("hostname", ""), - "category": item.get("category", ""), - "source_url": item.get("sourceUrl", ""), - "recommended_method": item.get("recommendedMethod", ""), - "proxies": item.get("proxies", False), - "install_count": item.get("installCount", 0), - }, + name=name, description=description, source="browse-sh", identifier=f"browse-sh/{slug}", + trust_level="community", tags=item.get("tags", []), + extra={"slug": slug, "hostname": item.get("hostname", ""), "category": item.get("category", ""), + "source_url": item.get("sourceUrl", ""), "recommended_method": item.get("recommendedMethod", ""), + "proxies": item.get("proxies", False), "install_count": item.get("installCount", 0)}, ) def search(self, query: str, limit: int = 10) -> List[SkillMeta]: @@ -524,9 +392,7 @@ class BrowseShSource(SkillSource): def _catalog_item(self, identifier: str) -> Optional[Dict]: slug = self._slug_from_identifier(identifier) - if not slug: - return None - return next((i for i in self._fetch_catalog() if i.get("slug") == slug), None) + return next((i for i in self._fetch_catalog() if i.get("slug") == slug), None) if slug else None def inspect(self, identifier: str) -> Optional[SkillMeta]: item = self._catalog_item(identifier) @@ -537,44 +403,28 @@ class BrowseShSource(SkillSource): if not item: return None slug = item["slug"] - md_url = self._resolve_skill_md_url(slug, item) content = _get_text(md_url, follow_redirects=True) if md_url else None if content is None: return None - meta = self._item_to_meta(item) return SkillBundle( - name=meta.name if meta else slug.split("/")[-1], - files={"SKILL.md": content}, - source="browse-sh", - identifier=identifier, - trust_level="community", - metadata={ - "slug": slug, - "hostname": item.get("hostname", ""), - "source_url": item.get("sourceUrl", ""), - "skill_md_url": md_url, - }, + name=meta.name if meta else slug.split("/")[-1], files={"SKILL.md": content}, source="browse-sh", + identifier=identifier, trust_level="community", + metadata={"slug": slug, "hostname": item.get("hostname", ""), "source_url": item.get("sourceUrl", ""), + "skill_md_url": md_url}, ) def _resolve_skill_md_url(self, slug: str, item: Dict) -> Optional[str]: - """``skillMdUrl`` from ``/api/skills/{slug}``; fallback to a - ``raw.githubusercontent.com`` catalog ``sourceUrl`` when present.""" + """``skillMdUrl`` from ``/api/skills/{slug}``; fallback to a ``raw.githubusercontent.com`` ``sourceUrl``.""" data = _get_json(self.SKILL_DETAIL_URL.format(slug=slug), follow_redirects=True) - if isinstance(data, dict): - md_url = data.get("skillMdUrl") - if isinstance(md_url, str) and md_url.startswith("http"): - return md_url - + md_url = data.get("skillMdUrl") if isinstance(data, dict) else None + if isinstance(md_url, str) and md_url.startswith("http"): + return md_url source_url = item.get("sourceUrl", "") if isinstance(item, dict) else "" from utils import base_url_host_matches - if source_url and base_url_host_matches(source_url, "raw.githubusercontent.com"): - return source_url - return None + return source_url if source_url and base_url_host_matches(source_url, "raw.githubusercontent.com") else None def _slug_from_identifier(self, identifier: str) -> str: """'browse-sh/airbnb.com/search-listings-abc' -> 'airbnb.com/search-listings-abc'.""" - if identifier.startswith("browse-sh/"): - return identifier[len("browse-sh/"):] - return identifier + return identifier[len("browse-sh/"):] if identifier.startswith("browse-sh/") else identifier