"""Skills Hub GitHub adapter: API auth, tap providers, and the Contents/Trees source.""" import json import logging import subprocess import time from pathlib import Path from typing import Dict, List, Optional, Tuple, Union from urllib.parse import quote import httpx from hermes_cli._subprocess_compat import windows_hide_flags from tools.skills_guard import TRUSTED_REPOS from tools.skills_hub_models import ( SkillBundle, SkillMeta, SkillSource, _dedupe_by_trust, _hermes_tags, _matches_query, _parse_frontmatter, _referenced_support_paths, _skill_meta_to_dict, _validate_bundle_rel_path, hub, ) logger = logging.getLogger("tools.skills_hub") # Maps a GitHub tap repo (owner/repo) to the provider label used by the # docs-site catalog (website/scripts/extract-skills.py::GITHUB_TAP_LABELS). # The runtime index collapses every tap into source="github"; the label in # ``extra.provider`` keeps per-tap identity searchable/filterable without # disturbing the dedup / floor / index-skip logic keyed on the bare source id. GITHUB_TAP_PROVIDERS = { "openai/skills": "OpenAI", "anthropics/skills": "Anthropic", "huggingface/skills": "HuggingFace", "nvidia/skills": "NVIDIA", "voltagent/awesome-agent-skills": "VoltAgent", "garrytan/gstack": "gstack", "minimax-ai/cli": "MiniMax", } # Accepted ``--source`` provider filters (lowercased). Not real source ids — # they narrow merged results to GitHub-tap skills carrying that ``extra.provider``. _PROVIDER_FILTER_VALUES = frozenset(v.lower() for v in GITHUB_TAP_PROVIDERS.values()) def github_provider_for(repo: str) -> Optional[str]: """Provider label for an ``owner/repo`` tap (case-insensitive), or None.""" if not repo: return None return GITHUB_TAP_PROVIDERS.get(repo.strip().lower()) def _filter_results_by_provider(results: List[SkillMeta], provider: str) -> List[SkillMeta]: """Keep only results whose ``extra.provider`` matches ``provider``. An explicit provider filter (``--source nvidia``) narrows to exactly that provider — the official catalog is NOT injected the way unfiltered browse does. """ want = provider.strip().lower() return [r for r in results if str((r.extra or {}).get("provider", "")).lower() == want] def _is_rate_limit_response(resp: httpx.Response) -> bool: """403 with exhausted quota, or any 429.""" return resp.status_code == 429 or ( resp.status_code == 403 and resp.headers.get("X-RateLimit-Remaining", "") == "0" ) # --------------------------------------------------------------------------- # GitHub Authentication # --------------------------------------------------------------------------- class GitHubAuth: """GitHub API authentication, tried in priority order: 1. GITHUB_TOKEN / GH_TOKEN (PAT) 2. `gh auth token` (gh CLI) 3. GitHub App JWT + installation token 4. Unauthenticated (60 req/hr, public repos only) """ def __init__(self): self._cached_token: Optional[str] = None self._cached_method: Optional[str] = None self._app_token_expiry: float = 0 def get_headers(self) -> Dict[str, str]: token = self._resolve_token() headers = {"Accept": "application/vnd.github.v3+json"} if token: headers["Authorization"] = f"token {token}" return headers def is_authenticated(self) -> bool: return self._resolve_token() is not None def auth_method(self) -> str: """'pat', 'gh-cli', 'github-app', or 'anonymous'.""" self._resolve_token() return self._cached_method or "anonymous" def _resolve_token(self) -> Optional[str]: if self._cached_token and ( self._cached_method != "github-app" or time.time() < self._app_token_expiry ): return self._cached_token # Profile-scoped secret lookup (multiplexed gateway safe). from agent.secret_scope import get_secret token = get_secret("GITHUB_TOKEN") or get_secret("GH_TOKEN") if token: self._cached_token, self._cached_method = token, "pat" return token token = self._try_gh_cli() if token: self._cached_token, self._cached_method = token, "gh-cli" return token token = self._try_github_app() if token: self._cached_token, self._cached_method = token, "github-app" self._app_token_expiry = time.time() + 3500 # ~58 min (tokens last 1 hour) return token self._cached_method = "anonymous" return None def _try_gh_cli(self) -> Optional[str]: try: result = subprocess.run( ["gh", "auth", "token"], capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5, stdin=subprocess.DEVNULL, creationflags=windows_hide_flags(), ) if result.returncode == 0 and result.stdout.strip(): return result.stdout.strip() except (FileNotFoundError, subprocess.TimeoutExpired) as e: logger.debug("gh CLI token lookup failed: %s", e) return None def _try_github_app(self) -> Optional[str]: from agent.secret_scope import get_secret app_id = get_secret("GITHUB_APP_ID") key_path = get_secret("GITHUB_APP_PRIVATE_KEY_PATH") installation_id = get_secret("GITHUB_APP_INSTALLATION_ID") if not all([app_id, key_path, installation_id]): return None try: import jwt # PyJWT except ImportError: logger.debug("PyJWT not installed, skipping GitHub App auth") return None try: key_file = Path(key_path) if not key_file.exists(): return None private_key = key_file.read_text(encoding="utf-8") now = int(time.time()) payload = {"iat": now - 60, "exp": now + (10 * 60), "iss": app_id} encoded_jwt = jwt.encode(payload, private_key, algorithm="RS256") resp = httpx.post( f"https://api.github.com/app/installations/{installation_id}/access_tokens", headers={ "Authorization": f"Bearer {encoded_jwt}", "Accept": "application/vnd.github.v3+json", }, timeout=10, ) if resp.status_code == 201: return resp.json().get("token") except Exception as e: logger.debug("GitHub App auth failed: %s", e) return None # --------------------------------------------------------------------------- # GitHub source adapter # --------------------------------------------------------------------------- def _skip_bundle_file(rel_path: str) -> bool: """Dotfiles, bytecode and __pycache__ never ship in a bundle.""" base = rel_path.rsplit("/", 1)[-1] return base.startswith(".") or base.endswith(".pyc") or "__pycache__" in rel_path.split("/") class GitHubSource(SkillSource): """Fetch skills from GitHub repos via the Contents API.""" DEFAULT_TAPS = [ # openai/skills keeps its content under skills/.curated/ and # skills/.system/; _list_skills_in_repo skips "."/"_" directories, # so both entries point at the inner paths directly. {"repo": "openai/skills", "path": "skills/.curated/"}, {"repo": "openai/skills", "path": "skills/.system/"}, {"repo": "anthropics/skills", "path": "skills/"}, {"repo": "huggingface/skills", "path": "skills/"}, # NVIDIA-verified skills (CUDA-X, NeMo, cuOpt, ...), each with a # signed skill.oms.sig + governance card; `trusted` via # tools/skills_guard.py::TRUSTED_REPOS. {"repo": "NVIDIA/skills", "path": "skills/"}, {"repo": "garrytan/gstack", "path": ""}, ] _parse_frontmatter_quick = staticmethod(_parse_frontmatter) def __init__(self, auth: GitHubAuth, extra_taps: Optional[List[Dict]] = None): self.auth = auth self.taps = list(self.DEFAULT_TAPS) if extra_taps: self.taps.extend(extra_taps) # Per-instance repo -> (default_branch, tree_entries); lives for one # search/install flow so repeated tree lookups cost no API calls. self._tree_cache: Dict[str, Tuple[str, List[dict]]] = {} self._tree_revisions: Dict[str, str] = {} # repo -> skills.sh.json grouping map; None = fetched, no sidecar. self._skillsh_groupings: Dict[str, Optional[Dict[str, str]]] = {} self._rate_limited: bool = False def source_id(self) -> str: return "github" @property def is_rate_limited(self) -> bool: """Whether GitHub API rate limit was hit during operations.""" return self._rate_limited def trust_level_for(self, identifier: str) -> str: # identifier format: "owner/repo/path/to/skill" parts = identifier.split("/", 2) if len(parts) >= 2 and f"{parts[0]}/{parts[1]}" in TRUSTED_REPOS: return "trusted" return "community" def search(self, query: str, limit: int = 10) -> List[SkillMeta]: """Substring-match all taps; dedupe by identifier preferring higher trust.""" results: List[SkillMeta] = [] query_lower = query.lower() for tap in self.taps: try: for skill in self._list_skills_in_repo(tap["repo"], tap.get("path", "")): if _matches_query(query_lower, skill.name, skill.description, skill.tags): results.append(skill) except Exception as e: logger.debug("Failed to search %s: %s", tap['repo'], e) continue return _dedupe_by_trust(results)[:limit] def fetch(self, identifier: str) -> Optional[SkillBundle]: """Download a skill; identifier format: "owner/repo/path/to/skill-dir".""" parts = identifier.split("/", 2) if len(parts) < 3: return None repo = f"{parts[0]}/{parts[1]}" skill_path = parts[2] skill_dir = skill_path.rstrip("/") # Resolve the tree FIRST so every byte fetch — SKILL.md included — is # pinned to the same revision; an unpinned /contents fetch floats to # HEAD and can serve bytes newer than the tree the paths were # validated against (TOCTOU). Idempotent + cached. tree = self._get_repo_tree(repo) pinned_ref = self._tree_revisions.get(repo) skill_md = self._fetch_file_content(repo, f"{skill_dir}/SKILL.md", ref=pinned_ref) if skill_md is None: return None referenced = _referenced_support_paths(skill_md) if referenced is None: return None files: Dict[str, Union[str, bytes]] = {"SKILL.md": skill_md} if tree is not None: branch, entries = tree if not self._collect_tree_files(repo, skill_dir, entries, pinned_ref, referenced, files): return None revision = pinned_ref or branch else: for rel_path in referenced: content = self._fetch_file_bytes(repo, f"{skill_dir}/{rel_path}") if content is None: logger.warning("Failed to fetch referenced skill support " "file; continuing without it: %s", rel_path) continue files[rel_path] = content revision = "" return SkillBundle( name=skill_dir.split("/")[-1], files=files, source="github", identifier=identifier, trust_level=self.trust_level_for(identifier), metadata={ "source_url": ( f"https://github.com/{repo}/tree/{revision}/{skill_path}" if revision else f"https://github.com/{repo}/{skill_path}" ), "source_revision": revision, }, ) def _collect_tree_files( self, repo: str, skill_path: str, entries: List[dict], ref: Optional[str], referenced: set, files: Dict[str, Union[str, bytes]], ) -> bool: """Download the FULL skill directory from the pinned tree into ``files``. Link-driven fetching silently dropped support files under non-canonical dirs (``reference/``, ``agents/``, root LICENSE); everything still goes through quarantine + scan, and the scanner sees MORE this way. Returns False (bundle rejected) on an unsafe path or a SKILL.md-linked path that exists in the tree as a symlink/non-blob — that shape is an escape attempt. A linked path that is simply absent is a dangling link (repo-only dev tool, prose over-match): warn and install without it. """ prefix = f"{skill_path}/" symlinked: set = set() for item in entries: item_path = item.get("path", "") if not item_path.startswith(prefix): continue rel_path = item_path[len(prefix):] if item.get("type") != "blob" or item.get("mode") == "120000": symlinked.add(rel_path) continue if rel_path == "SKILL.md" or _skip_bundle_file(rel_path): continue try: rel_path = _validate_bundle_rel_path(rel_path) except ValueError: logger.warning("Rejected unsafe file path in skill bundle: %s", item_path) return False content = self._fetch_file_bytes(repo, item_path, ref=ref) if content is None: logger.warning("Failed to fetch referenced skill support " "file; continuing without it: %s", item_path) continue files[rel_path] = content for rel_path in sorted(referenced): if rel_path in symlinked: logger.warning( "Rejected non-regular referenced file in skill " "bundle: %s%s", prefix, rel_path, ) return False if rel_path not in files: logger.warning( "Referenced skill support file is missing; " "continuing without it: %s%s", prefix, rel_path, ) return True def inspect(self, identifier: str) -> Optional[SkillMeta]: """Fetch just the SKILL.md metadata for preview.""" parts = identifier.split("/", 2) if len(parts) < 3: return None repo = f"{parts[0]}/{parts[1]}" skill_path = parts[2].rstrip("/") content = self._fetch_file_content(repo, f"{skill_path}/SKILL.md") if not content: return None fm = _parse_frontmatter(content) tags = _hermes_tags(fm) if not tags: raw_tags = fm.get("tags", []) tags = raw_tags if isinstance(raw_tags, list) else [] provider = github_provider_for(repo) return SkillMeta( name=fm.get("name", skill_path.split("/")[-1]), description=str(fm.get("description", "")), source="github", identifier=identifier, trust_level=self.trust_level_for(identifier), repo=repo, path=skill_path, tags=[str(t) for t in tags], extra={"provider": provider} if provider else {}, ) # -- Internal helpers -- def _list_skills_in_repo(self, repo: str, path: str) -> List[SkillMeta]: """List skill directories in a GitHub repo path, using cached index.""" cache_key = f"{repo}_{path}".replace("/", "_").replace(" ", "_") cached = self._read_cache(cache_key) if cached is not None: return [SkillMeta(**s) for s in cached] url = f"https://api.github.com/repos/{repo}/contents/{path.rstrip('/')}" resp = self._github_get(url) if resp is None or resp.status_code != 200: return [] entries = resp.json() if not isinstance(entries, list): return [] skills: List[SkillMeta] = [] groupings = self._get_skillsh_groupings(repo) prefix = path.rstrip("/") for entry in entries: if entry.get("type") != "dir": continue dir_name = entry["name"] if dir_name.startswith((".", "_")): continue meta = self.inspect(f"{repo}/{prefix}/{dir_name}" if prefix else f"{repo}/{dir_name}") if meta: if groupings: category = groupings.get(meta.name) or groupings.get(dir_name) if category: meta.extra["category"] = category skills.append(meta) self._write_cache(cache_key, [_skill_meta_to_dict(s) for s in skills]) return skills def _get_repo_tree(self, repo: str) -> Optional[Tuple[str, List[dict]]]: """Cached ``(default_branch, tree_entries)`` for a repo, or None. One install may need the tree several times; caching saves the ``GET /repos/{repo}`` + ``GET .../git/trees/{branch}`` pair each time (~12 of the 60/hr unauthenticated budget before). """ if repo in self._tree_cache: return self._tree_cache[repo] headers = self.auth.get_headers() try: resp = httpx.get( f"https://api.github.com/repos/{repo}", headers=headers, timeout=15, follow_redirects=True, ) if resp.status_code != 200: self._check_rate_limit_response(resp) return None default_branch = resp.json().get("default_branch", "main") except (httpx.HTTPError, ValueError): return None try: resp = httpx.get( f"https://api.github.com/repos/{repo}/git/trees/{default_branch}", params={"recursive": "1"}, headers=headers, timeout=30, follow_redirects=True, ) if resp.status_code != 200: self._check_rate_limit_response(resp) return None tree_data = resp.json() if tree_data.get("truncated"): logger.debug("Git tree truncated for %s, cannot cache", repo) return None except (httpx.HTTPError, ValueError): return None entries = tree_data.get("tree", []) revision = tree_data.get("sha") if isinstance(revision, str) and revision: self._tree_revisions[repo] = revision self._tree_cache[repo] = (default_branch, entries) return (default_branch, entries) def _check_rate_limit_response(self, resp: httpx.Response) -> None: """Flag the instance as rate-limited when GitHub returns 403 + exhausted quota.""" if _is_rate_limit_response(resp): self._rate_limited = True logger.warning( "GitHub API rate limit exhausted (unauthenticated: 60 req/hr). " "Set GITHUB_TOKEN or install the gh CLI to raise the limit to 5,000/hr." ) def _github_get( self, url: str, *, params: Optional[Dict] = None, headers: Optional[Dict] = None, timeout: float = 15.0, max_retries: int = 3, ) -> Optional[httpx.Response]: """GET against the GitHub API with retry/backoff on transient failures. Returns the final response (caller inspects status) or None when every attempt raised a transport error. Retries rate-limit 403/429 (waiting until ``Retry-After``/``X-RateLimit-Reset`` when present, capped 60s — one shared limit zeroes every GitHub tap at once during an index build), 5xx, and transport errors with exponential backoff. Terminal rate-limit exhaustion flags the instance so an index build fails loud instead of silently shipping zero GitHub skills. """ hdrs = headers if headers is not None else self.auth.get_headers() backoff = 1.0 last_resp: Optional[httpx.Response] = None for attempt in range(max_retries): last_attempt = attempt >= max_retries - 1 try: resp = httpx.get( url, params=params, headers=hdrs, timeout=timeout, follow_redirects=True, ) except httpx.HTTPError as e: logger.debug("GitHub GET %s failed (attempt %d/%d): %s", url, attempt + 1, max_retries, e) if last_attempt: return None time.sleep(backoff) backoff = min(backoff * 2, 30.0) continue last_resp = resp if resp.status_code == 200: return resp if resp.status_code in (403, 429): if not _is_rate_limit_response(resp) or last_attempt: self._check_rate_limit_response(resp) return resp wait = backoff reset = resp.headers.get("X-RateLimit-Reset", "") retry_after = resp.headers.get("Retry-After", "") if retry_after.isdigit(): wait = min(float(retry_after), 60.0) elif reset.isdigit(): delta = float(reset) - time.time() if 0 < delta <= 60.0: wait = delta logger.debug( "GitHub rate limited on %s, waiting %.1fs (attempt %d/%d)", url, wait, attempt + 1, max_retries, ) time.sleep(wait) backoff = min(backoff * 2, 30.0) continue if 500 <= resp.status_code < 600 and not last_attempt: time.sleep(backoff) backoff = min(backoff * 2, 30.0) continue return resp return last_resp def _find_skill_in_repo_tree(self, repo: str, skill_name: str) -> Optional[str]: """Locate ``/SKILL.md`` anywhere in the repo tree (one API call). Returns the full identifier (``repo/path/to/skill``) or None. """ cached = self._get_repo_tree(repo) if cached is None: return None _default_branch, tree_entries = cached skill_md_suffix = f"/{skill_name}/SKILL.md" for entry in tree_entries: if entry.get("type") != "blob": continue path = entry.get("path", "") if path.endswith(skill_md_suffix) or path == f"{skill_name}/SKILL.md": return f"{repo}/{path[: -len('/SKILL.md')]}" return None def _fetch_file_content( self, repo: str, path: str, ref: Optional[str] = None ) -> Optional[str]: """Fetch a single text file from GitHub (None on miss or non-UTF-8).""" content = self._fetch_file_bytes(repo, path, ref=ref) if content is None: return None try: return content.decode("utf-8") except UnicodeDecodeError: return None def _fetch_file_bytes( self, repo: str, path: str, ref: Optional[str] = None ) -> Optional[bytes]: """Fetch exact file bytes. ``ref`` pins to a tree SHA (see ``fetch`` on the TOCTOU); None keeps the legacy unpinned behavior.""" encoded_path = quote(path, safe="/") url = f"https://api.github.com/repos/{repo}/contents/{encoded_path}" resp = self._github_get( url, params={"ref": ref} if ref else None, headers={**self.auth.get_headers(), "Accept": "application/vnd.github.v3.raw"}, ) if resp is not None and resp.status_code == 200: return resp.content return None def _get_skillsh_groupings(self, repo: str) -> Optional[Dict[str, str]]: """Repo-root ``skills.sh.json`` groupings flattened to ``{skill_name: title}``. ``skills.sh.json`` is a cross-ecosystem standard (``$schema: https://skills.sh/schemas/skills.sh.schema.json``); any tap shipping it gets category pills for free. None when absent/unparsable. Cached per repo on the instance. """ if repo in self._skillsh_groupings: return self._skillsh_groupings[repo] content = self._fetch_file_content(repo, "skills.sh.json") groupings = self._parse_skillsh_groupings(content) if content else None self._skillsh_groupings[repo] = groupings return groupings @staticmethod def _parse_skillsh_groupings(content: str) -> Optional[Dict[str, str]]: """Flatten ``{"groupings": [{"title", "skills": [...]}]}``; None if not usable.""" try: data = json.loads(content) except (json.JSONDecodeError, TypeError): return None if not isinstance(data, dict): return None groupings = data.get("groupings") if not isinstance(groupings, list): return None mapping: Dict[str, str] = {} for group in groupings: if not isinstance(group, dict): continue title = group.get("title") members = group.get("skills") if not isinstance(title, str) or not isinstance(members, list): continue for member in members: if isinstance(member, str) and member: mapping.setdefault(member, title) # first grouping wins return mapping def _read_cache(self, key: str) -> Optional[list]: return hub()._read_index_cache(key) def _write_cache(self, key: str, data: list) -> None: hub()._write_index_cache(key, data)