diff --git a/hermes_cli/azure_detect.py b/hermes_cli/azure_detect.py index d5e24c2039..179a38f92b 100644 --- a/hermes_cli/azure_detect.py +++ b/hermes_cli/azure_detect.py @@ -20,17 +20,16 @@ from hermes_cli.urllib_security import open_credentialed_url logger = logging.getLogger(__name__) +TokenProvider = Optional[Callable[[], str]] -# Default Azure OpenAI ``api-version`` to probe with. The v1 GA endpoint -# accepts requests without ``api-version`` entirely, so this is only used -# as a fallback for pre-v1 resources that still require it. +# Azure OpenAI ``api-version`` fallbacks for pre-v1 resources; the v1 GA endpoint accepts requests +# without ``api-version`` entirely, so these are only probed second. _AZURE_OPENAI_PROBE_API_VERSIONS = ( "2025-04-01-preview", "2024-10-21", # oldest GA that supports /models ) -# Default Azure Anthropic ``api-version``. Matches the value used by -# ``agent/anthropic_adapter.py`` when building the Anthropic client. +# Matches the value ``agent/anthropic_adapter.py`` uses when building the Anthropic client. _AZURE_ANTHROPIC_API_VERSION = "2025-04-15" @@ -38,37 +37,25 @@ _AZURE_ANTHROPIC_API_VERSION = "2025-04-15" class DetectionResult: """Everything auto-detection could gather from a base URL + API key.""" - #: Detected API transport: ``"chat_completions"``, - #: ``"anthropic_messages"``, or ``None`` when detection failed. + #: ``"chat_completions"``, ``"anthropic_messages"``, or ``None`` when detection failed. api_mode: Optional[str] = None - - #: Deployment / model IDs returned by ``/models`` (best effort). - #: Empty when the endpoint doesn't expose the list with an API key. + #: Deployment / model IDs returned by ``/models`` (best effort; empty when not exposed). models: list[str] = field(default_factory=list) - #: Lowercased host from the base URL (used for display messages). hostname: str = "" - - #: Human-readable reason the detector chose ``api_mode``. Useful - #: for explaining auto-detection to the user in the wizard. + #: Human-readable reason the detector chose ``api_mode`` (shown by the wizard). reason: str = "" - #: ``True`` when ``/models`` returned a valid OpenAI-shaped payload. models_probe_ok: bool = False - - #: ``True`` when the URL was determined to be an Anthropic-style - #: endpoint (from path suffix or live probe). + #: ``True`` when the URL was determined to be Anthropic-style (path suffix or live probe). is_anthropic: bool = False -def _resolve_credential(api_key: Any, - token_provider: Optional[Callable[[], str]] = None, - ) -> tuple[Optional[str], str]: - """Coerce wizard inputs into a (token, mode) pair. +def _resolve_credential(api_key: Any, token_provider: TokenProvider = None) -> tuple[Optional[str], str]: + """Coerce wizard inputs into ``(token_or_None, mode)``. - Returns ``(token_or_None, mode)`` where ``mode`` is: - ``"entra_id"`` when a callable token - provider was supplied — the returned token is a freshly minted bearer JWT, sent ONLY in - ``Authorization: Bearer``. + ``mode`` is ``"entra_id"`` when a callable token provider was supplied (the token is a freshly + minted bearer JWT, sent ONLY in ``Authorization: Bearer``), else ``"api_key"``. """ # Token-provider path (callable wins when both supplied). for provider, label in ((token_provider, "token_provider"), (api_key, "api_key callable")): @@ -79,7 +66,6 @@ def _resolve_credential(api_key: Any, except Exception as exc: logger.debug("azure_detect: %s failed: %s", label, exc) return None, "entra_id" - # API-key path. if isinstance(api_key, str) and api_key: return api_key, "api_key" return None, "api_key" @@ -91,25 +77,18 @@ def _authed_request(url: str, api_key: Any, token_provider, *, method: str = "GE token, mode = _resolve_credential(api_key, token_provider) req = urllib_request.Request(url, method=method, data=data) if token: + # Legacy broad-compat behaviour sends both headers so we land on any Azure resource. In + # entra_id mode send Bearer ONLY — api-key would log a JWT in a slot meant for static keys. if mode != "entra_id": - # Legacy broad-compat behaviour: send both headers so we land on - # any Azure resource regardless of which it accepts. req.add_header("api-key", token) - # Bearer-only in entra_id mode: do NOT also set api-key, which would - # log a JWT in a header slot intended for static keys. req.add_header("Authorization", f"Bearer {token}") req.add_header("User-Agent", "hermes-agent/azure-detect") return req -def _http_get_json(url: str, - api_key: Any, - timeout: float = 6.0, - *, - token_provider: Optional[Callable[[], str]] = None, - ) -> tuple[int, Optional[dict]]: - """GET a URL with the appropriate auth headers. Return - ``(status_code, parsed_json_or_None)``. Never raises.""" +def _http_get_json(url: str, api_key: Any, timeout: float = 6.0, *, + token_provider: TokenProvider = None) -> tuple[int, Optional[dict]]: + """GET with auth headers; return ``(status_code, parsed_json_or_None)``. Never raises.""" req = _authed_request(url, api_key, token_provider) try: with open_credentialed_url(req, timeout=timeout) as resp: @@ -134,166 +113,115 @@ def _strip_trailing_v1(url: str) -> str: def _looks_like_anthropic_path(url: str) -> bool: - """Return True when the URL's path ends in ``/anthropic`` or contains a ``/anthropic/`` segment. - Used by Azure Foundry resources that route Claude traffic through a dedicated path. - """ + """True when the path ends in ``/anthropic`` or contains a ``/anthropic/`` segment (Foundry Claude routes).""" try: - parsed = urlparse(url) - path = (parsed.path or "").lower().rstrip("/") + path = (urlparse(url).path or "").lower().rstrip("/") return path.endswith("/anthropic") or "/anthropic/" in path + "/" except Exception: return False def _extract_model_ids(payload: dict) -> list[str]: - """Extract a list of model IDs from an OpenAI-shaped ``/models`` - response. Returns ``[]`` on any shape mismatch.""" + """Model IDs from an OpenAI-shaped ``/models`` response; ``[]`` on any shape mismatch.""" data = payload.get("data") if isinstance(payload, dict) else None if not isinstance(data, list): return [] ids: list[str] = [] for item in data: - if not isinstance(item, dict): - continue - # OpenAI shape: {"id": "gpt-5.4", "object": "model", ...} - mid = item.get("id") or item.get("model") or item.get("name") - if isinstance(mid, str) and mid: - ids.append(mid) + if isinstance(item, dict): + mid = item.get("id") or item.get("model") or item.get("name") + if isinstance(mid, str) and mid: + ids.append(mid) return ids -def _probe_openai_models(base_url: str, - api_key: Any, - *, - token_provider: Optional[Callable[[], str]] = None, - ) -> tuple[bool, list[str]]: +def _probe_openai_models(base_url: str, api_key: Any, *, token_provider: TokenProvider = None) -> tuple[bool, list[str]]: """Probe ``/models`` for an OpenAI-shaped response.""" base_url = base_url.rstrip("/") - - # Azure OpenAI v1: {resource}.openai.azure.com/openai/v1 — no - # api-version required for GA paths, so probe without first. - # Fallback: explicit api-version for pre-v1 resources - candidates = [f"{base_url}/models"] + [ - f"{base_url}/models?api-version={v}" for v in _AZURE_OPENAI_PROBE_API_VERSIONS - ] + # Azure OpenAI v1 needs no api-version for GA paths, so probe without first; then fall back to + # explicit api-versions for pre-v1 resources. + candidates = [f"{base_url}/models"] + [f"{base_url}/models?api-version={v}" for v in _AZURE_OPENAI_PROBE_API_VERSIONS] for url in candidates: status, body = _http_get_json(url, api_key, token_provider=token_provider) if status == 200 and body is not None: ids = _extract_model_ids(body) if ids: - logger.info( - "azure_detect: /models probe OK at %s (%d models)", - url, len(ids), - ) + logger.info("azure_detect: /models probe OK at %s (%d models)", url, len(ids)) return True, ids - # 200 + empty list still counts as "OpenAI shape, no models - # listed" — let the user proceed with manual entry. + # 200 + empty list still counts as "OpenAI shape, no models listed". if isinstance(body, dict) and "data" in body: return True, [] return False, [] -def _probe_anthropic_messages(base_url: str, - api_key: Any, - *, - token_provider: Optional[Callable[[], str]] = None, - ) -> bool: - """Send a zero-token request to ``/v1/messages`` and check whether the endpoint at least - *recognises* the Anthropic Messages shape (any 4xx that mentions ``messages`` or ``model``, or a - 400 ``invalid_request`` with an Anthropic error shape). Never completes a real chat. +def _probe_anthropic_messages(base_url: str, api_key: Any, *, token_provider: TokenProvider = None) -> bool: + """Zero-token POST to ``/v1/messages``: does the endpoint *recognise* the Anthropic shape? + + Any 4xx mentioning ``messages``/``model``, or an Anthropic-shaped error body, counts. Never + completes a real chat. """ - base = _strip_trailing_v1(base_url) - url = f"{base}/v1/messages?api-version={_AZURE_ANTHROPIC_API_VERSION}" - payload = json.dumps({ - "model": "probe", - "max_tokens": 1, - "messages": [{"role": "user", "content": "ping"}], - }).encode("utf-8") + url = f"{_strip_trailing_v1(base_url)}/v1/messages?api-version={_AZURE_ANTHROPIC_API_VERSION}" + payload = json.dumps({"model": "probe", "max_tokens": 1, "messages": [{"role": "user", "content": "ping"}]}).encode("utf-8") req = _authed_request(url, api_key, token_provider, method="POST", data=payload) req.add_header("anthropic-version", "2023-06-01") req.add_header("content-type", "application/json") try: with open_credentialed_url(req, timeout=6.0) as resp: - # Should never 200 — "probe" isn't a real deployment. But - # if it does, the endpoint definitely speaks Anthropic. + # Should never 200 — "probe" isn't a real deployment — but if it does, it speaks Anthropic. return resp.status < 500 except HTTPError as exc: - # 4xx with an Anthropic-shaped error body = Anthropic endpoint. try: - body = exc.read().decode("utf-8", errors="replace") - lowered = body.lower() + lowered = exc.read().decode("utf-8", errors="replace").lower() if "anthropic" in lowered or '"type"' in lowered and '"error"' in lowered: return True - # Pre-Azure-v1 Azure Foundry returns a plain 404 for - # Anthropic-style calls on non-Anthropic deployments. A - # 400 "model not found" IS Anthropic though. - if exc.code == 400 and ("messages" in lowered or "model" in lowered): - return True - return False + # Pre-Azure-v1 Foundry returns a plain 404 for Anthropic-style calls on non-Anthropic + # deployments. A 400 "model not found" IS Anthropic though. + return exc.code == 400 and ("messages" in lowered or "model" in lowered) except Exception: return False - except (URLError, TimeoutError, OSError): - return False - except Exception: # pragma: no cover + except Exception: # URLError, TimeoutError, OSError, anything else return False -def detect(base_url: str, - api_key: Any = "", - *, - token_provider: Optional[Callable[[], str]] = None, - ) -> DetectionResult: - """Inspect an Azure endpoint and describe its transport + models. - - Call this from the wizard before asking the user to pick an API mode manually. The caller should - treat the returned :class:`DetectionResult` as *advisory* — if ``api_mode`` is None, fall back - to asking the user. +def detect(base_url: str, api_key: Any = "", *, token_provider: TokenProvider = None) -> DetectionResult: + """Inspect an Azure endpoint and describe its transport + models (advisory — None api_mode means ask the user). ``api_key`` may be a string (legacy API-key auth — sends both ``api-key:`` and ``Authorization: Bearer``) or a callable returning a bearer JWT (Entra ID auth — sends ONLY ``Authorization: - Bearer``). ``token_provider`` is an alternative explicit name for the callable form; if both are - supplied the callable wins. + Bearer``). ``token_provider`` is an explicit name for the callable form; the callable wins. """ result = DetectionResult() - try: - parsed = urlparse(base_url) - result.hostname = (parsed.hostname or "").lower() + result.hostname = (urlparse(base_url).hostname or "").lower() except Exception: result.hostname = "" - # 1. Path sniff. Azure Foundry exposes Anthropic-style deployments - # under a dedicated ``/anthropic`` path. + # 1. Path sniff: Foundry exposes Anthropic-style deployments under a dedicated /anthropic path. if _looks_like_anthropic_path(base_url): result.is_anthropic = True result.api_mode = "anthropic_messages" result.reason = "URL path ends in /anthropic → Anthropic Messages API" return result - # 2. Try the OpenAI-style /models probe. If this works, the - # endpoint definitely speaks OpenAI wire. + # 2. OpenAI-style /models probe — success means the endpoint definitely speaks OpenAI wire. ok, models = _probe_openai_models(base_url, api_key, token_provider=token_provider) if ok: result.models_probe_ok = True result.models = models result.api_mode = "chat_completions" result.reason = ( - f"GET /models returned {len(models)} model(s) — OpenAI-style endpoint" - if models + f"GET /models returned {len(models)} model(s) — OpenAI-style endpoint" if models else "GET /models returned an OpenAI-shaped empty list — OpenAI-style endpoint" ) return result - # 3. Fallback: probe the Anthropic Messages shape. Slower and more - # intrusive than /models, so only run it when the OpenAI probe - # failed. + # 3. Anthropic Messages probe — slower and more intrusive, so only when /models failed. if _probe_anthropic_messages(base_url, api_key, token_provider=token_provider): result.is_anthropic = True result.api_mode = "anthropic_messages" result.reason = "Endpoint accepts Anthropic Messages shape" return result - # Nothing matched. Caller falls back to manual selection. result.reason = ( "Could not probe endpoint (private network, missing model list, or " "non-standard path) — falling back to manual API-mode selection" @@ -301,40 +229,27 @@ def detect(base_url: str, return result -def lookup_context_length(model: str, - base_url: str, - api_key: Any = "", - *, - token_provider: Optional[Callable[[], str]] = None, - ) -> Optional[int]: - """Thin wrapper around :func:`agent.model_metadata.get_model_context_length` that returns ``None`` - when only the fallback default (128k) would fire, so the wizard can distinguish "we actually - know this" from "we guessed. +def lookup_context_length(model: str, base_url: str, api_key: Any = "", *, + token_provider: TokenProvider = None) -> Optional[int]: + """``get_model_context_length`` that returns None when only the fallback default would fire, so + the wizard can distinguish "we actually know this" from "we guessed". """ model_id = str(model or "").strip() if not model_id: return None try: - from agent.model_metadata import ( - DEFAULT_FALLBACK_CONTEXT, - get_model_context_length, - ) + from agent.model_metadata import DEFAULT_FALLBACK_CONTEXT, get_model_context_length except Exception: return None - # Resolve the credential once. For Entra mode this calls the token - # provider; for legacy api_key this is a no-op string pass-through. + # Resolve the credential once: Entra mode calls the provider; api_key is a string pass-through. token, _mode = _resolve_credential(api_key, token_provider) - try: n = get_model_context_length(model_id, base_url=base_url, api_key=token or "") except Exception as exc: logger.debug("azure_detect: context length lookup failed: %s", exc) return None - - if isinstance(n, int) and n > 0 and n != DEFAULT_FALLBACK_CONTEXT: - return n - return None + return n if isinstance(n, int) and n > 0 and n != DEFAULT_FALLBACK_CONTEXT else None __all__ = ["DetectionResult", "detect", "lookup_context_length"]