diff --git a/agent/agent_init.py b/agent/agent_init.py index 33956ff539..dfbb4cae60 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -372,8 +372,9 @@ _EXPLICIT_API_MODES = { def _resolve_api_mode(agent, api_mode, provider_name, base_url): """Set ``agent.api_mode`` (and provider rewrites) — ordered ladder, first match wins.""" + from hermes_cli.providers import is_actual_route host, url = agent._base_url_hostname, agent._base_url_lower - if agent.provider == "actual": + if is_actual_route(agent.provider, base_url): agent.api_mode = "chat_completions" elif api_mode in _EXPLICIT_API_MODES: agent.api_mode = api_mode @@ -418,6 +419,7 @@ def _resolve_api_mode(agent, api_mode, provider_name, base_url): def _finalize_routing(agent, api_mode, credential_pool): + from hermes_cli.providers import is_actual_route # Credential-pool validation runs AFTER provider auto-detection so a pool scoped to # "anthropic" isn't rejected for provider=None + anthropic.com URL. # Regression from #63048 which placed this check before the URL-based auto-detection block above (fixed @@ -470,6 +472,7 @@ def _finalize_routing(agent, api_mode, credential_pool): # upgrade for Azure (openai.azure.com), even though it looks OpenAI-compatible. api_mode is None and agent.api_mode == "chat_completions" + and not is_actual_route(agent.provider, agent.base_url) and agent.provider != "copilot-acp" and not _base_lower.startswith(("acp://", "acp+tcp://")) and not agent._is_azure_openai_url() @@ -2243,6 +2246,10 @@ def init_agent( agent.skip_background_review = bool(skip_background_review) agent.log_prefix = f"{log_prefix} " if log_prefix else "" # Effective base URL for feature detection (prompt caching, reasoning, etc.) + from hermes_cli.providers import is_actual_route + if is_actual_route(provider, base_url): + from hermes_cli.auth import normalize_actual_base_url + base_url = normalize_actual_base_url(base_url) agent.base_url = base_url or "" provider_name = provider.strip().lower() if isinstance(provider, str) and provider.strip() else None agent.provider = provider_name or "" diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index b5de714299..da3ad84ef1 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -889,7 +889,8 @@ def _apply_primary_runtime_fields(agent, rt: Dict[str, Any]) -> None: agent.provider = rt["provider"] agent.requested_provider = rt.get("requested_provider", agent.provider) agent.base_url = rt["base_url"] # setter updates _base_url_lower - agent.api_mode = rt["api_mode"] + from hermes_cli.providers import is_actual_route + agent.api_mode = "chat_completions" if is_actual_route(agent.provider, agent.base_url) else rt["api_mode"] if hasattr(agent, "_transport_cache"): agent._transport_cache.clear() agent.api_key = rt["api_key"] @@ -1841,7 +1842,7 @@ def _restore_switch_snapshot(agent, snapshot: Dict[str, Any]) -> None: def _resolve_switch_destination(agent, new_model, new_provider, base_url, api_mode, capabilities, old_norm, new_norm): """Resolve ``(api_mode, base_url, destination_capabilities)`` for the switch target.""" - from hermes_cli.providers import determine_api_mode + from hermes_cli.providers import determine_api_mode, is_actual_route from agent.native_compaction import resolve_native_compaction_capabilities from hermes_cli.models import opencode_provider_family # Pass model so dual-wire providers (Nous Portal anthropic/* -> Messages) resolve correctly. @@ -1855,6 +1856,11 @@ def _resolve_switch_destination(agent, new_model, new_provider, base_url, api_mo effective_base_url = base_url if not effective_base_url and old_norm == new_norm: effective_base_url = getattr(agent, "base_url", "") + if is_actual_route(new_provider, effective_base_url): + api_mode = "chat_completions" + if effective_base_url: + from hermes_cli.auth import normalize_actual_base_url + base_url = normalize_actual_base_url(effective_base_url) destination_capabilities = ( dict(capabilities) if isinstance(capabilities, dict) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 2f643b6710..28d8cff386 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -936,6 +936,9 @@ def _to_openai_base_url(base_url: str) -> str: without it). Anthropic-only gateways keep their path. """ url = str(base_url or "").strip().rstrip("/") + if base_url_hostname(url) == "api.actual.inc": + from hermes_cli.auth import normalize_actual_base_url + return normalize_actual_base_url(url) if url.endswith("/anthropic"): if base_url_host_matches(url, "open.bigmodel.cn") or base_url_host_matches(url, "api.z.ai"): rewritten = url[: -len("/anthropic")] + "/coding/paas/v4" @@ -4469,12 +4472,32 @@ def _log_once_debug(seen: set, key: Any, msg: str, *args: Any) -> None: logger.debug(msg, *args) +def _is_actual_auxiliary_route(req: _ResolveRequest, base_url: str) -> bool: + from hermes_cli.auth import normalize_actual_base_url + from hermes_cli.providers import is_actual_route + from hermes_cli.route_identity import normalize_route_base_url + + if is_actual_route(req.provider, base_url): + return True + runtime = _normalize_main_runtime(req.main_runtime) + return bool( + base_url + and is_actual_route(runtime.get("provider", ""), runtime.get("base_url", "")) + and normalize_route_base_url(normalize_actual_base_url(base_url)) + == normalize_route_base_url( + normalize_actual_base_url(runtime.get("base_url", "")) + ) + ) + + def _wrap_transport(req: _ResolveRequest, client_obj: Any, final_model_str: str, base_url_str: str = "", api_key_str: str = ""): """Wrap a plain OpenAI client in the right transport adapter; specialized wrappers pass through. Codex (Responses API): explicit ``api_mode=codex_responses``, else — with no explicit api_mode — api.openai.com + codex model. Anthropic (Messages): ``api_mode=anthropic_messages``, any ``/anthropic`` suffix, ``api.kimi.com/coding``, or ``api.anthropic.com``.""" + if _is_actual_auxiliary_route(req, base_url_str): + return client_obj._real_client if isinstance(client_obj, CodexAuxiliaryClient) else client_obj needs_codex = not (isinstance(client_obj, CodexAuxiliaryClient) or req.raw_codex) and ( req.api_mode == "codex_responses" or (not req.api_mode and base_url_hostname(base_url_str) == "api.openai.com" @@ -4616,6 +4639,9 @@ def _resolve_custom_branch(req: _ResolveRequest) -> _ResolveResult: if _main_base and _main_key: custom_base, custom_key = _main_base, _main_key if custom_base and custom_key: + if _is_actual_auxiliary_route(req, custom_base): + from hermes_cli.auth import normalize_actual_base_url + custom_base = normalize_actual_base_url(custom_base) final_model = _normalize_resolved_model( model or (main_runtime.get("model") if main_runtime else None) or "gpt-4o-mini", provider, ) @@ -4678,8 +4704,12 @@ def _resolve_named_custom_branch(req: _ResolveRequest) -> Optional[_ResolveResul logger.warning("resolve_provider_client: named custom provider %r has no resolvable " "api_key — request will be sent with placeholder no-key-required " "and will 401 on auth-required endpoints", custom_entry.get("name") or provider) - # Explicit per-task api_mode override wins over the provider entry's. + # Actual's wire protocol takes precedence over persisted task/provider modes. entry_api_mode = (req.api_mode or custom_entry.get("api_mode") or "").strip() + if _is_actual_auxiliary_route(req, custom_base): + from hermes_cli.auth import normalize_actual_base_url + custom_base = normalize_actual_base_url(custom_base) + entry_api_mode = "chat_completions" if not custom_base: logger.warning("resolve_provider_client: named custom provider %r has no base_url", provider) return None, None @@ -4766,7 +4796,7 @@ def _resolve_api_key_branch(req: _ResolveRequest, pconfig: Any, resolve_creds: C return None, None base_url = _to_openai_base_url(raw_base_url) # Explicit base_url override: a fallback_model/custom_providers entry pointing a built-in name elsewhere. - if req.explicit_base_url: + if req.explicit_base_url and provider != "actual": base_url = _to_openai_base_url(req.explicit_base_url.strip().rstrip("/")) final_model = _normalize_resolved_model(req.model or _get_aux_model_for_provider(provider), provider) if provider == "gemini": diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index b9cf6a1d30..2eeb6c89e9 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -1872,7 +1872,10 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool logger.warning("Could not normalize fallback model %r for provider %r: %s", fb_model, fb_provider, _norm_err) fb_base_url = str(fb_client.base_url) - if not fb_api_mode_explicit and fb_api_mode == "chat_completions": + from hermes_cli.providers import is_actual_route + if is_actual_route(fb_provider, fb_base_url): + fb_api_mode = "chat_completions" + elif not fb_api_mode_explicit and fb_api_mode == "chat_completions": fb_api_mode = _fallback_api_mode_resolved(agent, fb_provider, fb_model, fb_base_url) old_model, old_provider, old_base_url = agent.model, agent.provider, agent.base_url diff --git a/agent/client_lifecycle.py b/agent/client_lifecycle.py index 071b6c9855..1563dcc610 100644 --- a/agent/client_lifecycle.py +++ b/agent/client_lifecycle.py @@ -915,6 +915,13 @@ class ClientLifecycleMixin: def _swap_credential(self, entry) -> None: runtime_key = getattr(entry, "runtime_api_key", None) or getattr(entry, "access_token", "") runtime_base = getattr(entry, "runtime_base_url", None) or getattr(entry, "base_url", None) or self.base_url + from hermes_cli.providers import is_actual_route + if is_actual_route(getattr(self, "provider", ""), runtime_base): + from hermes_cli.auth import normalize_actual_base_url + runtime_base = normalize_actual_base_url(runtime_base) + self.api_mode = "chat_completions" + if hasattr(self, "_transport_cache"): + self._transport_cache.clear() self._credential_pool_entry_id = getattr(entry, "id", None) from hermes_cli.route_identity import normalize_route_base_url route_changed = normalize_route_base_url(self.base_url) != normalize_route_base_url(runtime_base) diff --git a/agent/transports/codex.py b/agent/transports/codex.py index 121d348edf..a7c1bd0ab4 100644 --- a/agent/transports/codex.py +++ b/agent/transports/codex.py @@ -11,7 +11,7 @@ import re from typing import Any, Callable, Optional from agent.reasoning_effort import ( - ACTUAL_RELAY_EFFORTS, CODEX_ASTRA_EFFORTS, CODEX_LEGACY_EFFORTS, + CODEX_ASTRA_EFFORTS, CODEX_LEGACY_EFFORTS, XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort, is_astra_model, # Same declared vocabulary + shared clamp as the main Codex transport (agent.reasoning_effort): # per-model — "max" is gpt-5.6-only, "minimal"/"ultra" always rejected (live-verified, #68365). @@ -221,8 +221,6 @@ def _resolve_reasoning(model: str, params: dict[str, Any]) -> tuple[Any, bool]: # Grok 4.6 accepts xhigh; older Grok tops out at high. supported = XAI_GROK46_EFFORTS if is_grok_46_family(model) else XAI_LEGACY_EFFORTS - elif (params.get("provider") or "").strip().lower() == "actual": - supported = ACTUAL_RELAY_EFFORTS else: declared = _profile_declared_efforts(params.get("provider"), model, params.get("base_url")) if declared is not None and not declared: diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index 0e6aa4a5ec..ac49fb7e5a 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -1344,7 +1344,8 @@ def _resolve_switch_credentials(st: _Switch) -> Optional[ModelSwitchResult]: # Fills an empty mode (alias cleared it) and overrides a STALE mode carried from previous # session state when the host mandates one wire protocol (e.g. gpt-5.x on api.openai.com # would otherwise 400 on tools+reasoning). - mandated_mode = host_mandated_api_mode(st.base_url) + from hermes_cli.providers import is_actual_route + mandated_mode = "chat_completions" if is_actual_route(st.target_provider, st.base_url) else host_mandated_api_mode(st.base_url) if mandated_mode is not None: st.api_mode = mandated_mode st.api_mode = st.api_mode or determine_api_mode(st.target_provider, st.base_url) diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index 3982ea98e6..54380eb718 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -167,6 +167,14 @@ def normalize_provider(name: str) -> str: return ALIASES.get(key, key) +def is_actual_route(provider: str = "", base_url: str = "") -> bool: + """Identify Actual by provider/alias or its hosted endpoint, including custom routes.""" + return ( + normalize_provider(provider or "") == "actual" + or base_url_hostname(base_url) == "api.actual.inc" + ) + + def _models_dev_info(canonical: str, allow_network: bool = True): """models.dev entry or None. Single-arg call on the default path: test sites monkeypatch ``get_provider_info`` with single-arg lambdas.""" @@ -289,6 +297,8 @@ def host_mandated_api_mode(base_url: str = "") -> Optional[str]: return None url_lower = base_url.rstrip("/").lower() hostname = base_url_hostname(base_url) + if hostname == "api.actual.inc": + return "chat_completions" # Exact-hostname matching only — never bare substring — so lookalike hosts # (api.openai.com.attacker.test) and path-segment spoofs (proxy.test/api.openai.com/v1) are NOT treated # as the real endpoint. (#32243) @@ -343,6 +353,8 @@ def determine_api_mode(provider: str, base_url: str = "", model: str = "") -> st """API mode (wire protocol) for a provider/endpoint: host-mandated mode, then Nous dual-wire (model-derived — the overlay alone says openai_chat and would pin Claude on the wrong wire), then the known provider's transport, then bedrock, else ``chat_completions``.""" + if is_actual_route(provider, base_url): + return "chat_completions" mandated = host_mandated_api_mode(base_url) if mandated is not None: return mandated diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index a48aa64be9..37dc514bcf 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -32,7 +32,7 @@ from hermes_cli.auth import ( # resolve_external_process_provider_credentials i from hermes_cli import config as _config_mod from hermes_cli import models as _models # attribute access keeps ``hermes_cli.models.`` patches effective from hermes_constants import OPENROUTER_BASE_URL -from hermes_cli.providers import determine_api_mode, is_official_openai_host, nous_api_mode +from hermes_cli.providers import determine_api_mode, is_actual_route, is_official_openai_host, nous_api_mode from utils import base_url_host_matches, base_url_hostname, env_int @@ -91,7 +91,7 @@ def _config_base_url_trustworthy_for_bare_custom(cfg_base_url: str, cfg_provider # so the runtime resolver stays in lockstep: api.meta.ai — prompt caching only on Responses; # api.router.com — /v1/chat/completions is a minimal shim; api.anthropic.com — native Messages. _HOST_MANDATED_API_MODES = { - "api.x.ai": "codex_responses", "api.meta.ai": "codex_responses", "api.actual.inc": "codex_responses", + "api.x.ai": "codex_responses", "api.meta.ai": "codex_responses", "api.actual.inc": "chat_completions", "api.router.com": "codex_responses", "api.anthropic.com": "anthropic_messages", } @@ -144,12 +144,16 @@ def _fallback_api_mode(provider: str, base_url: str, model: str = "") -> str: first, then the transport the provider overlay declares via ``providers.determine_api_mode`` (``openai-api`` pointed at us.api.openai.com 400'd on every tool call without it), then ``chat_completions``.""" + if is_actual_route(provider, base_url): + return "chat_completions" return _detect_api_mode_for_url(base_url) or determine_api_mode(provider, base_url, model) or "chat_completions" def _resolve_plain_custom_api_mode(model_cfg: Dict[str, Any], base_url: str) -> str: """api_mode for legacy/plain ``provider: custom`` endpoints — conservative by default: only direct OpenAI/xAI/Meta URLs imply Responses; named custom providers opt in via ``api_mode``.""" + if is_actual_route(base_url=base_url): + return "chat_completions" configured_mode = _parse_api_mode(model_cfg.get("api_mode")) detected_mode = _detect_api_mode_for_url(base_url) if configured_mode == "codex_responses" and detected_mode != "codex_responses": @@ -247,6 +251,9 @@ _NO_ANTHROPIC_CREDENTIALS_MSG = ("No Anthropic credentials found. Set ANTHROPIC_ def _runtime(provider: str, api_mode: str, base_url: Any, api_key: Any, **extra: Any) -> Dict[str, Any]: """Build a resolved-runtime dict; ``extra`` carries source/requested_provider/provider-specific keys.""" + if is_actual_route(provider, base_url): + api_mode = "chat_completions" + base_url = normalize_actual_base_url(base_url) return {"provider": provider, "api_mode": api_mode, "base_url": base_url, "api_key": api_key, **extra} diff --git a/run_agent.py b/run_agent.py index c62443588d..1498cddb43 100644 --- a/run_agent.py +++ b/run_agent.py @@ -636,10 +636,11 @@ class AIAgent( @staticmethod def _provider_model_requires_responses_api(model: str, *, provider: Optional[str] = None) -> bool: """Return True when this provider/model pair should use Responses API.""" + from hermes_cli.providers import is_actual_route normalized_provider = (provider or "").strip().lower() # Nous serves GPT-5.x via chat completions (its /v1/responses returns 404); generic custom endpoints # may relay GPT-5 without full Responses semantics — only direct OpenAI/xAI URLs auto-upgrade. - if normalized_provider in ("nous", "custom", "actual"): + if normalized_provider in ("nous", "custom") or is_actual_route(provider): return False if normalized_provider == "copilot": try: diff --git a/tests/agent/test_actual_auxiliary_routing.py b/tests/agent/test_actual_auxiliary_routing.py index 3ee59067e0..1cfc06d03c 100644 --- a/tests/agent/test_actual_auxiliary_routing.py +++ b/tests/agent/test_actual_auxiliary_routing.py @@ -3,6 +3,7 @@ import asyncio from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer import json +import socket import threading import time @@ -11,8 +12,24 @@ import yaml @pytest.fixture -def actual_endpoint(): +def actual_endpoint(monkeypatch): + from agent.auxiliary_client import ( + shutdown_cached_clients, + _reset_aux_unhealthy_cache, + ) + + shutdown_cached_clients() + _reset_aux_unhealthy_cache() requests = [] + resolve_address = socket.getaddrinfo + + def local_actual_address(host, *args, **kwargs): + if host in ("api.actual.inc", b"api.actual.inc"): + host = "127.0.0.1" + return resolve_address(host, *args, **kwargs) + + monkeypatch.setattr(socket, "getaddrinfo", local_actual_address) + monkeypatch.setenv("NO_PROXY", "127.0.0.1,localhost,api.actual.inc") class Handler(BaseHTTPRequestHandler): def do_POST(self): @@ -21,6 +38,20 @@ def actual_endpoint(): if self.path != "/v1/chat/completions": self.send_error(404) return + if payload["model"] == "unavailable-model": + body = json.dumps({ + "error": { + "message": "The model is not supported with this account", + "type": "invalid_request_error", + "code": "unsupported_model", + } + }).encode() + self.send_response(400) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + return content = ( '{"title":"Actual background routing"}' if "response_format" in payload @@ -62,20 +93,43 @@ def actual_endpoint(): pass server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) - thread = threading.Thread(target=server.serve_forever, daemon=True) + thread = threading.Thread( + target=server.serve_forever, kwargs={"poll_interval": 0.01}, daemon=True + ) thread.start() try: yield f"http://127.0.0.1:{server.server_port}", requests finally: + shutdown_cached_clients() server.shutdown() server.server_close() thread.join(timeout=5) -@pytest.mark.parametrize("aux_provider", ["auto", "actual", "aci", "custom"]) -@pytest.mark.parametrize("use_api_key", [False, True]) +@pytest.mark.parametrize( + "aux_provider", + [ + "auto", + "actual", + "actual-computer", + "actualcomputer", + "aci", + "custom", + "custom:actual-relay", + ], +) +@pytest.mark.parametrize( + "hosted,use_api_key", [(False, False), (False, True), (True, True)] +) +@pytest.mark.parametrize("stale_mode", [None, "codex_responses"]) def test_actual_background_tasks_reach_chat_completions( - tmp_path, monkeypatch, actual_endpoint, aux_provider, use_api_key + tmp_path, + monkeypatch, + actual_endpoint, + aux_provider, + hosted, + use_api_key, + stale_mode, ): from agent.auxiliary_client import async_call_llm from agent.context_compressor import ContextCompressor @@ -83,11 +137,13 @@ def test_actual_background_tasks_reach_chat_completions( from hermes_cli.runtime_provider import resolve_runtime_provider base_url, requests = actual_endpoint + if hosted: + base_url = base_url.replace("127.0.0.1", "api.actual.inc") monkeypatch.setenv("HERMES_HOME", str(tmp_path)) if use_api_key: monkeypatch.setenv("ACTUAL_API_KEY", "actual-test-key") monkeypatch.setenv("ACTUAL_BASE_URL", "http://127.0.0.1:1") - aux_model = "override-model" if aux_provider == "custom" else "test-model" + aux_model = "override-model" if aux_provider.startswith("custom") else "test-model" config = { "model": { "provider": "aci" if aux_provider == "aci" else "actual", @@ -98,7 +154,12 @@ def test_actual_background_tasks_reach_chat_completions( task: { "provider": aux_provider, "model": aux_model, - **({"base_url": base_url + "/v1"} if aux_provider == "custom" else {}), + "api_mode": stale_mode, + **( + {"base_url": base_url if stale_mode else base_url + "/v1"} + if aux_provider == "custom" + else {} + ), } for task in ( "compression", @@ -107,10 +168,19 @@ def test_actual_background_tasks_reach_chat_completions( ) }, } + config["providers"] = { + "actual-relay": { + "base_url": base_url if stale_mode else base_url + "/v1", + "key_env": "ACTUAL_API_KEY", + "transport": stale_mode or "chat_completions", + } + } (tmp_path / "config.yaml").write_text(yaml.safe_dump(config), encoding="utf-8") runtime = resolve_runtime_provider(requested="actual") runtime["model"] = "test-model" assert runtime["api_mode"] == "chat_completions" + if stale_mode: + runtime["api_mode"] = stale_mode assert ( generate_title( "Stale conversation", main_runtime=runtime, runtime_validator=lambda: False @@ -145,10 +215,174 @@ def test_actual_background_tasks_reach_chat_completions( ) assert response.choices[0].message.content == "The task is complete." assert len(requests) == 3 - assert all( - path == "/v1/chat/completions" and payload["model"] == aux_model - for path, payload in requests + assert [(path, payload["model"]) for path, payload in requests] == [ + ("/v1/chat/completions", aux_model) + ] * 3 + + +@pytest.mark.parametrize( + "provider,hosted", + [ + ("actual", False), + ("aci", False), + ("actual-computer", False), + ("actualcomputer", False), + ("custom", True), + ("custom:actual-relay", True), + ], +) +@pytest.mark.parametrize( + "entrypoint", ["init", "auto", "switch", "fallback", "restore", "rotation"] +) +def test_actual_runtime_transitions_reach_chat_completions( + tmp_path, monkeypatch, actual_endpoint, provider, hosted, entrypoint +): + from agent.error_classifier import FailoverReason + from hermes_cli.runtime_provider import resolve_runtime_provider + from run_agent import AIAgent + + base_url, requests = actual_endpoint + if hosted: + base_url = base_url.replace("127.0.0.1", "api.actual.inc") + base_url += "/v1" + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + monkeypatch.setenv("ACTUAL_API_KEY", "actual-test-key") + monkeypatch.setenv("OPENAI_API_KEY", "actual-test-key") + config = { + "model": { + "provider": provider, + "default": "gpt-5.4", + "base_url": base_url, + "api_mode": "codex_responses", + }, + "providers": { + "actual-relay": { + "base_url": base_url, + "key_env": "ACTUAL_API_KEY", + "transport": "codex_responses", + } + }, + } + (tmp_path / "config.yaml").write_text(yaml.safe_dump(config), encoding="utf-8") + runtime = resolve_runtime_provider(requested=provider) + assert runtime["api_mode"] == "chat_completions" + agent = AIAgent( + provider=provider, + base_url=base_url.replace("api.actual.inc", "127.0.0.1") + if entrypoint == "rotation" + else base_url.removesuffix("/v1"), + api_key="actual-test-key", + api_mode=None if entrypoint == "auto" else "codex_responses", + model="primary-model" if entrypoint == "fallback" else "gpt-5.4", + enabled_toolsets=[], + quiet_mode=True, + skip_context_files=True, + skip_memory=True, + save_trajectories=False, + fallback_model={ + "provider": provider, + "model": "gpt-5.4", + "base_url": base_url.removesuffix("/v1"), + "api_key": "actual-test-key", + "api_mode": "codex_responses", + }, ) + try: + if entrypoint == "switch": + agent.switch_model( + "gpt-5.4", + provider, + api_key="actual-test-key", + base_url=base_url.removesuffix("/v1"), + api_mode="codex_responses", + ) + elif entrypoint == "fallback": + assert agent._try_activate_fallback(FailoverReason.rate_limit) + elif entrypoint == "restore": + agent._primary_runtime["api_mode"] = "codex_responses" + agent._fallback_activated = True + assert agent._restore_primary_runtime() + elif entrypoint == "rotation": + from agent.credential_pool import PooledCredential + + agent._swap_credential( + PooledCredential.from_dict( + provider, + { + "id": "rotated", + "access_token": "actual-rotated-key", + "base_url": base_url.removesuffix("/v1"), + }, + ) + ) + assert agent.api_mode == "chat_completions" + response = agent._interruptible_api_call( + agent._build_api_kwargs([{"role": "user", "content": "Reply briefly."}], []) + ) + assert response.choices[0].message.content == "The task is complete." + inference_requests = [ + (path, body) for path, body in requests if path != "/api/show" + ] + assert len(inference_requests) == 1 + assert inference_requests[0][0] == "/v1/chat/completions" + assert inference_requests[0][1]["model"] == "gpt-5.4" + finally: + agent.client.close() + + +@pytest.mark.parametrize("provider", ["actual", "aci", "custom:actual-relay"]) +@pytest.mark.parametrize("async_mode", [False, True]) +def test_actual_auxiliary_fallback_reaches_chat_completions( + tmp_path, monkeypatch, actual_endpoint, provider, async_mode +): + from agent.auxiliary_client import call_llm, async_call_llm + + local_url, requests = actual_endpoint + actual_url = local_url.replace("127.0.0.1", "api.actual.inc") + "/v1" + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + monkeypatch.setenv("ACTUAL_API_KEY", "actual-test-key") + config = { + "model": { + "provider": "actual", + "default": "test-model", + "base_url": actual_url, + }, + "providers": { + "actual-relay": { + "base_url": actual_url, + "key_env": "ACTUAL_API_KEY", + "transport": "codex_responses", + } + }, + "auxiliary": { + "session_search": { + "provider": "custom", + "model": "unavailable-model", + "base_url": local_url + "/v1", + "fallback_chain": [ + { + "provider": provider, + "model": "test-model", + "api_mode": "codex_responses", + "base_url": actual_url, + } + ], + } + }, + } + (tmp_path / "config.yaml").write_text(yaml.safe_dump(config), encoding="utf-8") + kwargs = { + "task": "session_search", + "messages": [{"role": "user", "content": "Summarize the results."}], + "timeout": 5, + } + response = ( + asyncio.run(async_call_llm(**kwargs)) if async_mode else call_llm(**kwargs) + ) + assert response.choices[0].message.content == "The task is complete." + assert requests[0][1]["model"] == "unavailable-model" + assert requests[-1][1]["model"] == "test-model" + assert all(path == "/v1/chat/completions" for path, _payload in requests) @pytest.mark.parametrize("override", ["", "http://127.0.0.1:8081", "invalid-url"]) diff --git a/tests/hermes_cli/test_actual_provider.py b/tests/hermes_cli/test_actual_provider.py index 4c5e12ffac..5c16d3fe52 100644 --- a/tests/hermes_cli/test_actual_provider.py +++ b/tests/hermes_cli/test_actual_provider.py @@ -159,12 +159,29 @@ def test_actual_runtime_ignores_legacy_mode_environment(monkeypatch): assert resolved["api_mode"] == "chat_completions" -def test_actual_hostname_detection_preserves_custom_responses_route(): +def test_actual_hostname_detection_repairs_custom_responses_route(): + from hermes_cli.providers import is_actual_route + base_url = "https://api.actual.inc/v1" - assert rp._detect_api_mode_for_url(base_url) == "codex_responses" - assert rp._fallback_api_mode("custom", base_url) == "codex_responses" - assert rp._resolve_plain_custom_api_mode({}, base_url) == "codex_responses" + assert rp._detect_api_mode_for_url(base_url) == "chat_completions" + assert rp._fallback_api_mode("custom", base_url) == "chat_completions" + assert rp._resolve_plain_custom_api_mode({}, base_url) == "chat_completions" + assert ( + rp._resolve_plain_custom_api_mode({"api_mode": "codex_responses"}, base_url) + == "chat_completions" + ) + for unrelated_url in ( + "https://api.actual.inc.example/v1", + "https://proxy.example/api.actual.inc/v1", + ): + assert not is_actual_route("custom", unrelated_url) + assert ( + rp._runtime("custom", "codex_responses", unrelated_url, "test-key")[ + "api_mode" + ] + == "codex_responses" + ) def test_actual_runtime_uses_local_env_without_key(monkeypatch): diff --git a/website/docs/integrations/providers.md b/website/docs/integrations/providers.md index d4202387a9..286abcee2a 100644 --- a/website/docs/integrations/providers.md +++ b/website/docs/integrations/providers.md @@ -602,7 +602,7 @@ The base URL can be overridden with `GMI_BASE_URL` (default: `https://api.gmi-se ### Actual Computer -Your own hardware as a private inference cluster via [Actual Computer](https://actual.inc). Two serving modes, both OpenAI-compatible (Hermes defaults to Chat Completions so reasoning and final content are returned together): +Your own hardware as a private inference cluster via [Actual Computer](https://actual.inc). Two serving modes, both using Chat Completions so reasoning and final content are returned together: - **Hosted relay** — `https://api.actual.inc`, end-to-end encrypted, routes to *your* cluster. Authenticate with an `ac_` inference key from [actual.inc/user/keys](https://actual.inc/user/keys). - **Local daemon** — on-device at `http://127.0.0.1:8080`, fully offline. No API key needed: Hermes detects the loopback base URL and authenticates with an internal placeholder automatically. @@ -626,7 +626,7 @@ model: Notes: - Model IDs come from your cluster's `GET /v1/models` — discover with `hermes model` or `curl -s https://api.actual.inc/v1/models -H "Authorization: Bearer $ACTUAL_API_KEY"`. - Bare hosts in `model.base_url` are normalized: `http://127.0.0.1:8080` becomes `http://127.0.0.1:8080/v1` automatically. The legacy `ACTUAL_BASE_URL` environment variable is a fallback when no Actual URL is configured in YAML. -- The built-in Actual provider uses Chat Completions for chat, compaction, title generation, and other auxiliary tasks. Stale built-in Responses modes are repaired. For an endpoint that requires Responses, configure a named custom provider with `transport: codex_responses` under `providers` in `config.yaml`; per-task `auxiliary..api_mode` overrides are also supported. +- Actual uses `/v1/chat/completions` for chat, compaction, title generation, and every other auxiliary task. This also applies to custom providers targeting `api.actual.inc`, model switches, and fallbacks. Legacy Responses settings in the main model, custom provider, or auxiliary task configuration are overridden automatically. - Reasoning effort is clamped to Actual's supported range (`none/low/medium/high/max`) — a global `xhigh`/`ultra` setting will not 400 requests. - Small local models: Hermes' full default toolset plus the system prompt can exceed a 32k context window, producing an empty-stream error from llama.cpp-family servers. Restrict the toolset (`-t file,web`) or load the model with a larger context. The optional `actual-setup` skill (`hermes skills install official/devops/actual-setup`) covers setup and troubleshooting in detail. - Aliases: `actual-computer`, `actualcomputer`, `aci`.