Files
EvoScientist/tests/test_model_factory.py
T
m4 b2e28249fd fix(model-registry): inject async safe clients and split ollama transports
Review fixes for the Task 3 contract layer:

- build_chat_model now accepts http_async_client alongside http_client
  (at least one required) and wires it into ChatOpenAI
  (http_async_client), ChatAnthropic (seeded _async_client), and
  ChatOllama (async_client_kwargs transport), closing the unsafe
  default-async-client gap.
- ChatOllama safe transports move from the shared client_kwargs to
  sync_client_kwargs/async_client_kwargs; langchain-ollama merges shared
  kwargs into both clients, which poisoned the async client with a sync
  transport and crashed ainvoke.
- Unsupported parameters now actually execute the contract-declared
  normalizer (reject_non_auto) instead of a hardcoded raise, with a
  fallback rejection if a normalizer would let a value through.
- build_chat_model rejects overlapping client_options/request_options
  keys instead of silently overwriting.
2026-07-20 22:40:33 +08:00

465 lines
17 KiB
Python

"""Tests for build_chat_model (design doc section 6.4).
``build_chat_model`` is the single model-construction entry point: it calls
``adapter.build_request`` for the resolved configuration, injects the Task 2
safe HTTP client, and never falls back to provider API-key environment
variables. Per-adapter fakes capture the construction arguments and request
options to prove the registry resolution matches the outbound request.
"""
from __future__ import annotations
import json
import httpx
import pytest
from langchain_core.messages import HumanMessage
from EvoScientist.model_registry import factory
from EvoScientist.model_registry.errors import (
ADAPTER_NOT_SUPPORTED,
CREDENTIAL_NOT_CONFIGURED,
ModelRegistryError,
)
from EvoScientist.model_registry.factory import build_chat_model
from EvoScientist.model_registry.schemas import ResolvedModelConfig
def _resolved_config(**overrides):
payload = {
"model_ref": {"provider_id": "zhipu-glm", "model_key": "glm-5.2"},
"role": "primary",
"adapter_id": "openai-compatible",
"adapter_spec_revision": 1,
"upstream_model_id": "glm-5.2",
"base_url": "https://open.bigmodel.cn/api/paas/v4",
"auth_ref": {
"mode": "api_key",
"credential_id": "zhipu-primary",
"credential_revision": 1,
},
"client_options": {"timeout_seconds": 120, "max_retries": 2},
"request_options": {
"max_output_tokens": 32768,
"temperature": 0.7,
"top_p": 0.95,
"reasoning_effort": "auto",
},
"budget": {
"resolved_input_limit": 1015808,
"fixed_reserves": {
"fixed_system_reserve_tokens": 4096,
"fixed_tools_reserve_tokens": 8192,
"fixed_attachments_reserve_tokens": 4096,
},
"message_budget": 1000000,
},
"effective_capabilities": {
"tools": True,
"vision": False,
"structured_output": True,
},
}
payload.update(overrides)
return ResolvedModelConfig.model_validate(payload)
def _http_client() -> httpx.Client:
return httpx.Client(transport=httpx.MockTransport(lambda request: None))
def _http_async_client() -> httpx.AsyncClient:
return httpx.AsyncClient(transport=httpx.MockTransport(lambda request: None))
class _FakeChatModel:
"""Captures construction arguments instead of opening a connection."""
def __init__(self, **kwargs):
self.kwargs = kwargs
class _FakeBuilder:
"""Records the options and HTTP clients a builder receives."""
def __init__(self):
self.calls = []
def __call__(self, options, http_client, http_async_client):
self.calls.append((dict(options), http_client, http_async_client))
return _FakeChatModel(**options)
@pytest.fixture
def fake_builders(monkeypatch):
builders = {}
for name in ("ChatOpenAI", "ChatAnthropic", "ChatOllama"):
builder = _FakeBuilder()
builders[name] = builder
monkeypatch.setitem(factory.CHAT_MODEL_BUILDERS, name, builder)
return builders
class TestInterfaceContract:
def test_rejects_extra_keyword_arguments(self):
with pytest.raises(TypeError):
build_chat_model(_resolved_config(), _http_client(), unexpected_option=True)
def test_rejects_positional_credential(self):
with pytest.raises(TypeError):
build_chat_model(_resolved_config(), _http_client(), "secret")
def test_rejects_non_resolved_config(self):
with pytest.raises(TypeError):
build_chat_model({"adapter_id": "openai"}, _http_client())
def test_rejects_non_httpx_client(self):
with pytest.raises(TypeError):
build_chat_model(_resolved_config(), object())
def test_unknown_adapter_spec_revision_fails_loudly(self):
resolved = _resolved_config(adapter_spec_revision=99)
with pytest.raises(ModelRegistryError) as excinfo:
build_chat_model(resolved, _http_client(), credential="secret")
assert excinfo.value.code == ADAPTER_NOT_SUPPORTED
class TestEnvironmentIsolation:
def test_openai_api_key_env_is_never_read(self, monkeypatch):
monkeypatch.setenv("OPENAI_API_KEY", "env-key")
with pytest.raises(ModelRegistryError) as excinfo:
build_chat_model(_resolved_config(), _http_client())
assert excinfo.value.code == CREDENTIAL_NOT_CONFIGURED
def test_explicit_credential_wins_over_env(self, monkeypatch):
monkeypatch.setenv("OPENAI_API_KEY", "env-key")
model = build_chat_model(
_resolved_config(), _http_client(), credential="real-key"
)
assert model.openai_api_key.get_secret_value() == "real-key"
def test_anthropic_api_key_env_is_never_read(self, monkeypatch):
monkeypatch.setenv("ANTHROPIC_API_KEY", "env-ant-key")
resolved = _resolved_config(
adapter_id="anthropic",
upstream_model_id="claude-x",
base_url="https://api.anthropic.com",
)
with pytest.raises(ModelRegistryError) as excinfo:
build_chat_model(resolved, _http_client())
assert excinfo.value.code == CREDENTIAL_NOT_CONFIGURED
def test_ollama_mode_none_strips_env_authorization(self, monkeypatch):
monkeypatch.setenv("OLLAMA_API_KEY", "ollama-env-secret")
resolved = _resolved_config(
adapter_id="ollama",
upstream_model_id="llama3",
base_url="http://localhost:11434",
auth_ref={"mode": "none", "credential_id": None},
)
model = build_chat_model(
resolved, _http_client(), http_async_client=_http_async_client()
)
headers = model._client._client.headers
assert "authorization" not in headers
async_headers = model._async_client._client.headers
assert "authorization" not in async_headers
class TestPerAdapterConstruction:
def test_openai_compatible_glm_options(self, fake_builders):
client = _http_client()
build_chat_model(_resolved_config(), client, credential="test-secret")
options, seen_client, seen_async = fake_builders["ChatOpenAI"].calls[0]
assert seen_client is client
assert seen_async is None
assert options == {
"model": "glm-5.2",
"base_url": "https://open.bigmodel.cn/api/paas/v4",
"timeout": 120,
"max_retries": 2,
"api_key": "test-secret",
"max_tokens": 32768,
"temperature": 0.7,
"top_p": 0.95,
}
def test_openai_adapter(self, fake_builders):
resolved = _resolved_config(
adapter_id="openai",
upstream_model_id="gpt-x",
base_url="https://api.openai.com/v1",
)
build_chat_model(resolved, _http_client(), credential="sk-test")
options, _, _ = fake_builders["ChatOpenAI"].calls[0]
assert options["model"] == "gpt-x"
assert options["max_tokens"] == 32768
assert options["api_key"] == "sk-test"
def test_anthropic_adapter(self, fake_builders):
resolved = _resolved_config(
adapter_id="anthropic",
upstream_model_id="claude-x",
base_url="https://api.anthropic.com",
)
client = _http_client()
build_chat_model(resolved, client, credential="sk-ant")
options, seen_client, _ = fake_builders["ChatAnthropic"].calls[0]
assert seen_client is client
assert options["model"] == "claude-x"
assert options["max_tokens"] == 32768
assert options["api_key"] == "sk-ant"
assert "reasoning_effort" not in options
def test_anthropic_compatible_adapter(self, fake_builders):
resolved = _resolved_config(
adapter_id="anthropic-compatible",
upstream_model_id="claude-x",
base_url="https://gateway.example.com",
)
build_chat_model(resolved, _http_client(), credential="sk-ant")
options, _, _ = fake_builders["ChatAnthropic"].calls[0]
assert options["base_url"] == "https://gateway.example.com"
assert options["max_tokens"] == 32768
def test_ollama_adapter(self, fake_builders):
resolved = _resolved_config(
adapter_id="ollama",
upstream_model_id="llama3",
base_url="http://localhost:11434",
auth_ref={"mode": "none", "credential_id": None},
)
client = _http_client()
build_chat_model(resolved, client)
options, seen_client, _ = fake_builders["ChatOllama"].calls[0]
assert seen_client is client
assert options["model"] == "llama3"
assert options["base_url"] == "http://localhost:11434"
assert options["num_predict"] == 32768
assert options["client_kwargs"] == {"timeout": 120}
assert "api_key" not in options
class TestSafeHttpClientInjection:
def test_chat_openai_receives_safe_client(self):
client = _http_client()
model = build_chat_model(_resolved_config(), client, credential="key")
assert model.http_client is client
def test_chat_anthropic_egress_uses_safe_client(self):
client = _http_client()
resolved = _resolved_config(
adapter_id="anthropic",
upstream_model_id="claude-x",
base_url="https://api.anthropic.com",
)
model = build_chat_model(resolved, client, credential="sk-ant")
assert model._client._client is client
def test_chat_ollama_egress_uses_safe_transport(self):
client = _http_client()
resolved = _resolved_config(
adapter_id="ollama",
upstream_model_id="llama3",
base_url="http://localhost:11434",
auth_ref={"mode": "none", "credential_id": None},
)
model = build_chat_model(resolved, client)
assert model._client._client._transport is client._transport
class TestOutboundRequestCapture:
def test_glm_request_matches_resolved_config(self, monkeypatch):
monkeypatch.delenv("OPENAI_API_KEY", raising=False)
captured = []
def handler(request: httpx.Request) -> httpx.Response:
captured.append(request)
return httpx.Response(
200,
json={
"id": "chatcmpl-test",
"object": "chat.completion",
"created": 1,
"model": "glm-5.2",
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "hello"},
"finish_reason": "stop",
}
],
"usage": {
"prompt_tokens": 3,
"completion_tokens": 1,
"total_tokens": 4,
},
},
)
client = httpx.Client(transport=httpx.MockTransport(handler))
model = build_chat_model(_resolved_config(), client, credential="real-key")
response = model.invoke([HumanMessage(content="hi")])
assert response.content == "hello"
assert len(captured) == 1
request = captured[0]
assert request.url.path == "/api/paas/v4/chat/completions"
assert request.headers["authorization"] == "Bearer real-key"
body = json.loads(request.content)
assert body["model"] == "glm-5.2"
# The contract maps max_output_tokens to the LangChain `max_tokens`
# constructor option; LangChain encodes it as max_completion_tokens
# on the wire.
assert body["max_completion_tokens"] == 32768
assert "max_tokens" not in body
assert body["temperature"] == 0.7
assert body["top_p"] == 0.95
assert "reasoning_effort" not in body
assert "reasoning" not in body
class TestAsyncClientInjection:
def _ollama_resolved(self):
return _resolved_config(
adapter_id="ollama",
upstream_model_id="llama3",
base_url="http://localhost:11434",
auth_ref={"mode": "none", "credential_id": None},
)
def test_requires_at_least_one_client(self):
with pytest.raises(TypeError):
build_chat_model(_resolved_config())
def test_rejects_non_httpx_async_client(self):
with pytest.raises(TypeError):
build_chat_model(_resolved_config(), http_async_client=object())
def test_chat_openai_receives_async_client(self):
async_client = _http_async_client()
model = build_chat_model(
_resolved_config(), http_async_client=async_client, credential="key"
)
assert model.http_async_client is async_client
def test_chat_anthropic_async_egress_uses_safe_client(self):
async_client = _http_async_client()
resolved = _resolved_config(
adapter_id="anthropic",
upstream_model_id="claude-x",
base_url="https://api.anthropic.com",
)
model = build_chat_model(
resolved, http_async_client=async_client, credential="sk-ant"
)
assert model._async_client._client is async_client
def test_chat_ollama_sync_and_async_transports_are_split(self):
sync_client = _http_client()
async_client = _http_async_client()
model = build_chat_model(
self._ollama_resolved(), sync_client, http_async_client=async_client
)
sync_transport = model._client._client._transport
async_transport = model._async_client._client._transport
assert sync_transport is sync_client._transport
assert async_transport is async_client._transport
assert async_transport is not sync_client._transport
async def test_ollama_ainvoke_uses_async_transport(self):
captured = []
def handler(request: httpx.Request) -> httpx.Response:
captured.append(request)
return httpx.Response(
200,
json={
"model": "llama3",
"created_at": "2026-01-01T00:00:00Z",
"message": {"role": "assistant", "content": "hello"},
"done": True,
"done_reason": "stop",
"total_duration": 1,
"prompt_eval_count": 3,
"eval_count": 1,
},
)
sync_client = httpx.Client(transport=httpx.MockTransport(handler))
async_client = httpx.AsyncClient(transport=httpx.MockTransport(handler))
model = build_chat_model(
self._ollama_resolved(), sync_client, http_async_client=async_client
)
response = await model.ainvoke([HumanMessage(content="hi")])
assert response.content == "hello"
assert len(captured) == 1
assert captured[0].url.path == "/api/chat"
class TestOptionKeyOverlap:
def test_overlapping_client_and_request_keys_rejected(self, monkeypatch):
from EvoScientist.model_registry.adapters import Adapter
from EvoScientist.model_registry.schemas import (
AdapterParameterSpec,
AuthSpec,
Capabilities,
ConnectionSpec,
ParameterRule,
)
spec = AdapterParameterSpec(
adapter_id="ollama",
spec_revision=1,
model_selector="*",
auth_specs={
"none": AuthSpec(
credential_required=False,
credential_kind="none",
target="adapter_internal",
target_name=None,
)
},
parameters={
"timeout_seconds": ParameterRule(
supported=True,
value_type="integer",
nullable="forbidden",
target="client_option",
target_name="shared",
normalizer="identity",
),
"max_retries": ParameterRule(
supported=True,
value_type="integer",
nullable="forbidden",
target="client_option",
target_name="max_retries",
normalizer="identity",
),
"max_output_tokens": ParameterRule(
supported=True,
value_type="integer",
nullable="forbidden",
target="request_option",
target_name="shared",
normalizer="identity",
),
},
protocol_capabilities=Capabilities(),
connection=ConnectionSpec(
chat_model="ChatOllama", model_field="model", base_url_field="base_url"
),
)
monkeypatch.setattr(
factory, "get_adapter", lambda *args, **kwargs: Adapter(spec)
)
resolved = _resolved_config(
adapter_id="ollama",
upstream_model_id="llama3",
base_url="http://localhost:11434",
auth_ref={"mode": "none", "credential_id": None},
)
with pytest.raises(ValueError, match="overlap"):
build_chat_model(resolved, _http_client())