5a581c78a2
Build / build (push) Has been cancelled
Docker / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
Introduce provider, model, and invocation contracts with encrypted configuration persistence. Add web runtime fencing, route fallback, recovery middleware, workspace scoping, and comprehensive tests.
1038 lines
35 KiB
Python
1038 lines
35 KiB
Python
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
import httpx
|
|
import pytest
|
|
|
|
import EvoScientist.llm.adapter_registry as adapter_registry
|
|
from EvoScientist.llm.adapter_registry import NormalizedUsage, get_adapter_registry
|
|
from EvoScientist.llm.contracts import EvoRuntimeError
|
|
from EvoScientist.llm.model_config import (
|
|
EvoModelConfig,
|
|
convert_v2_to_v3_draft,
|
|
invocation_fingerprint,
|
|
route_semantics_hash,
|
|
)
|
|
from EvoScientist.llm.secret_store import EncryptedModelSecretStore
|
|
|
|
|
|
def _model(model_key: str, model_id: str, *, api_mode: str) -> dict:
|
|
return {
|
|
"model_key": model_key,
|
|
"provider_model_id": model_id,
|
|
"version_policy": "rolling",
|
|
"resolved_model_revision": None,
|
|
"display_name": model_key,
|
|
"description": "fixture",
|
|
"enabled": True,
|
|
"tags": ["fixture"],
|
|
"invocation": {"api_mode": api_mode, "tool_call_transport": "native"},
|
|
"capabilities": {"text": True},
|
|
"limits": {"context_tokens": 128000, "max_output_tokens": 8192},
|
|
"parameters": {
|
|
"defaults": {},
|
|
"purpose_overrides": {},
|
|
"user_options": {},
|
|
"constraints": [],
|
|
},
|
|
"access": {"visibility": "authenticated", "roles": []},
|
|
"billing": {
|
|
"sku": f"internal/{model_key}",
|
|
"pricing_revision": "internal-unmetered-v1",
|
|
"currency": "CNY",
|
|
"unit_scale": 1000000,
|
|
"input_microunits_per_million": 0,
|
|
"output_microunits_per_million": 0,
|
|
"cached_microunits_per_million": 0,
|
|
"multiplier": 1.25,
|
|
},
|
|
}
|
|
|
|
|
|
def v3_payload() -> dict:
|
|
specs = (
|
|
(
|
|
"anthropic-prod",
|
|
"anthropic",
|
|
"anthropic-v1",
|
|
"anthropic_native",
|
|
"messages",
|
|
"claude-fixture",
|
|
),
|
|
(
|
|
"openai-prod",
|
|
"openai",
|
|
"openai-v1",
|
|
"openai_native",
|
|
"responses",
|
|
"gpt-fixture",
|
|
),
|
|
(
|
|
"gemini-prod",
|
|
"google-gemini",
|
|
"google-gemini-v1",
|
|
"gemini_native",
|
|
"interactions",
|
|
"gemini-fixture",
|
|
),
|
|
("xai-prod", "xai", "xai-v1", "openai_compatible", "responses", "grok-fixture"),
|
|
)
|
|
providers = []
|
|
aliases = []
|
|
for provider_id, adapter_id, revision, wire, mode, model_id in specs:
|
|
providers.append(
|
|
{
|
|
"provider_id": provider_id,
|
|
"display_name": provider_id,
|
|
"adapter_id": adapter_id,
|
|
"adapter_revision": revision,
|
|
"wire_protocol": wire,
|
|
"enabled": True,
|
|
"connection": {
|
|
"base_url": get_adapter_registry()
|
|
.get(adapter_id, revision)
|
|
.recommended_base_url,
|
|
"credential_ref": f"secret://model-providers/{provider_id}#1",
|
|
},
|
|
"defaults": {},
|
|
"models": [
|
|
_model("general", model_id, api_mode=mode),
|
|
_model("fast", model_id + "-fast", api_mode=mode),
|
|
],
|
|
}
|
|
)
|
|
for model_key in ("general", "fast"):
|
|
aliases.append(
|
|
{
|
|
"alias": f"{provider_id}-{model_key}",
|
|
"display_name": f"{provider_id} {model_key}",
|
|
"provider_ref": provider_id,
|
|
"model_ref": model_key,
|
|
"enabled": True,
|
|
"access": {"visibility": "authenticated", "roles": []},
|
|
"defaults": {},
|
|
}
|
|
)
|
|
return {
|
|
"schema_version": 3,
|
|
"config_revision": 1,
|
|
"config_identity_key_id": "identity-v1",
|
|
"runtime_defaults": {},
|
|
"providers": providers,
|
|
"aliases": aliases,
|
|
"purpose_defaults": {
|
|
name: {}
|
|
for name in (
|
|
"main_agent",
|
|
"tool_selector",
|
|
"deepagents_summarizer",
|
|
"title",
|
|
)
|
|
},
|
|
"purpose_routes": {
|
|
"main_agent": {"default_alias": "openai-prod-general"},
|
|
"tool_selector": "inherit_main",
|
|
"deepagents_summarizer": "inherit_main",
|
|
"title": {"default_alias": "openai-prod-fast"},
|
|
},
|
|
"purpose_call_limits": {
|
|
"main_agent": {"max_output_tokens": 8192, "max_attempts_per_run": 2},
|
|
"tool_selector": {"max_output_tokens": 4096, "max_attempts_per_run": 2},
|
|
"deepagents_summarizer": {
|
|
"max_output_tokens": 4096,
|
|
"max_attempts_per_run": 2,
|
|
},
|
|
"title": {"max_output_tokens": 256, "max_attempts_per_run": 1},
|
|
},
|
|
"health_policy": {"provider_connection": {}, "model_route": {}},
|
|
"web_runtime": {},
|
|
"capability_evidence": [],
|
|
}
|
|
|
|
|
|
def test_v3_supports_four_native_adapters_and_multiple_models() -> None:
|
|
config = EvoModelConfig.parse(v3_payload(), require_evidence=False)
|
|
assert config.schema_version == 3
|
|
assert len(config.providers) == 4
|
|
assert all(len(provider.models) == 2 for provider in config.providers.values())
|
|
assert config.endpoint_pools == {}
|
|
assert config.tool_protocol_fallbacks == {}
|
|
assert all(
|
|
model.quote.multiplier == "1.25"
|
|
for provider in config.providers.values()
|
|
for model in provider.models.values()
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"base_url",
|
|
[
|
|
"",
|
|
"not a url",
|
|
"ftp://user:pass@example.test/path?key=value#fragment",
|
|
"https://169.254.169.254/latest/meta-data",
|
|
"custom://model.internal:70000/path",
|
|
],
|
|
)
|
|
def test_v3_provider_base_url_is_not_validated(base_url) -> None:
|
|
payload = v3_payload()
|
|
payload["providers"][0]["connection"]["base_url"] = base_url
|
|
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
|
|
endpoints = config.providers["anthropic-prod"].endpoints
|
|
assert len(endpoints) == 1
|
|
assert next(iter(endpoints.values())).base_url == base_url
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_model_discovery_does_not_prevalidate_base_url(monkeypatch) -> None:
|
|
requested = {}
|
|
|
|
class Response:
|
|
def raise_for_status(self):
|
|
return None
|
|
|
|
def json(self):
|
|
return {"data": [{"id": "model-a"}]}
|
|
|
|
class Client:
|
|
def __init__(self, **_kwargs):
|
|
pass
|
|
|
|
async def __aenter__(self):
|
|
return self
|
|
|
|
async def __aexit__(self, *_args):
|
|
return None
|
|
|
|
async def get(self, url, *, headers):
|
|
requested["url"] = url
|
|
requested["headers"] = headers
|
|
return Response()
|
|
|
|
monkeypatch.setattr(httpx, "AsyncClient", Client)
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
|
|
result = await registration.discover_models(
|
|
base_url="custom://model.internal:70000/path?key=value#fragment",
|
|
api_key="sk-test",
|
|
)
|
|
|
|
assert requested["url"] == (
|
|
"custom://model.internal:70000/path?key=value#fragment/models"
|
|
)
|
|
assert result == ({"provider_model_id": "model-a", "display_name": "model-a"},)
|
|
|
|
|
|
def test_v3_provider_can_use_current_credential_without_secret_ref(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
payload = v3_payload()
|
|
for provider in payload["providers"]:
|
|
provider["connection"].pop("credential_ref")
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
assert config.providers["openai-prod"].endpoints["openai-prod"].auth.ref == (
|
|
"provider://openai-prod"
|
|
)
|
|
|
|
store = EncryptedModelSecretStore(
|
|
tmp_path / "model_secrets.sqlite", master_secret="x" * 32
|
|
)
|
|
store.put(
|
|
"model-providers/openai-prod",
|
|
"sk-current",
|
|
created_by="admin",
|
|
status="active",
|
|
)
|
|
resolved = store.resolve(
|
|
config.providers["openai-prod"].endpoints["openai-prod"].auth
|
|
)
|
|
assert resolved.value == "sk-current"
|
|
assert resolved.authoritative_version == "1"
|
|
|
|
|
|
def test_v3_rejects_removed_endpoint_field() -> None:
|
|
payload = v3_payload()
|
|
payload["providers"][0]["endpoints"] = []
|
|
with pytest.raises(EvoRuntimeError, match="unknown fields"):
|
|
EvoModelConfig.parse(payload, require_evidence=False)
|
|
|
|
|
|
def test_qwen_37_descriptor_has_correct_bounds_and_parameter_range() -> None:
|
|
registration = get_adapter_registry().get("dashscope", "dashscope-v1")
|
|
descriptor = registration.resolve_model_descriptor(
|
|
"qwen3.7-plus",
|
|
"chat_completions",
|
|
context_tokens=1_000_000,
|
|
max_output_tokens=65_536,
|
|
declared_capabilities={
|
|
"text": True,
|
|
"tools": True,
|
|
"thinking": True,
|
|
"structured_output": True,
|
|
"vision": True,
|
|
},
|
|
)
|
|
assert descriptor.context_tokens == 1_000_000
|
|
assert descriptor.max_output_tokens == 65_536
|
|
assert descriptor.capabilities == frozenset(
|
|
{"text", "vision", "tools", "thinking", "structured_output"}
|
|
)
|
|
# Adapter catalogs supply defaults only. Product policy may enable a
|
|
# capability before the static catalog has been updated.
|
|
registration.resolve_model_descriptor(
|
|
"qwen3.7-plus",
|
|
"chat_completions",
|
|
context_tokens=1_000_000,
|
|
max_output_tokens=65_536,
|
|
declared_capabilities={"text": True, "documents": True},
|
|
)
|
|
with pytest.raises(EvoRuntimeError) as exc:
|
|
registration.validate_parameters({"temperature": 2}, path="parameters")
|
|
assert exc.value.code == "MODEL_PARAMETER_INVALID"
|
|
|
|
|
|
def test_dashscope_chat_compilation_sends_explicit_thinking_and_output_bounds() -> None:
|
|
registration = get_adapter_registry().get("dashscope", "dashscope-v1")
|
|
|
|
disabled = registration.compile_runtime_parameters(
|
|
"chat_completions", {"reasoning": "off"}, 65_536
|
|
)
|
|
enabled = registration.compile_runtime_parameters(
|
|
"chat_completions",
|
|
{"reasoning": "high", "reasoning_budget_tokens": 32_768},
|
|
65_536,
|
|
)
|
|
|
|
assert disabled["max_completion_tokens"] == 65_536
|
|
assert disabled["extra_body"] == {"enable_thinking": False}
|
|
assert enabled["extra_body"] == {
|
|
"enable_thinking": True,
|
|
"thinking_budget": 32_768,
|
|
}
|
|
|
|
|
|
def test_dashscope_request_level_policy_disables_thinking_for_json_and_forced_tools() -> None:
|
|
registration = get_adapter_registry().get("dashscope", "dashscope-v1")
|
|
|
|
structured = registration.compile_runtime_parameters(
|
|
"chat_completions", {"structured_output": True}, 65_536
|
|
)
|
|
forced_tool = registration.compile_runtime_parameters(
|
|
"chat_completions", {"tool_choice": "required"}, 65_536
|
|
)
|
|
|
|
assert structured["extra_body"] == {"enable_thinking": False}
|
|
assert structured["response_format"] == {"type": "json_object"}
|
|
assert forced_tool["extra_body"] == {"enable_thinking": False}
|
|
|
|
with pytest.raises(EvoRuntimeError) as structured_conflict:
|
|
registration.compile_runtime_parameters(
|
|
"chat_completions",
|
|
{"structured_output": True, "reasoning": "high"},
|
|
65_536,
|
|
)
|
|
assert structured_conflict.value.code == "MODEL_PARAMETER_CONFLICT"
|
|
|
|
with pytest.raises(EvoRuntimeError) as forced_tool_conflict:
|
|
registration.compile_runtime_parameters(
|
|
"chat_completions",
|
|
{"tool_choice": "required", "reasoning": "high"},
|
|
65_536,
|
|
)
|
|
assert forced_tool_conflict.value.code == "MODEL_PARAMETER_CONFLICT"
|
|
|
|
|
|
def test_openai_gpt5_chat_uses_max_completion_tokens_for_runtime_and_connection() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
|
|
params = registration.compile_runtime_parameters(
|
|
"chat_completions", {}, 16_384, provider_model_id="gpt-5.6"
|
|
)
|
|
_, _, body = registration.build_probe_request(
|
|
base_url="https://api.openai.com/v1",
|
|
api_key="sk-test",
|
|
provider_model_id="gpt-5.6",
|
|
api_mode="chat_completions",
|
|
probe_kind="connectivity",
|
|
)
|
|
|
|
assert params["max_completion_tokens"] == 16_384
|
|
assert "max_tokens" not in params
|
|
assert body["max_completion_tokens"] == 16
|
|
assert "max_tokens" not in body
|
|
|
|
|
|
def test_kimi_k3_chat_uses_max_completion_tokens_for_runtime_and_connection() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
|
|
params = registration.compile_runtime_parameters(
|
|
"chat_completions", {"reasoning": "high"}, 65_000, provider_model_id="k3"
|
|
)
|
|
_, _, body = registration.build_probe_request(
|
|
base_url="https://api.kimi.com/coding/v1",
|
|
api_key="sk-test",
|
|
provider_model_id="k3",
|
|
api_mode="chat_completions",
|
|
probe_kind="reasoning",
|
|
)
|
|
|
|
assert params == {
|
|
"max_completion_tokens": 65_000,
|
|
"use_responses_api": False,
|
|
"reasoning_effort": "high",
|
|
}
|
|
assert body["max_completion_tokens"] == 16
|
|
assert body["reasoning_effort"] == "low"
|
|
assert "max_tokens" not in body
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("adapter_id", "revision", "api_mode", "probe_kind", "path", "auth_header"),
|
|
[
|
|
("anthropic", "anthropic-v1", "messages", "tools", "/v1/messages", "x-api-key"),
|
|
("openai", "openai-v1", "responses", "structured_output", "/responses", "Authorization"),
|
|
("google-gemini", "google-gemini-v1", "interactions", "reasoning", "/v1beta/interactions", "x-goog-api-key"),
|
|
("xai", "xai-v1", "chat_completions", "vision", "/chat/completions", "Authorization"),
|
|
("dashscope", "dashscope-v1", "chat_completions", "reasoning", "/chat/completions", "Authorization"),
|
|
],
|
|
)
|
|
def test_adapter_capability_probe_requests_are_provider_specific(
|
|
adapter_id: str,
|
|
revision: str,
|
|
api_mode: str,
|
|
probe_kind: str,
|
|
path: str,
|
|
auth_header: str,
|
|
) -> None:
|
|
registration = get_adapter_registry().get(adapter_id, revision)
|
|
|
|
url, headers, body = registration.build_probe_request(
|
|
base_url=registration.recommended_base_url,
|
|
api_key="sk-probe",
|
|
provider_model_id="probe-model",
|
|
api_mode=api_mode,
|
|
probe_kind=probe_kind,
|
|
)
|
|
|
|
assert url.endswith(path)
|
|
assert auth_header in headers
|
|
assert body["model"] == "probe-model" if "model" in body else True
|
|
assert "sk-probe" not in str(body)
|
|
|
|
|
|
def test_adapter_rejects_capability_without_controlled_probe_fixture() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
|
|
with pytest.raises(EvoRuntimeError) as raised:
|
|
registration.build_probe_request(
|
|
base_url=registration.recommended_base_url,
|
|
api_key="sk-probe",
|
|
provider_model_id="probe-model",
|
|
api_mode="responses",
|
|
probe_kind="video",
|
|
)
|
|
|
|
assert raised.value.code == "MODEL_CAPABILITY_PROBE_UNSUPPORTED"
|
|
|
|
|
|
def test_probe_requires_semantic_tool_call_not_only_http_success() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
payloads = registration._decode_probe_payloads(
|
|
b'{"id":"resp_1","model":"gpt-test","output":[]}'
|
|
)
|
|
|
|
with pytest.raises(EvoRuntimeError) as raised:
|
|
registration._validate_probe_payloads("tools", payloads)
|
|
|
|
assert raised.value.code == "MODEL_CAPABILITY_PROBE_FAILED"
|
|
|
|
|
|
def test_probe_accepts_bounded_sse_reasoning_evidence() -> None:
|
|
registration = get_adapter_registry().get("dashscope", "dashscope-v1")
|
|
payloads = registration._decode_probe_payloads(
|
|
b'data: {"id":"1","model":"qwen-test","choices":[{"delta":{"reasoning_content":"x"}}]}\n\n'
|
|
b'data: [DONE]\n\n'
|
|
)
|
|
|
|
revision = registration._validate_probe_payloads("reasoning", payloads)
|
|
|
|
assert revision == "qwen-test"
|
|
|
|
|
|
def test_probe_validates_structured_output_shape() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
payloads = registration._decode_probe_payloads(
|
|
b'{"id":"1","choices":[{"message":{"content":"{\\"ok\\":true}"}}]}'
|
|
)
|
|
|
|
assert registration._validate_probe_payloads("structured_output", payloads) == ""
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_probe_retries_one_transient_provider_failure(monkeypatch) -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
requests = 0
|
|
|
|
def respond(request: httpx.Request) -> httpx.Response:
|
|
nonlocal requests
|
|
requests += 1
|
|
if requests == 1:
|
|
return httpx.Response(503, json={"error": {"code": "unavailable"}})
|
|
return httpx.Response(
|
|
200,
|
|
json={
|
|
"model": "k3",
|
|
"choices": [{"message": {"content": "OK"}, "finish_reason": "stop"}],
|
|
},
|
|
)
|
|
|
|
original_client = httpx.AsyncClient
|
|
transport = httpx.MockTransport(respond)
|
|
|
|
def client(**kwargs):
|
|
return original_client(transport=transport, **kwargs)
|
|
|
|
monkeypatch.setattr(httpx, "AsyncClient", client)
|
|
monkeypatch.setattr(adapter_registry, "_PROBE_RETRY_BASE_SECONDS", 0.0)
|
|
|
|
result = await registration.probe_model(
|
|
base_url="https://provider.example/v1",
|
|
api_key="sk-probe",
|
|
provider_model_id="k3",
|
|
api_mode="chat_completions",
|
|
probe_kinds=("connectivity",),
|
|
timeout_seconds=5,
|
|
max_attempts=2,
|
|
)
|
|
|
|
assert result == {"connectivity": "supported", "resolved_model_revision": "k3"}
|
|
assert requests == 2
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_probe_reports_exhausted_attempt_details(monkeypatch) -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
requests = 0
|
|
|
|
def respond(request: httpx.Request) -> httpx.Response:
|
|
nonlocal requests
|
|
requests += 1
|
|
return httpx.Response(503, json={"error": {"code": "unavailable"}})
|
|
|
|
original_client = httpx.AsyncClient
|
|
transport = httpx.MockTransport(respond)
|
|
|
|
def client(**kwargs):
|
|
return original_client(transport=transport, **kwargs)
|
|
|
|
monkeypatch.setattr(httpx, "AsyncClient", client)
|
|
monkeypatch.setattr(adapter_registry, "_PROBE_RETRY_BASE_SECONDS", 0.0)
|
|
|
|
with pytest.raises(EvoRuntimeError) as raised:
|
|
await registration.probe_model(
|
|
base_url="https://provider.example/v1",
|
|
api_key="sk-probe",
|
|
provider_model_id="k3",
|
|
api_mode="chat_completions",
|
|
probe_kinds=("reasoning",),
|
|
timeout_seconds=5,
|
|
max_attempts=2,
|
|
)
|
|
|
|
assert raised.value.code == "MODEL_PROVIDER_ERROR"
|
|
assert raised.value.details == (
|
|
{
|
|
"path": "probe.reasoning",
|
|
"code": "MODEL_PROVIDER_ERROR",
|
|
"probe_kind": "reasoning",
|
|
"attempts": 2,
|
|
"retryable": True,
|
|
},
|
|
)
|
|
assert requests == 2
|
|
|
|
|
|
def test_secret_lifecycle_revocation_is_immediate(tmp_path: Path) -> None:
|
|
store = EncryptedModelSecretStore(
|
|
tmp_path / "model_secrets.sqlite", master_secret="x" * 32
|
|
)
|
|
pending = store.create_pending(
|
|
"openai-prod", "sk-secret", created_by="admin", operation_id="create-1"
|
|
)
|
|
assert pending.status == "pending"
|
|
active = store.activate(pending.secret_id, pending.version, operation_id="commit-1")
|
|
assert active.status == "active"
|
|
revoked = store.revoke(
|
|
pending.secret_id,
|
|
pending.version,
|
|
revoked_by="admin",
|
|
reason="compromised",
|
|
operation_id="revoke-1",
|
|
)
|
|
assert revoked.status == "revoked"
|
|
from EvoScientist.llm.model_config import SecretReference
|
|
|
|
with pytest.raises(EvoRuntimeError) as exc:
|
|
store.resolve(SecretReference(pending.ref, pending.version))
|
|
assert exc.value.code == "MODEL_CREDENTIAL_REVOKED"
|
|
|
|
|
|
def test_retired_secret_remains_resolvable_for_frozen_runs(tmp_path: Path) -> None:
|
|
from EvoScientist.llm.model_config import SecretReference
|
|
|
|
store = EncryptedModelSecretStore(
|
|
tmp_path / "model_secrets.sqlite", master_secret="x" * 32
|
|
)
|
|
first = store.create_pending(
|
|
"openai-prod", "sk-old", created_by="admin", operation_id="create-old"
|
|
)
|
|
store.activate(first.secret_id, first.version, operation_id="activate-old")
|
|
second = store.create_pending(
|
|
"openai-prod", "sk-new", created_by="admin", operation_id="create-new"
|
|
)
|
|
store.activate(second.secret_id, second.version, operation_id="activate-new")
|
|
|
|
metadata = {item.version: item for item in store.list_metadata()}
|
|
assert metadata[first.version].status == "retired"
|
|
assert store.resolve(SecretReference(first.ref, first.version)).value == "sk-old"
|
|
|
|
|
|
def test_v2_converter_refuses_to_guess_multiple_endpoints() -> None:
|
|
from tests.v3_fixtures import v3_payload as legacy_v2_payload
|
|
|
|
payload = legacy_v2_payload(revision=1)
|
|
provider = payload["providers"]["custom-openai"]
|
|
provider["endpoints"].append(
|
|
{
|
|
**provider["endpoints"][0],
|
|
"name": "secondary",
|
|
"base_url": "https://secondary.example/v1",
|
|
}
|
|
)
|
|
payload["endpoint_pools"]["default"]["endpoints"].append(
|
|
{"name": "secondary", "weight": 1}
|
|
)
|
|
draft, report = convert_v2_to_v3_draft(
|
|
payload,
|
|
target_revision=2,
|
|
config_identity_key_id="identity-v1",
|
|
)
|
|
assert draft["schema_version"] == 3
|
|
assert not draft["providers"]
|
|
assert report.blocking_issues[0]["code"] == "MULTIPLE_ENDPOINTS_REQUIRE_SPLIT"
|
|
|
|
|
|
def test_v2_converter_projects_single_endpoint_for_direct_editing() -> None:
|
|
from tests.v3_fixtures import v3_payload as legacy_v2_payload
|
|
|
|
payload = legacy_v2_payload(revision=4)
|
|
draft, report = convert_v2_to_v3_draft(
|
|
payload,
|
|
target_revision=4,
|
|
config_identity_key_id="identity-v1",
|
|
)
|
|
|
|
assert not report.blocking_issues
|
|
assert draft["schema_version"] == 3
|
|
assert draft["providers"][0]["provider_id"] == "custom-openai"
|
|
assert draft["providers"][0]["models"][0]["provider_model_id"] == "model-id"
|
|
assert draft["purpose_routes"]["main_agent"]["default_alias"] == "visible-model"
|
|
EvoModelConfig.parse(draft, require_evidence=False)
|
|
|
|
|
|
def test_stateless_adapter_invariants_and_partial_usage() -> None:
|
|
registry = get_adapter_registry()
|
|
for adapter_id, revision in (("openai", "openai-v1"), ("xai", "xai-v1")):
|
|
params = registry.get(adapter_id, revision).compile_runtime_parameters(
|
|
"responses", {"reasoning_effort": "high"}, 4096
|
|
)
|
|
assert params["store"] is False
|
|
assert "previous_response_id" not in params
|
|
gemini = registry.get("google-gemini", "google-gemini-v1")
|
|
assert gemini.compile_runtime_parameters("interactions", {}, 4096)["store"] is False
|
|
assert (
|
|
NormalizedUsage(10, 5, None, finality="partial").confirmed_projection() is None
|
|
)
|
|
|
|
|
|
def test_openai_chat_compilation_preserves_reasoning_effort() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
|
|
params = registration.compile_runtime_parameters(
|
|
"chat_completions", {"reasoning": "max"}, 8192
|
|
)
|
|
|
|
assert params == {
|
|
"max_tokens": 8192,
|
|
"use_responses_api": False,
|
|
"reasoning_effort": "max",
|
|
}
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("adapter_id", "adapter_revision", "model_id", "api_mode", "limit", "expected"),
|
|
[
|
|
(
|
|
"generic-openai-compatible",
|
|
"generic-openai-compatible-v1",
|
|
"qwen3.7-plus",
|
|
"chat_completions",
|
|
65_536,
|
|
{"max_tokens": 65_536, "use_responses_api": False},
|
|
),
|
|
(
|
|
"openai",
|
|
"openai-v1",
|
|
"kimi-for-coding",
|
|
"chat_completions",
|
|
65_000,
|
|
{"max_completion_tokens": 65_000, "use_responses_api": False},
|
|
),
|
|
(
|
|
"openai",
|
|
"openai-v1",
|
|
"kimi-for-coding-highspeed",
|
|
"chat_completions",
|
|
65_000,
|
|
{"max_completion_tokens": 65_000, "use_responses_api": False},
|
|
),
|
|
(
|
|
"openai",
|
|
"openai-v1",
|
|
"k3",
|
|
"chat_completions",
|
|
65_000,
|
|
{"max_completion_tokens": 65_000, "use_responses_api": False},
|
|
),
|
|
(
|
|
"openai",
|
|
"openai-v1",
|
|
"gpt-5.6",
|
|
"responses",
|
|
65_000,
|
|
{
|
|
"max_output_tokens": 65_000,
|
|
"use_responses_api": True,
|
|
"store": False,
|
|
},
|
|
),
|
|
(
|
|
"openai",
|
|
"openai-v1",
|
|
"gpt-5.6-sol",
|
|
"responses",
|
|
65_000,
|
|
{
|
|
"max_output_tokens": 65_000,
|
|
"use_responses_api": True,
|
|
"store": False,
|
|
},
|
|
),
|
|
(
|
|
"openai",
|
|
"openai-v1",
|
|
"gpt-5.6-terra",
|
|
"responses",
|
|
65_000,
|
|
{
|
|
"max_output_tokens": 65_000,
|
|
"use_responses_api": True,
|
|
"store": False,
|
|
},
|
|
),
|
|
],
|
|
)
|
|
def test_current_model_call_plans_compile_to_one_api_envelope(
|
|
adapter_id: str,
|
|
adapter_revision: str,
|
|
model_id: str,
|
|
api_mode: str,
|
|
limit: int,
|
|
expected: dict[str, int | bool],
|
|
) -> None:
|
|
params = get_adapter_registry().get(
|
|
adapter_id, adapter_revision
|
|
).compile_runtime_parameters(
|
|
api_mode, {}, limit, provider_model_id=model_id
|
|
)
|
|
|
|
assert params == expected
|
|
|
|
|
|
def test_openai_gpt_chat_plan_uses_completion_tokens_not_responses_tokens() -> None:
|
|
params = get_adapter_registry().get("openai", "openai-v1").compile_runtime_parameters(
|
|
"chat_completions", {}, 65_000, provider_model_id="gpt-5.6"
|
|
)
|
|
|
|
assert params == {"max_completion_tokens": 65_000, "use_responses_api": False}
|
|
|
|
|
|
def test_kimi_discovery_descriptor_is_partial_and_has_official_reasoning_policy() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
|
|
descriptor = registration.resolve_discovery_descriptor("k3")
|
|
|
|
assert descriptor is not None
|
|
assert descriptor.context_tokens == 1_048_576
|
|
assert descriptor.max_output_tokens is None
|
|
assert descriptor.reasoning_efforts == ("low", "high", "max")
|
|
assert descriptor.default_reasoning_effort == "high"
|
|
|
|
|
|
def test_model_reasoning_policy_can_restrict_efforts_and_enable_max() -> None:
|
|
payload = v3_payload()
|
|
provider = next(item for item in payload["providers"] if item["adapter_id"] == "openai")
|
|
model = provider["models"][0]
|
|
model["capabilities"]["thinking"] = True
|
|
model["parameters"]["reasoning_policy"] = {
|
|
"mode": "effort",
|
|
"allowed_efforts": ["low", "high", "max"],
|
|
"default_effort": "high",
|
|
}
|
|
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
parsed = config.providers[provider["provider_id"]].models[model["model_key"]]
|
|
|
|
assert parsed.supports_reasoning is True
|
|
assert parsed.allowed_reasoning_efforts == ("high", "low", "max")
|
|
assert parsed.reasoning_mode == "effort"
|
|
assert parsed.reasoning_enabled_params == {"reasoning": "high"}
|
|
|
|
|
|
def test_generic_openai_compatible_adapter_supports_standard_model_discovery() -> None:
|
|
registration = get_adapter_registry().get(
|
|
"generic-openai-compatible", "generic-openai-compatible-v1"
|
|
)
|
|
|
|
assert registration.discovery_capability is True
|
|
|
|
|
|
def test_generic_adapter_does_not_publish_unimplemented_reasoning_capability() -> None:
|
|
payload = v3_payload()
|
|
provider = payload["providers"][1]
|
|
provider.update(
|
|
{
|
|
"adapter_id": "generic-openai-compatible",
|
|
"adapter_revision": "generic-openai-compatible-v1",
|
|
"wire_protocol": "openai_compatible",
|
|
}
|
|
)
|
|
provider["connection"]["base_url"] = "https://provider.example/v1"
|
|
for model in provider["models"]:
|
|
model["invocation"]["api_mode"] = "chat_completions"
|
|
model["capabilities"]["thinking"] = True
|
|
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
model = config.providers["openai-prod"].models["general"]
|
|
|
|
assert model.capabilities["thinking"] is True
|
|
assert model.reasoning_mode == "none"
|
|
assert model.supports_reasoning is False
|
|
assert model.allowed_reasoning_efforts == ()
|
|
|
|
|
|
def test_aliases_share_capability_evidence_but_not_invocation_identity() -> None:
|
|
payload = v3_payload()
|
|
openai_model = payload["providers"][1]["models"][0]
|
|
openai_model["parameters"]["user_options"]["temperature"] = {
|
|
"default": 0.2,
|
|
"applies_to": ["main_agent"],
|
|
"minimum": 0,
|
|
"maximum_exclusive": 2,
|
|
}
|
|
payload["aliases"][2]["defaults"] = {"temperature": 0.2}
|
|
payload["aliases"].append(
|
|
{
|
|
**payload["aliases"][2],
|
|
"alias": "openai-prod-creative",
|
|
"display_name": "OpenAI creative",
|
|
"defaults": {"temperature": 0.8},
|
|
}
|
|
)
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
general = config.concrete_routes(
|
|
config.main_routes.selectable["openai-prod-general"]
|
|
)[0]
|
|
creative = config.concrete_routes(
|
|
config.main_routes.selectable["openai-prod-creative"]
|
|
)[0]
|
|
key = b"identity-test-key-32-bytes-long!"
|
|
|
|
assert route_semantics_hash(config, general, key) == route_semantics_hash(
|
|
config, creative, key
|
|
)
|
|
assert invocation_fingerprint(
|
|
config, general, "main_agent", {"temperature": 0.2}, key
|
|
) != invocation_fingerprint(
|
|
config, creative, "main_agent", {"temperature": 0.8}, key
|
|
)
|
|
option = (
|
|
config.providers["openai-prod"].models["general"].user_options["temperature"]
|
|
)
|
|
assert option["minimum"] == 0
|
|
assert option["maximum_exclusive"] == 2
|
|
|
|
|
|
def test_user_option_cannot_loosen_adapter_bounds() -> None:
|
|
payload = v3_payload()
|
|
payload["providers"][1]["models"][0]["parameters"]["user_options"][
|
|
"temperature"
|
|
] = {"default": 2, "maximum": 2}
|
|
with pytest.raises(EvoRuntimeError) as exc:
|
|
EvoModelConfig.parse(payload, require_evidence=False)
|
|
assert exc.value.code == "MODEL_PARAMETER_INVALID"
|
|
|
|
|
|
def test_explicit_system_purpose_alias_is_preserved() -> None:
|
|
payload = v3_payload()
|
|
payload["purpose_routes"]["tool_selector"] = {
|
|
"default_alias": "anthropic-prod-fast"
|
|
}
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
selector_id = config.purpose_selector_ids["tool_selector"]
|
|
assert config.route_selectors[selector_id].alias == "anthropic-prod-fast"
|
|
|
|
|
|
def test_validate_rejects_conflict_after_alias_parameter_merge() -> None:
|
|
payload = v3_payload()
|
|
model = payload["providers"][1]["models"][0]
|
|
model["parameters"]["user_options"]["temperature"] = {"default": 0.2}
|
|
payload["aliases"][2]["defaults"] = {"temperature": 0.8}
|
|
payload["purpose_defaults"]["main_agent"] = {"top_p": 0.9}
|
|
with pytest.raises(EvoRuntimeError) as exc:
|
|
EvoModelConfig.parse(payload, require_evidence=False)
|
|
assert exc.value.code == "MODEL_PARAMETER_CONFLICT"
|
|
|
|
|
|
def test_anthropic_thinking_budget_is_strictly_below_output_limit() -> None:
|
|
registration = get_adapter_registry().get("anthropic", "anthropic-v1")
|
|
params = registration.compile_runtime_parameters(
|
|
"messages", {"thinking_enabled": True}, 4096
|
|
)
|
|
assert 1024 <= params["thinking"]["budget_tokens"] < params["max_tokens"]
|
|
|
|
|
|
def test_gemini_interactions_preserves_thought_signature() -> None:
|
|
from types import SimpleNamespace
|
|
|
|
from EvoScientist.llm.gemini_interactions import _chat_result, _compile_messages
|
|
|
|
class Block:
|
|
def model_dump(self, **_kwargs):
|
|
return {"type": "thought", "signature": "signed-opaque", "summary": []}
|
|
|
|
response = SimpleNamespace(
|
|
outputs=[Block()],
|
|
usage=SimpleNamespace(
|
|
total_input_tokens=2,
|
|
total_cached_tokens=0,
|
|
total_output_tokens=3,
|
|
total_thought_tokens=1,
|
|
total_tokens=5,
|
|
),
|
|
id="provider-id",
|
|
status="completed",
|
|
model=SimpleNamespace(id="gemini-fixture"),
|
|
)
|
|
message = _chat_result(response).generations[0].message
|
|
turns, _ = _compile_messages([message])
|
|
assert turns[0]["content"][0]["signature"] == "signed-opaque"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_gemini_interactions_streams_and_replays_signed_blocks(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.gemini_interactions import (
|
|
GeminiInteractionsChatModel,
|
|
_compile_messages,
|
|
create_gemini_interactions_model,
|
|
)
|
|
|
|
events = [
|
|
{
|
|
"event_type": "content.start",
|
|
"index": 0,
|
|
"content": {"type": "text", "text": ""},
|
|
},
|
|
{
|
|
"event_type": "content.delta",
|
|
"index": 0,
|
|
"delta": {"type": "text", "text": "hello"},
|
|
},
|
|
{"event_type": "content.stop", "index": 0},
|
|
{
|
|
"event_type": "content.start",
|
|
"index": 1,
|
|
"content": {"type": "thought", "summary": []},
|
|
},
|
|
{
|
|
"event_type": "content.delta",
|
|
"index": 1,
|
|
"delta": {"type": "thought_signature", "signature": "signed-stream"},
|
|
},
|
|
{"event_type": "content.stop", "index": 1},
|
|
{
|
|
"event_type": "content.start",
|
|
"index": 2,
|
|
"content": {
|
|
"type": "function_call",
|
|
"id": "call-1",
|
|
"name": "probe",
|
|
"arguments": {"value": "ok"},
|
|
},
|
|
},
|
|
{"event_type": "content.stop", "index": 2},
|
|
{
|
|
"event_type": "interaction.complete",
|
|
"interaction": {
|
|
"id": "request-1",
|
|
"status": "completed",
|
|
"model": {"id": "gemini-fixture"},
|
|
"usage": {
|
|
"total_input_tokens": 2,
|
|
"total_cached_tokens": 0,
|
|
"total_output_tokens": 3,
|
|
},
|
|
},
|
|
},
|
|
]
|
|
|
|
class Stream:
|
|
def __aiter__(self):
|
|
self.iterator = iter(events)
|
|
return self
|
|
|
|
async def __anext__(self):
|
|
try:
|
|
return next(self.iterator)
|
|
except StopIteration as exc:
|
|
raise StopAsyncIteration from exc
|
|
|
|
class Interactions:
|
|
async def create(self, **request):
|
|
assert request["stream"] is True
|
|
assert request["store"] is False
|
|
return Stream()
|
|
|
|
class Client:
|
|
aio = type("AsyncClient", (), {"interactions": Interactions()})()
|
|
|
|
monkeypatch.setattr(GeminiInteractionsChatModel, "_client", lambda self: Client())
|
|
model = create_gemini_interactions_model(model="gemini-fixture", api_key="secret")
|
|
chunks = [chunk async for chunk in model._astream([HumanMessage("hi")])]
|
|
combined = chunks[0].message
|
|
for chunk in chunks[1:]:
|
|
combined += chunk.message
|
|
|
|
assert combined.text == "hello"
|
|
assert combined.tool_calls[0]["name"] == "probe"
|
|
assert combined.usage_metadata["cached_input_tokens"] == 0
|
|
turns, _ = _compile_messages([combined])
|
|
assert turns[0]["content"][1]["signature"] == "signed-stream"
|