1052 lines
35 KiB
Python
1052 lines
35 KiB
Python
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
import httpx
|
|
import pytest
|
|
|
|
import EvoScientist.llm.adapter_registry as adapter_registry
|
|
from EvoScientist.llm.adapter_registry import NormalizedUsage, get_adapter_registry
|
|
from EvoScientist.llm.contracts import EvoRuntimeError
|
|
from EvoScientist.llm.model_config import (
|
|
EvoModelConfig,
|
|
convert_v2_to_v3_draft,
|
|
invocation_fingerprint,
|
|
route_semantics_hash,
|
|
)
|
|
from EvoScientist.llm.secret_store import EncryptedModelSecretStore
|
|
|
|
|
|
def _model(model_key: str, model_id: str, *, api_mode: str) -> dict:
|
|
return {
|
|
"model_key": model_key,
|
|
"provider_model_id": model_id,
|
|
"version_policy": "rolling",
|
|
"resolved_model_revision": None,
|
|
"display_name": model_key,
|
|
"description": "fixture",
|
|
"enabled": True,
|
|
"tags": ["fixture"],
|
|
"invocation": {"api_mode": api_mode, "tool_call_transport": "native"},
|
|
"capabilities": {"text": True},
|
|
"limits": {"context_tokens": 128000, "max_output_tokens": 8192},
|
|
"parameters": {
|
|
"defaults": {},
|
|
"purpose_overrides": {},
|
|
"user_options": {},
|
|
"constraints": [],
|
|
},
|
|
"access": {"visibility": "authenticated", "roles": []},
|
|
"billing": {
|
|
"sku": f"internal/{model_key}",
|
|
"pricing_revision": "internal-unmetered-v1",
|
|
"currency": "CNY",
|
|
"unit_scale": 1000000,
|
|
"input_microunits_per_million": 0,
|
|
"output_microunits_per_million": 0,
|
|
"cached_microunits_per_million": 0,
|
|
"multiplier": 1.25,
|
|
},
|
|
}
|
|
|
|
|
|
def v3_payload() -> dict:
|
|
specs = (
|
|
(
|
|
"anthropic-prod",
|
|
"anthropic",
|
|
"anthropic-v1",
|
|
"anthropic_native",
|
|
"messages",
|
|
"claude-fixture",
|
|
),
|
|
(
|
|
"openai-prod",
|
|
"openai",
|
|
"openai-v1",
|
|
"openai_native",
|
|
"responses",
|
|
"gpt-fixture",
|
|
),
|
|
(
|
|
"gemini-prod",
|
|
"google-gemini",
|
|
"google-gemini-v1",
|
|
"gemini_native",
|
|
"interactions",
|
|
"gemini-fixture",
|
|
),
|
|
("xai-prod", "xai", "xai-v1", "openai_compatible", "responses", "grok-fixture"),
|
|
)
|
|
providers = []
|
|
aliases = []
|
|
for provider_id, adapter_id, revision, wire, mode, model_id in specs:
|
|
providers.append(
|
|
{
|
|
"provider_id": provider_id,
|
|
"display_name": provider_id,
|
|
"adapter_id": adapter_id,
|
|
"adapter_revision": revision,
|
|
"wire_protocol": wire,
|
|
"enabled": True,
|
|
"connection": {
|
|
"base_url": get_adapter_registry()
|
|
.get(adapter_id, revision)
|
|
.recommended_base_url,
|
|
"credential_ref": f"secret://model-providers/{provider_id}#1",
|
|
},
|
|
"defaults": {},
|
|
"models": [
|
|
_model("general", model_id, api_mode=mode),
|
|
_model("fast", model_id + "-fast", api_mode=mode),
|
|
],
|
|
}
|
|
)
|
|
for model_key in ("general", "fast"):
|
|
aliases.append(
|
|
{
|
|
"alias": f"{provider_id}-{model_key}",
|
|
"display_name": f"{provider_id} {model_key}",
|
|
"provider_ref": provider_id,
|
|
"model_ref": model_key,
|
|
"enabled": True,
|
|
"access": {"visibility": "authenticated", "roles": []},
|
|
"defaults": {},
|
|
}
|
|
)
|
|
return {
|
|
"schema_version": 3,
|
|
"config_revision": 1,
|
|
"config_identity_key_id": "identity-v1",
|
|
"runtime_defaults": {},
|
|
"providers": providers,
|
|
"aliases": aliases,
|
|
"purpose_defaults": {
|
|
name: {}
|
|
for name in (
|
|
"main_agent",
|
|
"tool_selector",
|
|
"deepagents_summarizer",
|
|
"title",
|
|
)
|
|
},
|
|
"purpose_routes": {
|
|
"main_agent": {"default_alias": "openai-prod-general"},
|
|
"tool_selector": "inherit_main",
|
|
"deepagents_summarizer": "inherit_main",
|
|
"title": {"default_alias": "openai-prod-fast"},
|
|
},
|
|
"purpose_call_limits": {
|
|
"main_agent": {"max_output_tokens": 8192, "max_attempts_per_run": 2},
|
|
"tool_selector": {"max_output_tokens": 4096, "max_attempts_per_run": 2},
|
|
"deepagents_summarizer": {
|
|
"max_output_tokens": 4096,
|
|
"max_attempts_per_run": 2,
|
|
},
|
|
"title": {"max_output_tokens": 256, "max_attempts_per_run": 1},
|
|
},
|
|
"health_policy": {"provider_connection": {}, "model_route": {}},
|
|
"web_runtime": {},
|
|
"capability_evidence": [],
|
|
}
|
|
|
|
|
|
def test_v3_supports_four_native_adapters_and_multiple_models() -> None:
|
|
config = EvoModelConfig.parse(v3_payload(), require_evidence=False)
|
|
assert config.schema_version == 3
|
|
assert len(config.providers) == 4
|
|
assert all(len(provider.models) == 2 for provider in config.providers.values())
|
|
assert config.endpoint_pools == {}
|
|
assert config.tool_protocol_fallbacks == {}
|
|
assert all(
|
|
model.quote.multiplier == "1.25"
|
|
for provider in config.providers.values()
|
|
for model in provider.models.values()
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"base_url",
|
|
[
|
|
"",
|
|
"not a url",
|
|
"ftp://user:pass@example.test/path?key=value#fragment",
|
|
"https://169.254.169.254/latest/meta-data",
|
|
"custom://model.internal:70000/path",
|
|
],
|
|
)
|
|
def test_v3_provider_base_url_is_not_validated(base_url) -> None:
|
|
payload = v3_payload()
|
|
payload["providers"][0]["connection"]["base_url"] = base_url
|
|
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
|
|
endpoints = config.providers["anthropic-prod"].endpoints
|
|
assert len(endpoints) == 1
|
|
assert next(iter(endpoints.values())).base_url == base_url
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_model_discovery_does_not_prevalidate_base_url(monkeypatch) -> None:
|
|
requested = {}
|
|
|
|
class Response:
|
|
def raise_for_status(self):
|
|
return None
|
|
|
|
def json(self):
|
|
return {"data": [{"id": "model-a"}]}
|
|
|
|
class Client:
|
|
def __init__(self, **_kwargs):
|
|
pass
|
|
|
|
async def __aenter__(self):
|
|
return self
|
|
|
|
async def __aexit__(self, *_args):
|
|
return None
|
|
|
|
async def get(self, url, *, headers):
|
|
requested["url"] = url
|
|
requested["headers"] = headers
|
|
return Response()
|
|
|
|
monkeypatch.setattr(httpx, "AsyncClient", Client)
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
|
|
result = await registration.discover_models(
|
|
base_url="custom://model.internal:70000/path?key=value#fragment",
|
|
api_key="sk-test",
|
|
)
|
|
|
|
assert requested["url"] == (
|
|
"custom://model.internal:70000/path?key=value#fragment/models"
|
|
)
|
|
assert result == ({"provider_model_id": "model-a", "display_name": "model-a"},)
|
|
|
|
|
|
def test_v3_provider_can_use_current_credential_without_secret_ref(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
payload = v3_payload()
|
|
for provider in payload["providers"]:
|
|
provider["connection"].pop("credential_ref")
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
assert config.providers["openai-prod"].endpoints["openai-prod"].auth.ref == (
|
|
"provider://openai-prod"
|
|
)
|
|
|
|
store = EncryptedModelSecretStore(
|
|
tmp_path / "model_secrets.sqlite", master_secret="x" * 32
|
|
)
|
|
store.put(
|
|
"model-providers/openai-prod",
|
|
"sk-current",
|
|
created_by="admin",
|
|
status="active",
|
|
)
|
|
resolved = store.resolve(
|
|
config.providers["openai-prod"].endpoints["openai-prod"].auth
|
|
)
|
|
assert resolved.value == "sk-current"
|
|
assert resolved.authoritative_version == "1"
|
|
|
|
|
|
def test_v3_rejects_removed_endpoint_field() -> None:
|
|
payload = v3_payload()
|
|
payload["providers"][0]["endpoints"] = []
|
|
with pytest.raises(EvoRuntimeError, match="unknown fields"):
|
|
EvoModelConfig.parse(payload, require_evidence=False)
|
|
|
|
|
|
def test_qwen_37_descriptor_has_correct_bounds_and_parameter_range() -> None:
|
|
registration = get_adapter_registry().get("dashscope", "dashscope-v1")
|
|
descriptor = registration.resolve_model_descriptor(
|
|
"qwen3.7-plus",
|
|
"chat_completions",
|
|
context_tokens=1_000_000,
|
|
max_output_tokens=65_536,
|
|
declared_capabilities={
|
|
"text": True,
|
|
"tools": True,
|
|
"thinking": True,
|
|
"structured_output": True,
|
|
"vision": True,
|
|
},
|
|
)
|
|
assert descriptor.context_tokens == 1_000_000
|
|
assert descriptor.max_output_tokens == 65_536
|
|
assert descriptor.capabilities == frozenset(
|
|
{"text", "vision", "tools", "thinking", "structured_output"}
|
|
)
|
|
# Adapter catalogs supply defaults only. Product policy may enable a
|
|
# capability before the static catalog has been updated.
|
|
registration.resolve_model_descriptor(
|
|
"qwen3.7-plus",
|
|
"chat_completions",
|
|
context_tokens=1_000_000,
|
|
max_output_tokens=65_536,
|
|
declared_capabilities={"text": True, "documents": True},
|
|
)
|
|
with pytest.raises(EvoRuntimeError) as exc:
|
|
registration.validate_parameters({"temperature": 2}, path="parameters")
|
|
assert exc.value.code == "MODEL_PARAMETER_INVALID"
|
|
|
|
|
|
def test_dashscope_chat_compilation_sends_explicit_thinking_and_output_bounds() -> None:
|
|
registration = get_adapter_registry().get("dashscope", "dashscope-v1")
|
|
|
|
disabled = registration.compile_runtime_parameters(
|
|
"chat_completions", {"reasoning": "off"}, 65_536
|
|
)
|
|
enabled = registration.compile_runtime_parameters(
|
|
"chat_completions",
|
|
{"reasoning": "high", "reasoning_budget_tokens": 32_768},
|
|
65_536,
|
|
)
|
|
|
|
assert disabled["max_completion_tokens"] == 65_536
|
|
assert disabled["extra_body"] == {"enable_thinking": False}
|
|
assert enabled["extra_body"] == {
|
|
"enable_thinking": True,
|
|
"thinking_budget": 32_768,
|
|
}
|
|
|
|
|
|
def test_dashscope_request_level_policy_disables_thinking_for_json_and_forced_tools() -> None:
|
|
registration = get_adapter_registry().get("dashscope", "dashscope-v1")
|
|
|
|
structured = registration.compile_runtime_parameters(
|
|
"chat_completions", {"structured_output": True}, 65_536
|
|
)
|
|
forced_tool = registration.compile_runtime_parameters(
|
|
"chat_completions", {"tool_choice": "required"}, 65_536
|
|
)
|
|
|
|
assert structured["extra_body"] == {"enable_thinking": False}
|
|
assert structured["response_format"] == {"type": "json_object"}
|
|
assert forced_tool["extra_body"] == {"enable_thinking": False}
|
|
|
|
with pytest.raises(EvoRuntimeError) as structured_conflict:
|
|
registration.compile_runtime_parameters(
|
|
"chat_completions",
|
|
{"structured_output": True, "reasoning": "high"},
|
|
65_536,
|
|
)
|
|
assert structured_conflict.value.code == "MODEL_PARAMETER_CONFLICT"
|
|
|
|
with pytest.raises(EvoRuntimeError) as forced_tool_conflict:
|
|
registration.compile_runtime_parameters(
|
|
"chat_completions",
|
|
{"tool_choice": "required", "reasoning": "high"},
|
|
65_536,
|
|
)
|
|
assert forced_tool_conflict.value.code == "MODEL_PARAMETER_CONFLICT"
|
|
|
|
|
|
def test_openai_gpt5_chat_uses_max_completion_tokens_for_runtime_and_connection() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
|
|
params = registration.compile_runtime_parameters(
|
|
"chat_completions", {}, 16_384, provider_model_id="gpt-5.6"
|
|
)
|
|
_, _, body = registration.build_probe_request(
|
|
base_url="https://api.openai.com/v1",
|
|
api_key="sk-test",
|
|
provider_model_id="gpt-5.6",
|
|
api_mode="chat_completions",
|
|
probe_kind="connectivity",
|
|
)
|
|
|
|
assert params["max_completion_tokens"] == 16_384
|
|
assert "max_tokens" not in params
|
|
assert body["max_completion_tokens"] == 16
|
|
assert "max_tokens" not in body
|
|
|
|
|
|
def test_kimi_k3_chat_uses_max_completion_tokens_for_runtime_and_connection() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
|
|
params = registration.compile_runtime_parameters(
|
|
"chat_completions", {"reasoning": "high"}, 65_000, provider_model_id="k3"
|
|
)
|
|
_, _, body = registration.build_probe_request(
|
|
base_url="https://api.kimi.com/coding/v1",
|
|
api_key="sk-test",
|
|
provider_model_id="k3",
|
|
api_mode="chat_completions",
|
|
probe_kind="reasoning",
|
|
)
|
|
|
|
assert params == {
|
|
"max_completion_tokens": 65_000,
|
|
"use_responses_api": False,
|
|
"reasoning_effort": "high",
|
|
}
|
|
assert body["max_completion_tokens"] == 16
|
|
assert body["reasoning_effort"] == "low"
|
|
assert "max_tokens" not in body
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("adapter_id", "revision", "api_mode", "probe_kind", "path", "auth_header"),
|
|
[
|
|
("anthropic", "anthropic-v1", "messages", "tools", "/v1/messages", "x-api-key"),
|
|
("openai", "openai-v1", "responses", "structured_output", "/responses", "Authorization"),
|
|
("google-gemini", "google-gemini-v1", "interactions", "reasoning", "/v1beta/interactions", "x-goog-api-key"),
|
|
("xai", "xai-v1", "chat_completions", "vision", "/chat/completions", "Authorization"),
|
|
("dashscope", "dashscope-v1", "chat_completions", "reasoning", "/chat/completions", "Authorization"),
|
|
],
|
|
)
|
|
def test_adapter_capability_probe_requests_are_provider_specific(
|
|
adapter_id: str,
|
|
revision: str,
|
|
api_mode: str,
|
|
probe_kind: str,
|
|
path: str,
|
|
auth_header: str,
|
|
) -> None:
|
|
registration = get_adapter_registry().get(adapter_id, revision)
|
|
|
|
url, headers, body = registration.build_probe_request(
|
|
base_url=registration.recommended_base_url,
|
|
api_key="sk-probe",
|
|
provider_model_id="probe-model",
|
|
api_mode=api_mode,
|
|
probe_kind=probe_kind,
|
|
)
|
|
|
|
assert url.endswith(path)
|
|
assert auth_header in headers
|
|
assert body["model"] == "probe-model" if "model" in body else True
|
|
assert "sk-probe" not in str(body)
|
|
|
|
|
|
def test_adapter_rejects_capability_without_controlled_probe_fixture() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
|
|
with pytest.raises(EvoRuntimeError) as raised:
|
|
registration.build_probe_request(
|
|
base_url=registration.recommended_base_url,
|
|
api_key="sk-probe",
|
|
provider_model_id="probe-model",
|
|
api_mode="responses",
|
|
probe_kind="video",
|
|
)
|
|
|
|
assert raised.value.code == "MODEL_CAPABILITY_PROBE_UNSUPPORTED"
|
|
|
|
|
|
def test_probe_requires_semantic_tool_call_not_only_http_success() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
payloads = registration._decode_probe_payloads(
|
|
b'{"id":"resp_1","model":"gpt-test","output":[]}'
|
|
)
|
|
|
|
with pytest.raises(EvoRuntimeError) as raised:
|
|
registration._validate_probe_payloads("tools", payloads)
|
|
|
|
assert raised.value.code == "MODEL_CAPABILITY_PROBE_FAILED"
|
|
|
|
|
|
def test_probe_accepts_bounded_sse_reasoning_evidence() -> None:
|
|
registration = get_adapter_registry().get("dashscope", "dashscope-v1")
|
|
payloads = registration._decode_probe_payloads(
|
|
b'data: {"id":"1","model":"qwen-test","choices":[{"delta":{"reasoning_content":"x"}}]}\n\n'
|
|
b'data: [DONE]\n\n'
|
|
)
|
|
|
|
revision = registration._validate_probe_payloads("reasoning", payloads)
|
|
|
|
assert revision == "qwen-test"
|
|
|
|
|
|
def test_probe_validates_structured_output_shape() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
payloads = registration._decode_probe_payloads(
|
|
b'{"id":"1","choices":[{"message":{"content":"{\\"ok\\":true}"}}]}'
|
|
)
|
|
|
|
assert registration._validate_probe_payloads("structured_output", payloads) == ""
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_probe_retries_one_transient_provider_failure(monkeypatch) -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
requests = 0
|
|
|
|
def respond(request: httpx.Request) -> httpx.Response:
|
|
nonlocal requests
|
|
requests += 1
|
|
if requests == 1:
|
|
return httpx.Response(503, json={"error": {"code": "unavailable"}})
|
|
return httpx.Response(
|
|
200,
|
|
json={
|
|
"model": "k3",
|
|
"choices": [{"message": {"content": "OK"}, "finish_reason": "stop"}],
|
|
},
|
|
)
|
|
|
|
original_client = httpx.AsyncClient
|
|
transport = httpx.MockTransport(respond)
|
|
|
|
def client(**kwargs):
|
|
return original_client(transport=transport, **kwargs)
|
|
|
|
monkeypatch.setattr(httpx, "AsyncClient", client)
|
|
monkeypatch.setattr(adapter_registry, "_PROBE_RETRY_BASE_SECONDS", 0.0)
|
|
|
|
result = await registration.probe_model(
|
|
base_url="https://provider.example/v1",
|
|
api_key="sk-probe",
|
|
provider_model_id="k3",
|
|
api_mode="chat_completions",
|
|
probe_kinds=("connectivity",),
|
|
timeout_seconds=5,
|
|
max_attempts=2,
|
|
)
|
|
|
|
assert result == {"connectivity": "supported", "resolved_model_revision": "k3"}
|
|
assert requests == 2
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_probe_reports_exhausted_attempt_details(monkeypatch) -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
requests = 0
|
|
|
|
def respond(request: httpx.Request) -> httpx.Response:
|
|
nonlocal requests
|
|
requests += 1
|
|
return httpx.Response(503, json={"error": {"code": "unavailable"}})
|
|
|
|
original_client = httpx.AsyncClient
|
|
transport = httpx.MockTransport(respond)
|
|
|
|
def client(**kwargs):
|
|
return original_client(transport=transport, **kwargs)
|
|
|
|
monkeypatch.setattr(httpx, "AsyncClient", client)
|
|
monkeypatch.setattr(adapter_registry, "_PROBE_RETRY_BASE_SECONDS", 0.0)
|
|
|
|
with pytest.raises(EvoRuntimeError) as raised:
|
|
await registration.probe_model(
|
|
base_url="https://provider.example/v1",
|
|
api_key="sk-probe",
|
|
provider_model_id="k3",
|
|
api_mode="chat_completions",
|
|
probe_kinds=("reasoning",),
|
|
timeout_seconds=5,
|
|
max_attempts=2,
|
|
)
|
|
|
|
assert raised.value.code == "MODEL_PROVIDER_ERROR"
|
|
assert raised.value.details == (
|
|
{
|
|
"path": "probe.reasoning",
|
|
"code": "MODEL_PROVIDER_ERROR",
|
|
"probe_kind": "reasoning",
|
|
"attempts": 2,
|
|
"retryable": True,
|
|
},
|
|
)
|
|
assert requests == 2
|
|
|
|
|
|
def test_secret_lifecycle_revocation_is_immediate(tmp_path: Path) -> None:
|
|
store = EncryptedModelSecretStore(
|
|
tmp_path / "model_secrets.sqlite", master_secret="x" * 32
|
|
)
|
|
pending = store.create_pending(
|
|
"openai-prod", "sk-secret", created_by="admin", operation_id="create-1"
|
|
)
|
|
assert pending.status == "pending"
|
|
active = store.activate(pending.secret_id, pending.version, operation_id="commit-1")
|
|
assert active.status == "active"
|
|
revoked = store.revoke(
|
|
pending.secret_id,
|
|
pending.version,
|
|
revoked_by="admin",
|
|
reason="compromised",
|
|
operation_id="revoke-1",
|
|
)
|
|
assert revoked.status == "revoked"
|
|
from EvoScientist.llm.model_config import SecretReference
|
|
|
|
with pytest.raises(EvoRuntimeError) as exc:
|
|
store.resolve(SecretReference(pending.ref, pending.version))
|
|
assert exc.value.code == "MODEL_CREDENTIAL_REVOKED"
|
|
|
|
|
|
def test_retired_secret_remains_resolvable_for_frozen_runs(tmp_path: Path) -> None:
|
|
from EvoScientist.llm.model_config import SecretReference
|
|
|
|
store = EncryptedModelSecretStore(
|
|
tmp_path / "model_secrets.sqlite", master_secret="x" * 32
|
|
)
|
|
first = store.create_pending(
|
|
"openai-prod", "sk-old", created_by="admin", operation_id="create-old"
|
|
)
|
|
store.activate(first.secret_id, first.version, operation_id="activate-old")
|
|
second = store.create_pending(
|
|
"openai-prod", "sk-new", created_by="admin", operation_id="create-new"
|
|
)
|
|
store.activate(second.secret_id, second.version, operation_id="activate-new")
|
|
|
|
metadata = {item.version: item for item in store.list_metadata()}
|
|
assert metadata[first.version].status == "retired"
|
|
assert store.resolve(SecretReference(first.ref, first.version)).value == "sk-old"
|
|
|
|
|
|
def test_v2_converter_refuses_to_guess_multiple_endpoints() -> None:
|
|
from tests.v3_fixtures import v3_payload as legacy_v2_payload
|
|
|
|
payload = legacy_v2_payload(revision=1)
|
|
provider = payload["providers"]["custom-openai"]
|
|
provider["endpoints"].append(
|
|
{
|
|
**provider["endpoints"][0],
|
|
"name": "secondary",
|
|
"base_url": "https://secondary.example/v1",
|
|
}
|
|
)
|
|
payload["endpoint_pools"]["default"]["endpoints"].append(
|
|
{"name": "secondary", "weight": 1}
|
|
)
|
|
draft, report = convert_v2_to_v3_draft(
|
|
payload,
|
|
target_revision=2,
|
|
config_identity_key_id="identity-v1",
|
|
)
|
|
assert draft["schema_version"] == 3
|
|
assert not draft["providers"]
|
|
assert report.blocking_issues[0]["code"] == "MULTIPLE_ENDPOINTS_REQUIRE_SPLIT"
|
|
|
|
|
|
def test_v2_converter_projects_single_endpoint_for_direct_editing() -> None:
|
|
from tests.v3_fixtures import v3_payload as legacy_v2_payload
|
|
|
|
payload = legacy_v2_payload(revision=4)
|
|
draft, report = convert_v2_to_v3_draft(
|
|
payload,
|
|
target_revision=4,
|
|
config_identity_key_id="identity-v1",
|
|
)
|
|
|
|
assert not report.blocking_issues
|
|
assert draft["schema_version"] == 3
|
|
assert draft["providers"][0]["provider_id"] == "custom-openai"
|
|
assert draft["providers"][0]["models"][0]["provider_model_id"] == "model-id"
|
|
assert draft["purpose_routes"]["main_agent"]["default_alias"] == "visible-model"
|
|
EvoModelConfig.parse(draft, require_evidence=False)
|
|
|
|
|
|
def test_stateless_adapter_invariants_and_partial_usage() -> None:
|
|
registry = get_adapter_registry()
|
|
for adapter_id, revision in (("openai", "openai-v1"), ("xai", "xai-v1")):
|
|
params = registry.get(adapter_id, revision).compile_runtime_parameters(
|
|
"responses", {"reasoning_effort": "high"}, 4096
|
|
)
|
|
assert params["store"] is False
|
|
assert "previous_response_id" not in params
|
|
gemini = registry.get("google-gemini", "google-gemini-v1")
|
|
assert gemini.compile_runtime_parameters("interactions", {}, 4096)["store"] is False
|
|
assert (
|
|
NormalizedUsage(10, 5, None, finality="partial").confirmed_projection() is None
|
|
)
|
|
|
|
|
|
def test_openai_chat_compilation_preserves_reasoning_effort() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
|
|
params = registration.compile_runtime_parameters(
|
|
"chat_completions", {"reasoning": "max"}, 8192
|
|
)
|
|
|
|
assert params == {
|
|
"max_tokens": 8192,
|
|
"use_responses_api": False,
|
|
"reasoning_effort": "max",
|
|
}
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("adapter_id", "adapter_revision", "model_id", "api_mode", "limit", "expected"),
|
|
[
|
|
(
|
|
"generic-openai-compatible",
|
|
"generic-openai-compatible-v1",
|
|
"qwen3.7-plus",
|
|
"chat_completions",
|
|
65_536,
|
|
{"max_tokens": 65_536, "use_responses_api": False},
|
|
),
|
|
(
|
|
"openai",
|
|
"openai-v1",
|
|
"kimi-for-coding",
|
|
"chat_completions",
|
|
65_000,
|
|
{"max_completion_tokens": 65_000, "use_responses_api": False},
|
|
),
|
|
(
|
|
"openai",
|
|
"openai-v1",
|
|
"kimi-for-coding-highspeed",
|
|
"chat_completions",
|
|
65_000,
|
|
{"max_completion_tokens": 65_000, "use_responses_api": False},
|
|
),
|
|
(
|
|
"openai",
|
|
"openai-v1",
|
|
"k3",
|
|
"chat_completions",
|
|
65_000,
|
|
{"max_completion_tokens": 65_000, "use_responses_api": False},
|
|
),
|
|
(
|
|
"openai",
|
|
"openai-v1",
|
|
"gpt-5.6",
|
|
"responses",
|
|
65_000,
|
|
{
|
|
"max_output_tokens": 65_000,
|
|
"use_responses_api": True,
|
|
"store": False,
|
|
},
|
|
),
|
|
(
|
|
"openai",
|
|
"openai-v1",
|
|
"gpt-5.6-sol",
|
|
"responses",
|
|
65_000,
|
|
{
|
|
"max_output_tokens": 65_000,
|
|
"use_responses_api": True,
|
|
"store": False,
|
|
},
|
|
),
|
|
(
|
|
"openai",
|
|
"openai-v1",
|
|
"gpt-5.6-terra",
|
|
"responses",
|
|
65_000,
|
|
{
|
|
"max_output_tokens": 65_000,
|
|
"use_responses_api": True,
|
|
"store": False,
|
|
},
|
|
),
|
|
],
|
|
)
|
|
def test_current_model_call_plans_compile_to_one_api_envelope(
|
|
adapter_id: str,
|
|
adapter_revision: str,
|
|
model_id: str,
|
|
api_mode: str,
|
|
limit: int,
|
|
expected: dict[str, int | bool],
|
|
) -> None:
|
|
params = get_adapter_registry().get(
|
|
adapter_id, adapter_revision
|
|
).compile_runtime_parameters(
|
|
api_mode, {}, limit, provider_model_id=model_id
|
|
)
|
|
|
|
assert params == expected
|
|
|
|
|
|
def test_openai_gpt_chat_plan_uses_completion_tokens_not_responses_tokens() -> None:
|
|
params = get_adapter_registry().get("openai", "openai-v1").compile_runtime_parameters(
|
|
"chat_completions", {}, 65_000, provider_model_id="gpt-5.6"
|
|
)
|
|
|
|
assert params == {"max_completion_tokens": 65_000, "use_responses_api": False}
|
|
|
|
|
|
def test_kimi_discovery_descriptor_is_partial_and_has_official_reasoning_policy() -> None:
|
|
registration = get_adapter_registry().get("openai", "openai-v1")
|
|
|
|
descriptor = registration.resolve_discovery_descriptor("k3")
|
|
|
|
assert descriptor is not None
|
|
assert descriptor.context_tokens == 1_048_576
|
|
assert descriptor.max_output_tokens is None
|
|
assert descriptor.reasoning_efforts == ("low", "high", "max")
|
|
assert descriptor.default_reasoning_effort == "high"
|
|
|
|
|
|
def test_model_reasoning_policy_can_restrict_efforts_and_enable_max() -> None:
|
|
payload = v3_payload()
|
|
provider = next(item for item in payload["providers"] if item["adapter_id"] == "openai")
|
|
model = provider["models"][0]
|
|
model["capabilities"]["thinking"] = True
|
|
model["parameters"]["reasoning_policy"] = {
|
|
"mode": "effort",
|
|
"allowed_efforts": ["low", "high", "max"],
|
|
"default_effort": "high",
|
|
}
|
|
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
parsed = config.providers[provider["provider_id"]].models[model["model_key"]]
|
|
|
|
assert parsed.supports_reasoning is True
|
|
assert parsed.allowed_reasoning_efforts == ("high", "low", "max")
|
|
assert parsed.reasoning_mode == "effort"
|
|
assert parsed.reasoning_enabled_params == {"reasoning": "high"}
|
|
|
|
|
|
def test_model_reasoning_policy_defaults_to_medium_without_explicit_effort() -> None:
|
|
payload = v3_payload()
|
|
provider = next(item for item in payload["providers"] if item["adapter_id"] == "openai")
|
|
model = provider["models"][0]
|
|
model["capabilities"]["thinking"] = True
|
|
# 不设置 reasoning_policy.default_effort → 兜底应为 medium
|
|
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
parsed = config.providers[provider["provider_id"]].models[model["model_key"]]
|
|
|
|
assert parsed.supports_reasoning is True
|
|
assert parsed.reasoning_enabled_params == {"reasoning": "medium"}
|
|
|
|
|
|
def test_generic_openai_compatible_adapter_supports_standard_model_discovery() -> None:
|
|
registration = get_adapter_registry().get(
|
|
"generic-openai-compatible", "generic-openai-compatible-v1"
|
|
)
|
|
|
|
assert registration.discovery_capability is True
|
|
|
|
|
|
def test_generic_adapter_does_not_publish_unimplemented_reasoning_capability() -> None:
|
|
payload = v3_payload()
|
|
provider = payload["providers"][1]
|
|
provider.update(
|
|
{
|
|
"adapter_id": "generic-openai-compatible",
|
|
"adapter_revision": "generic-openai-compatible-v1",
|
|
"wire_protocol": "openai_compatible",
|
|
}
|
|
)
|
|
provider["connection"]["base_url"] = "https://provider.example/v1"
|
|
for model in provider["models"]:
|
|
model["invocation"]["api_mode"] = "chat_completions"
|
|
model["capabilities"]["thinking"] = True
|
|
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
model = config.providers["openai-prod"].models["general"]
|
|
|
|
assert model.capabilities["thinking"] is True
|
|
assert model.reasoning_mode == "none"
|
|
assert model.supports_reasoning is False
|
|
assert model.allowed_reasoning_efforts == ()
|
|
|
|
|
|
def test_aliases_share_capability_evidence_but_not_invocation_identity() -> None:
|
|
payload = v3_payload()
|
|
openai_model = payload["providers"][1]["models"][0]
|
|
openai_model["parameters"]["user_options"]["temperature"] = {
|
|
"default": 0.2,
|
|
"applies_to": ["main_agent"],
|
|
"minimum": 0,
|
|
"maximum_exclusive": 2,
|
|
}
|
|
payload["aliases"][2]["defaults"] = {"temperature": 0.2}
|
|
payload["aliases"].append(
|
|
{
|
|
**payload["aliases"][2],
|
|
"alias": "openai-prod-creative",
|
|
"display_name": "OpenAI creative",
|
|
"defaults": {"temperature": 0.8},
|
|
}
|
|
)
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
general = config.concrete_routes(
|
|
config.main_routes.selectable["openai-prod-general"]
|
|
)[0]
|
|
creative = config.concrete_routes(
|
|
config.main_routes.selectable["openai-prod-creative"]
|
|
)[0]
|
|
key = b"identity-test-key-32-bytes-long!"
|
|
|
|
assert route_semantics_hash(config, general, key) == route_semantics_hash(
|
|
config, creative, key
|
|
)
|
|
assert invocation_fingerprint(
|
|
config, general, "main_agent", {"temperature": 0.2}, key
|
|
) != invocation_fingerprint(
|
|
config, creative, "main_agent", {"temperature": 0.8}, key
|
|
)
|
|
option = (
|
|
config.providers["openai-prod"].models["general"].user_options["temperature"]
|
|
)
|
|
assert option["minimum"] == 0
|
|
assert option["maximum_exclusive"] == 2
|
|
|
|
|
|
def test_user_option_cannot_loosen_adapter_bounds() -> None:
|
|
payload = v3_payload()
|
|
payload["providers"][1]["models"][0]["parameters"]["user_options"][
|
|
"temperature"
|
|
] = {"default": 2, "maximum": 2}
|
|
with pytest.raises(EvoRuntimeError) as exc:
|
|
EvoModelConfig.parse(payload, require_evidence=False)
|
|
assert exc.value.code == "MODEL_PARAMETER_INVALID"
|
|
|
|
|
|
def test_explicit_system_purpose_alias_is_preserved() -> None:
|
|
payload = v3_payload()
|
|
payload["purpose_routes"]["tool_selector"] = {
|
|
"default_alias": "anthropic-prod-fast"
|
|
}
|
|
config = EvoModelConfig.parse(payload, require_evidence=False)
|
|
selector_id = config.purpose_selector_ids["tool_selector"]
|
|
assert config.route_selectors[selector_id].alias == "anthropic-prod-fast"
|
|
|
|
|
|
def test_validate_rejects_conflict_after_alias_parameter_merge() -> None:
|
|
payload = v3_payload()
|
|
model = payload["providers"][1]["models"][0]
|
|
model["parameters"]["user_options"]["temperature"] = {"default": 0.2}
|
|
payload["aliases"][2]["defaults"] = {"temperature": 0.8}
|
|
payload["purpose_defaults"]["main_agent"] = {"top_p": 0.9}
|
|
with pytest.raises(EvoRuntimeError) as exc:
|
|
EvoModelConfig.parse(payload, require_evidence=False)
|
|
assert exc.value.code == "MODEL_PARAMETER_CONFLICT"
|
|
|
|
|
|
def test_anthropic_thinking_budget_is_strictly_below_output_limit() -> None:
|
|
registration = get_adapter_registry().get("anthropic", "anthropic-v1")
|
|
params = registration.compile_runtime_parameters(
|
|
"messages", {"thinking_enabled": True}, 4096
|
|
)
|
|
assert 1024 <= params["thinking"]["budget_tokens"] < params["max_tokens"]
|
|
|
|
|
|
def test_gemini_interactions_preserves_thought_signature() -> None:
|
|
from types import SimpleNamespace
|
|
|
|
from EvoScientist.llm.gemini_interactions import _chat_result, _compile_messages
|
|
|
|
class Block:
|
|
def model_dump(self, **_kwargs):
|
|
return {"type": "thought", "signature": "signed-opaque", "summary": []}
|
|
|
|
response = SimpleNamespace(
|
|
outputs=[Block()],
|
|
usage=SimpleNamespace(
|
|
total_input_tokens=2,
|
|
total_cached_tokens=0,
|
|
total_output_tokens=3,
|
|
total_thought_tokens=1,
|
|
total_tokens=5,
|
|
),
|
|
id="provider-id",
|
|
status="completed",
|
|
model=SimpleNamespace(id="gemini-fixture"),
|
|
)
|
|
message = _chat_result(response).generations[0].message
|
|
turns, _ = _compile_messages([message])
|
|
assert turns[0]["content"][0]["signature"] == "signed-opaque"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_gemini_interactions_streams_and_replays_signed_blocks(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.gemini_interactions import (
|
|
GeminiInteractionsChatModel,
|
|
_compile_messages,
|
|
create_gemini_interactions_model,
|
|
)
|
|
|
|
events = [
|
|
{
|
|
"event_type": "content.start",
|
|
"index": 0,
|
|
"content": {"type": "text", "text": ""},
|
|
},
|
|
{
|
|
"event_type": "content.delta",
|
|
"index": 0,
|
|
"delta": {"type": "text", "text": "hello"},
|
|
},
|
|
{"event_type": "content.stop", "index": 0},
|
|
{
|
|
"event_type": "content.start",
|
|
"index": 1,
|
|
"content": {"type": "thought", "summary": []},
|
|
},
|
|
{
|
|
"event_type": "content.delta",
|
|
"index": 1,
|
|
"delta": {"type": "thought_signature", "signature": "signed-stream"},
|
|
},
|
|
{"event_type": "content.stop", "index": 1},
|
|
{
|
|
"event_type": "content.start",
|
|
"index": 2,
|
|
"content": {
|
|
"type": "function_call",
|
|
"id": "call-1",
|
|
"name": "probe",
|
|
"arguments": {"value": "ok"},
|
|
},
|
|
},
|
|
{"event_type": "content.stop", "index": 2},
|
|
{
|
|
"event_type": "interaction.complete",
|
|
"interaction": {
|
|
"id": "request-1",
|
|
"status": "completed",
|
|
"model": {"id": "gemini-fixture"},
|
|
"usage": {
|
|
"total_input_tokens": 2,
|
|
"total_cached_tokens": 0,
|
|
"total_output_tokens": 3,
|
|
},
|
|
},
|
|
},
|
|
]
|
|
|
|
class Stream:
|
|
def __aiter__(self):
|
|
self.iterator = iter(events)
|
|
return self
|
|
|
|
async def __anext__(self):
|
|
try:
|
|
return next(self.iterator)
|
|
except StopIteration as exc:
|
|
raise StopAsyncIteration from exc
|
|
|
|
class Interactions:
|
|
async def create(self, **request):
|
|
assert request["stream"] is True
|
|
assert request["store"] is False
|
|
return Stream()
|
|
|
|
class Client:
|
|
aio = type("AsyncClient", (), {"interactions": Interactions()})()
|
|
|
|
monkeypatch.setattr(GeminiInteractionsChatModel, "_client", lambda self: Client())
|
|
model = create_gemini_interactions_model(model="gemini-fixture", api_key="secret")
|
|
chunks = [chunk async for chunk in model._astream([HumanMessage("hi")])]
|
|
combined = chunks[0].message
|
|
for chunk in chunks[1:]:
|
|
combined += chunk.message
|
|
|
|
assert combined.text == "hello"
|
|
assert combined.tool_calls[0]["name"] == "probe"
|
|
assert combined.usage_metadata["cached_input_tokens"] == 0
|
|
turns, _ = _compile_messages([combined])
|
|
assert turns[0]["content"][1]["signature"] == "signed-stream"
|