Files
EvoScientist-Multi/EvoScientist/llm/model_config.py
T

3310 lines
128 KiB
Python

"""Strict Evo-owned V2/V3 model-route configuration."""
from __future__ import annotations
import hashlib
import hmac
import json
import math
import os
import re
import secrets
import sqlite3
import stat
import tempfile
import time
from collections.abc import Mapping, Sequence
from dataclasses import asdict, dataclass, field
from decimal import Decimal, InvalidOperation
from pathlib import Path
from typing import Any
import yaml
from filelock import FileLock
from ..config.settings import get_config_dir
from .adapter_registry import get_adapter_registry
from .configuration import (
EndpointConfig,
ModelConfig,
ProviderConfig,
ResolvedSecret,
SecretReference,
SecretResolver,
)
from .contracts import (
AdminConfigGrant,
AdminConfigGrantVerifier,
EvoRuntimeError,
PricingQuote,
)
from .crypto import canonical_json_v1, hmac_id, sha256_id
from .user_options import (
project_user_options_for_purpose,
validate_parameter_constraints,
)
_PURPOSES = ("main_agent", "tool_selector", "deepagents_summarizer", "title")
_REASONING_EFFORTS = frozenset({"disabled", "low", "medium", "high", "max"})
_REASONING_MODES = frozenset({"effort", "boolean"})
_BLOCKED_PARAM_KEYS = frozenset(
{
"api_key",
"base_url",
"max_tokens",
"max_output_tokens",
"max_retries",
"streaming",
"disable_streaming",
"use_responses_api",
"reasoning",
"token",
"secret",
"password",
"authorization",
}
)
_SENSITIVE_HEADERS = frozenset(
{"authorization", "proxy-authorization", "cookie", "set-cookie", "x-api-key"}
)
_HEADER_RE = re.compile(r"^[!#$%&'*+.^_`|~0-9A-Za-z-]+$")
_ENV_RE = re.compile(r"^[A-Z_][A-Z0-9_]*$")
_BIGINT_MAX = 2**63 - 1
_PARAM_ALLOWLIST: Mapping[tuple[str, str], frozenset[str]] = {
("custom-openai", "chat_completions"): frozenset(
{
"temperature",
"top_p",
"seed",
"frequency_penalty",
"presence_penalty",
"timeout",
"extra_body",
"output_token_limit",
}
),
("custom-openai", "responses"): frozenset(
{"temperature", "top_p", "seed", "timeout", "extra_body", "output_token_limit"}
),
("openai", "chat_completions"): frozenset(
{
"temperature",
"top_p",
"seed",
"frequency_penalty",
"presence_penalty",
"timeout",
"extra_body",
"output_token_limit",
}
),
("openai", "responses"): frozenset(
{"temperature", "top_p", "seed", "timeout", "extra_body", "output_token_limit"}
),
("anthropic", "messages"): frozenset(
{"temperature", "top_p", "top_k", "timeout", "output_token_limit"}
),
}
class _UniqueKeyLoader(yaml.SafeLoader):
pass
def _construct_mapping(
loader: yaml.SafeLoader, node: yaml.MappingNode, deep: bool = False
) -> Any:
loader.flatten_mapping(node)
result: dict[Any, Any] = {}
for key_node, value_node in node.value:
key = loader.construct_object(key_node, deep=deep)
if key in result:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"duplicate YAML key: {key}"
)
result[key] = loader.construct_object(value_node, deep=deep)
return result
_UniqueKeyLoader.add_constructor(
yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG,
_construct_mapping,
)
def load_yaml_unique(text: str) -> Any:
try:
return yaml.load(text, Loader=_UniqueKeyLoader)
except EvoRuntimeError:
raise
except yaml.YAMLError as exc:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid model route YAML"
) from exc
def _mapping(value: Any, name: str) -> Mapping[str, Any]:
if not isinstance(value, Mapping):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be a mapping"
)
if not all(isinstance(key, str) for key in value):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} keys must be strings"
)
return value
def _sequence(value: Any, name: str, *, allow_empty: bool = True) -> Sequence[Any]:
if not isinstance(value, list | tuple):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be a list"
)
if not allow_empty and not value:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must not be empty"
)
return value
def _strict_keys(value: Mapping[str, Any], *, allowed: set[str], name: str) -> None:
unknown = set(value) - allowed
if unknown:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
f"{name} has unknown fields: {', '.join(sorted(unknown))}",
)
def _text(value: Any, name: str) -> str:
if not isinstance(value, str) or not value.strip():
raise EvoRuntimeError("LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} is required")
return value.strip()
def _integer(
value: Any, name: str, *, minimum: int = 0, maximum: int = _BIGINT_MAX
) -> int:
if isinstance(value, bool) or not isinstance(value, int):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be an integer"
)
if value < minimum or value > maximum:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} is out of range"
)
return value
def _positive_decimal(value: Any, name: str, *, default: str = "1") -> str:
if value is None:
value = default
if isinstance(value, bool):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be a positive number"
)
try:
parsed = Decimal(str(value))
except (InvalidOperation, TypeError, ValueError) as exc:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be a positive number"
) from exc
if not parsed.is_finite() or parsed <= 0:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be greater than zero"
)
return format(parsed.normalize(), "f")
def _bool(value: Any, name: str) -> bool:
if not isinstance(value, bool):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be boolean"
)
return value
def _string_set(
value: Any, name: str, *, allowed: frozenset[str] | None = None
) -> tuple[str, ...]:
items = _sequence(value, name)
normalized = tuple(sorted({_text(item, name) for item in items}))
if allowed is not None and not set(normalized) <= allowed:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} contains unsupported values"
)
return normalized
def _validate_params(
value: Any, *, name: str, allowed: frozenset[str]
) -> Mapping[str, Any]:
params = dict(_mapping(value or {}, name))
unknown = set(params) - allowed
if unknown:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
f"{name} has unsupported parameters: {', '.join(sorted(unknown))}",
)
def visit(item: Any, path: str) -> None:
if item is None or isinstance(item, bool | int | float | str):
canonical_json_v1(item)
return
if isinstance(item, Mapping):
for key, child in item.items():
if not isinstance(key, str):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{path} key must be text"
)
if key.lower() in _BLOCKED_PARAM_KEYS:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{path}.{key} is forbidden"
)
visit(child, f"{path}.{key}")
return
if isinstance(item, list | tuple):
for index, child in enumerate(item):
visit(child, f"{path}[{index}]")
return
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{path} is not JSON data"
)
visit(params, name)
return params
@dataclass(frozen=True, slots=True)
class AliasConfig:
alias: str
display_name: str
provider_ref: str
model_ref: str
enabled: bool
access: Mapping[str, Any]
defaults: Mapping[str, Any]
@dataclass(frozen=True, slots=True)
class PoolEndpoint:
name: str
weight: int
@dataclass(frozen=True, slots=True)
class EndpointPool:
pool_id: str
provider: str
strategy: LiteralStrategy
endpoints: tuple[PoolEndpoint, ...]
LiteralStrategy = str
@dataclass(frozen=True, slots=True)
class RouteSelector:
selector_id: str
provider: str
endpoint: str | None
endpoint_pool: str | None
model: str
api_mode: str
tool_call_transport: str
identity_selector_id: str = ""
alias: str = ""
alias_defaults: Mapping[str, Any] = field(default_factory=dict)
@dataclass(frozen=True, slots=True)
class RouteRef:
selector_id: str
provider: str
endpoint: str
model: str
api_mode: str
tool_call_transport: str
def key(self) -> str:
return ":".join(
(
self.provider,
self.endpoint,
self.model,
self.api_mode,
self.tool_call_transport,
)
)
@dataclass(frozen=True, slots=True)
class PurposeRoutes:
default_alias: str
selectable: Mapping[str, str]
@dataclass(frozen=True, slots=True)
class PurposeCallLimit:
max_attempts_per_run: int
def _effective_output_limit(model: ModelConfig, params: Mapping[str, Any]) -> int:
override = params.get("output_token_limit")
return model.max_output_tokens if override is None else int(override)
def _validate_model_output_limits(model: ModelConfig) -> None:
candidates = [model.params.get("output_token_limit")]
candidates.extend(
values.get("output_token_limit")
for values in model.purpose_overrides.values()
)
for value in candidates:
if value is not None and int(value) > model.max_output_tokens:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"model output_token_limit exceeds model capability",
)
@dataclass(frozen=True, slots=True)
class RouteHealthPolicy:
failure_threshold: int
cooldown_seconds: int
half_open_max_inflight: int
counted_error_codes: tuple[str, ...]
open_immediately_error_codes: tuple[str, ...]
@dataclass(frozen=True, slots=True)
class WebRuntimePolicy:
title_start_timeout_seconds: int
prepare_ttl_seconds: int
turn_lease_grace_seconds: int
active_run_timeout_seconds: int
max_run_journal_events: int
max_run_journal_bytes: int
max_prepared_runs_per_subject: int
max_prepared_runs_total: int
@dataclass(frozen=True, slots=True)
class CapabilityEvidence:
route: RouteRef
connectivity: str
tool_capability: str
route_semantics_hash: str
endpoint_fingerprint: str
config_identity_key_id: str
adapter_revision: str
fixture_digest: str
verified_at: str
results: Mapping[str, str] = field(default_factory=dict)
evidence_expires_at: str = ""
implementation_fingerprint: str = ""
resolved_model_revision: str | None = None
reproducible: bool = False
@dataclass(frozen=True, slots=True)
class EvoModelConfig:
config_revision: int
config_identity_key_id: str
runtime_defaults: Mapping[str, Any]
purpose_defaults: Mapping[str, Mapping[str, str]]
providers: Mapping[str, ProviderConfig]
endpoint_pools: Mapping[str, EndpointPool]
route_health: RouteHealthPolicy
route_selectors: Mapping[str, RouteSelector]
main_routes: PurposeRoutes
title_selector_id: str
purpose_call_limits: Mapping[str, PurposeCallLimit]
web_runtime: WebRuntimePolicy
capability_evidence: Mapping[str, CapabilityEvidence]
tool_protocol_fallbacks: Mapping[str, tuple[str, ...]]
raw: Mapping[str, Any]
schema_version: int = 2
aliases: Mapping[str, AliasConfig] = field(default_factory=dict)
provider_health: RouteHealthPolicy | None = None
adapter_registry_revision: str = ""
purpose_selector_ids: Mapping[str, str] = field(default_factory=dict)
@classmethod
def parse(cls, payload: Any, *, require_evidence: bool = True) -> EvoModelConfig:
# ``require_evidence`` is retained for callers of the legacy schema API.
# Capability evidence is historical audit data, not a publication or
# invocation prerequisite.
raw = _mapping(payload, "model_routes")
if raw.get("schema_version") == 3:
return _parse_v3_config(raw, require_evidence=require_evidence)
_strict_keys(
raw,
allowed={
"schema_version",
"config_revision",
"config_identity_key_id",
"runtime_defaults",
"purpose_defaults",
"providers",
"endpoint_pools",
"route_health",
"route_selectors",
"purpose_routes",
"purpose_call_limits",
"web_runtime",
"capability_evidence",
"tool_protocol_fallbacks",
},
name="model_routes",
)
if raw.get("schema_version") != 2:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "schema_version must be 2"
)
revision = _integer(raw.get("config_revision"), "config_revision", minimum=1)
identity_key_id = _text(
raw.get("config_identity_key_id"), "config_identity_key_id"
)
runtime_defaults = _mapping(raw.get("runtime_defaults"), "runtime_defaults")
_strict_keys(runtime_defaults, allowed={"max_retries"}, name="runtime_defaults")
if runtime_defaults.get("max_retries") != 0:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "runtime max_retries must be 0"
)
purpose_defaults = _parse_purpose_defaults(raw.get("purpose_defaults"))
providers = _parse_providers(raw.get("providers"))
pools = _parse_endpoint_pools(raw.get("endpoint_pools"), providers)
health = _parse_health(raw.get("route_health"))
selectors = _parse_selectors(raw.get("route_selectors"), providers, pools)
main_routes, title_selector_id = _parse_purpose_routes(
raw.get("purpose_routes"), selectors
)
limits = _parse_call_limits(raw.get("purpose_call_limits"))
web_runtime = _parse_web_runtime(raw.get("web_runtime"))
fallbacks = _parse_fallbacks(raw.get("tool_protocol_fallbacks"), selectors)
config = cls(
config_revision=revision,
config_identity_key_id=identity_key_id,
runtime_defaults=dict(runtime_defaults),
purpose_defaults=purpose_defaults,
providers=providers,
endpoint_pools=pools,
route_health=health,
route_selectors=selectors,
main_routes=main_routes,
title_selector_id=title_selector_id,
purpose_call_limits=limits,
web_runtime=web_runtime,
capability_evidence={},
tool_protocol_fallbacks=fallbacks,
raw=dict(raw),
)
config._validate(require_evidence=require_evidence)
return config
@property
def catalog_revision(self) -> int:
return self.config_revision
@property
def title_start_timeout_seconds(self) -> int:
return self.web_runtime.title_start_timeout_seconds
def resolve_main_selector(self, alias: str | None) -> RouteSelector:
selected_alias = str(alias or "").strip() or self.main_routes.default_alias
selector_id = self.main_routes.selectable.get(selected_alias)
if selector_id is None:
raise EvoRuntimeError("MODEL_ACCESS_DENIED")
return self.route_selectors[selector_id]
def concrete_routes(self, selector_id: str) -> tuple[RouteRef, ...]:
selector = self.route_selectors[selector_id]
if selector.endpoint is not None:
endpoints = (selector.endpoint,)
else:
assert selector.endpoint_pool is not None
endpoints = tuple(
item.name
for item in self.endpoint_pools[selector.endpoint_pool].endpoints
)
return tuple(
RouteRef(
selector_id=selector_id,
provider=selector.provider,
endpoint=endpoint,
model=selector.model,
api_mode=selector.api_mode,
tool_call_transport=selector.tool_call_transport,
)
for endpoint in endpoints
)
def fallback_selectors(self, primary_selector_id: str) -> tuple[str, ...]:
return self.tool_protocol_fallbacks.get(primary_selector_id, ())
def route_model(self, route: RouteRef) -> ModelConfig:
return self.providers[route.provider].models[route.model]
def required_concrete_routes(self) -> Mapping[str, tuple[str, ...]]:
if self.schema_version == 3:
required: dict[str, tuple[str, ...]] = {}
selector_ids = set(self.main_routes.selectable.values()) | {
self.title_selector_id
}
for selector_id in selector_ids:
route = self.concrete_routes(selector_id)[0]
model = self.route_model(route)
probes = ["connectivity"]
probes.extend(
key
for key, enabled in model.capabilities.items()
if enabled and key != "text"
)
required[route.key()] = tuple(dict.fromkeys(probes))
return required
main_selector_ids = set(self.main_routes.selectable.values())
reachable = set(main_selector_ids)
for selector_id in tuple(main_selector_ids):
reachable.update(self.fallback_selectors(selector_id))
result: dict[str, tuple[str, ...]] = {}
for selector_id in reachable:
for route in self.concrete_routes(selector_id):
result[route.key()] = ("connectivity", "tool_protocol")
for route in self.concrete_routes(self.title_selector_id):
result.setdefault(route.key(), ("connectivity",))
return result
def _validate(self, *, require_evidence: bool = True) -> None:
if self.schema_version == 3:
self._validate_v3(require_evidence=require_evidence)
return
main_limit = self.purpose_call_limits["main_agent"].max_attempts_per_run
for selector in self.route_selectors.values():
model = self.providers[selector.provider].models[selector.model]
_validate_model_output_limits(model)
for primary, fallbacks in self.tool_protocol_fallbacks.items():
chain = (primary, *fallbacks)
if len(chain) != len(set(chain)) or len(chain) > main_limit:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid fallback chain"
)
primary_models = [
self.route_model(route) for route in self.concrete_routes(primary)
]
base = primary_models[0]
for selector_id in fallbacks:
for candidate in self.concrete_routes(selector_id):
model = self.route_model(candidate)
if model.model_id != base.model_id or model.quote != base.quote:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"fallback billing differs",
)
def _validate_v3(self, *, require_evidence: bool) -> None:
if self.endpoint_pools or self.tool_protocol_fallbacks:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"V3 direct routes cannot contain pools or fallback chains",
)
if not self.aliases or not self.main_routes.selectable:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "aliases are required"
)
total_attempts = sum(
value.max_attempts_per_run for value in self.purpose_call_limits.values()
)
if total_attempts > 16:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"purpose attempts exceed the run limit",
)
for alias, selector_id in self.main_routes.selectable.items():
selector = self.route_selectors[selector_id]
model = self.providers[selector.provider].models[selector.model]
registration = get_adapter_registry().get(
self.providers[selector.provider].adapter_id,
self.providers[selector.provider].adapter_revision,
)
if not model.enabled or not self.providers[selector.provider].enabled:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
f"enabled alias {alias} has a disabled target",
)
_validate_model_output_limits(model)
alias_config = self.aliases[alias]
applicable_purposes = {"main_agent"}
for purpose in ("tool_selector", "deepagents_summarizer"):
explicit_selector = self.purpose_selector_ids.get(purpose)
if explicit_selector in {None, selector_id}:
applicable_purposes.add(purpose)
if self.title_selector_id == selector_id:
applicable_purposes.add("title")
for purpose in applicable_purposes:
params = dict(self.purpose_defaults[purpose])
params.update(
project_user_options_for_purpose(
values=model.params,
user_options=model.user_options,
purpose=purpose,
)
)
for name, option in model.user_options.items():
if "default" in option and purpose in set(
option.get("applies_to") or ("main_agent",)
):
params.setdefault(name, option["default"])
params.update(
project_user_options_for_purpose(
values=alias_config.defaults,
user_options=model.user_options,
purpose=purpose,
)
)
params.update(model.purpose_overrides.get(purpose, {}))
registration.validate_parameters(
params, path=f"aliases.{alias}.merged.{purpose}"
)
registration.compile_runtime_parameters(
selector.api_mode,
params,
_effective_output_limit(model, params),
provider_model_id=model.model_id,
)
_V3_RUNTIME_DEFAULTS: Mapping[str, tuple[int, int, int]] = {
"sdk_max_retries": (0, 0, 0),
"connect_timeout_seconds": (10, 1, 60),
"first_event_timeout_seconds": (60, 1, 300),
"stream_idle_timeout_seconds": (60, 1, 300),
"attempt_timeout_seconds": (600, 1, 3600),
"max_sse_event_bytes": (1_048_576, 4_096, 4_194_304),
"max_content_block_bytes": (4_194_304, 4_096, 16_777_216),
"max_output_bytes": (16_777_216, 65_536, 67_108_864),
"max_opaque_state_bytes": (8_388_608, 65_536, 33_554_432),
"stream_buffer_max_events": (256, 1, 1_024),
"stream_buffer_max_bytes": (2_097_152, 65_536, 8_388_608),
"max_tool_schema_bytes": (262_144, 4_096, 1_048_576),
"max_tool_arguments_bytes": (1_048_576, 4_096, 4_194_304),
"max_tool_schema_depth": (16, 1, 32),
"max_tool_argument_depth": (32, 1, 64),
}
_V3_CAPABILITIES = (
"text",
"vision",
"video",
"documents",
"tools",
"structured_output",
"thinking",
)
_V3_ACCESS_KEYS = {"visibility", "roles", "groups", "users"}
def _parse_v3_access(value: Any, name: str) -> Mapping[str, Any]:
raw = _mapping(value, name)
_strict_keys(raw, allowed=_V3_ACCESS_KEYS, name=name)
visibility = _text(raw.get("visibility"), f"{name}.visibility")
if visibility not in {"authenticated", "role_based", "private"}:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name}.visibility is invalid"
)
result = {
"visibility": visibility,
"roles": _string_set(raw.get("roles") or [], f"{name}.roles"),
"groups": _string_set(raw.get("groups") or [], f"{name}.groups"),
"users": _string_set(raw.get("users") or [], f"{name}.users"),
}
if visibility == "authenticated" and any(
result[key] for key in ("roles", "groups", "users")
):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
f"{name} authenticated access must be unscoped",
)
if visibility == "role_based" and not (result["roles"] or result["groups"]):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} role access is empty"
)
if visibility == "private" and not result["users"]:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} private access is empty"
)
return result
def _normalize_v3_user_option(
value: Mapping[str, Any], rule: Any, path: str
) -> Mapping[str, Any]:
option = dict(value)
numeric = rule.kind in {"integer", "number"}
bound_names = (
"minimum",
"minimum_exclusive",
"maximum",
"maximum_exclusive",
)
if any(name in option for name in bound_names) and not numeric:
raise EvoRuntimeError(
"MODEL_PARAMETER_INVALID", f"{path} bounds require a numeric parameter"
)
if "minimum" in option and "minimum_exclusive" in option:
raise EvoRuntimeError("MODEL_PARAMETER_INVALID", f"{path} has two lower bounds")
if "maximum" in option and "maximum_exclusive" in option:
raise EvoRuntimeError("MODEL_PARAMETER_INVALID", f"{path} has two upper bounds")
for name in bound_names:
if name not in option:
continue
bound = option[name]
if (
isinstance(bound, bool)
or not isinstance(bound, int | float)
or not math.isfinite(float(bound))
):
raise EvoRuntimeError(
"MODEL_PARAMETER_INVALID", f"{path}.{name} must be finite"
)
if numeric and rule.minimum is not None:
default_name = "minimum_exclusive" if rule.minimum_exclusive else "minimum"
option.setdefault(default_name, rule.minimum)
if numeric and rule.maximum is not None:
default_name = "maximum_exclusive" if rule.maximum_exclusive else "maximum"
option.setdefault(default_name, rule.maximum)
lower_name = next(
(name for name in ("minimum", "minimum_exclusive") if name in option), None
)
upper_name = next(
(name for name in ("maximum", "maximum_exclusive") if name in option), None
)
if lower_name and rule.minimum is not None:
lower = float(option[lower_name])
if lower < rule.minimum or (
lower == rule.minimum and rule.minimum_exclusive and lower_name == "minimum"
):
raise EvoRuntimeError(
"MODEL_PARAMETER_INVALID", f"{path} loosens the adapter minimum"
)
if upper_name and rule.maximum is not None:
upper = float(option[upper_name])
if upper > rule.maximum or (
upper == rule.maximum and rule.maximum_exclusive and upper_name == "maximum"
):
raise EvoRuntimeError(
"MODEL_PARAMETER_INVALID", f"{path} loosens the adapter maximum"
)
if lower_name and upper_name:
lower = float(option[lower_name])
upper = float(option[upper_name])
if lower > upper or (
lower == upper
and (lower_name == "minimum_exclusive" or upper_name == "maximum_exclusive")
):
raise EvoRuntimeError(
"MODEL_PARAMETER_INVALID", f"{path} has an empty range"
)
if "choices" in option:
choices = list(
_sequence(option["choices"], f"{path}.choices", allow_empty=False)
)
for index, choice in enumerate(choices):
rule.validate(choice, f"{path}.choices[{index}]")
if len({canonical_json_v1(choice) for choice in choices}) != len(choices):
raise EvoRuntimeError(
"MODEL_PARAMETER_INVALID", f"{path}.choices contains duplicates"
)
option["choices"] = choices
elif rule.kind == "enum":
option["choices"] = list(rule.choices)
if "default" in option:
default = option["default"]
rule.validate(default, f"{path}.default")
if lower_name and (
default < option[lower_name]
or (lower_name == "minimum_exclusive" and default == option[lower_name])
):
raise EvoRuntimeError(
"MODEL_PARAMETER_INVALID", f"{path}.default is below its range"
)
if upper_name and (
default > option[upper_name]
or (upper_name == "maximum_exclusive" and default == option[upper_name])
):
raise EvoRuntimeError(
"MODEL_PARAMETER_INVALID", f"{path}.default exceeds its range"
)
if option.get("choices") and default not in option["choices"]:
raise EvoRuntimeError(
"MODEL_PARAMETER_INVALID", f"{path}.default is not an allowed choice"
)
return option
def _parse_v3_billing(value: Any, name: str, *, require_evidence: bool) -> PricingQuote:
raw = _mapping(value, name)
_strict_keys(
raw,
allowed={
"sku",
"pricing_revision",
"currency",
"unit_scale",
"input_microunits_per_million",
"output_microunits_per_million",
"cached_microunits_per_million",
"multiplier",
},
name=name,
)
payload = {
"billing_sku": _text(raw.get("sku"), f"{name}.sku"),
"pricing_revision": _text(
raw.get("pricing_revision"), f"{name}.pricing_revision"
),
"currency": _text(raw.get("currency"), f"{name}.currency"),
"unit_scale": _integer(raw.get("unit_scale"), f"{name}.unit_scale", minimum=1),
"input_microunits_per_million": _integer(
raw.get("input_microunits_per_million"), f"{name}.input", minimum=0
),
"output_microunits_per_million": _integer(
raw.get("output_microunits_per_million"), f"{name}.output", minimum=0
),
"cached_input_microunits_per_million": _integer(
raw.get("cached_microunits_per_million"), f"{name}.cached", minimum=0
),
"multiplier": _positive_decimal(raw.get("multiplier"), f"{name}.multiplier"),
}
if payload["currency"] != "CNY" or payload["unit_scale"] != 1_000_000:
raise EvoRuntimeError("PRICING_DIMENSION_UNSUPPORTED")
if require_evidence and payload["pricing_revision"].startswith("draft-"):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "active pricing is not approved"
)
return PricingQuote(**payload, quote_id=sha256_id(payload))
def _parse_v3_runtime_defaults(value: Any) -> Mapping[str, int]:
raw = _mapping(value or {}, "runtime_defaults")
_strict_keys(raw, allowed=set(_V3_RUNTIME_DEFAULTS), name="runtime_defaults")
result = {
key: _integer(
raw.get(key, default),
f"runtime_defaults.{key}",
minimum=minimum,
maximum=maximum,
)
for key, (default, minimum, maximum) in _V3_RUNTIME_DEFAULTS.items()
}
if result["sdk_max_retries"] != 0:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "sdk_max_retries must be 0"
)
if result["attempt_timeout_seconds"] < max(
result["connect_timeout_seconds"],
result["first_event_timeout_seconds"],
result["stream_idle_timeout_seconds"],
):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "attempt timeout is too small"
)
return result
def _parse_v3_provider_defaults(
value: Any, runtime: Mapping[str, int]
) -> Mapping[str, int]:
raw = _mapping(value or {}, "provider.defaults")
allowed = {
"connect_timeout_seconds",
"first_event_timeout_seconds",
"stream_idle_timeout_seconds",
"attempt_timeout_seconds",
"max_inflight_requests",
"queue_timeout_seconds",
}
_strict_keys(raw, allowed=allowed, name="provider.defaults")
result: dict[str, int] = {}
for key in allowed - {"max_inflight_requests", "queue_timeout_seconds"}:
result[key] = _integer(
raw.get(key, runtime[key]),
f"provider.defaults.{key}",
minimum=1,
maximum=_V3_RUNTIME_DEFAULTS[key][2],
)
result["max_inflight_requests"] = _integer(
raw.get("max_inflight_requests", 16),
"provider.defaults.max_inflight_requests",
minimum=1,
maximum=256,
)
result["queue_timeout_seconds"] = _integer(
raw.get("queue_timeout_seconds", 5),
"provider.defaults.queue_timeout_seconds",
minimum=1,
maximum=60,
)
if result["attempt_timeout_seconds"] < max(
result["connect_timeout_seconds"],
result["first_event_timeout_seconds"],
result["stream_idle_timeout_seconds"],
):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "provider attempt timeout is too small"
)
return result
def _parse_v3_health(value: Any, name: str) -> RouteHealthPolicy:
raw = _mapping(value or {}, name)
_strict_keys(
raw,
allowed={
"failure_threshold",
"cooldown_seconds",
"half_open_max_inflight",
"counted_error_codes",
"open_immediately_error_codes",
},
name=name,
)
counted = _string_set(
raw.get("counted_error_codes") or [], f"{name}.counted_error_codes"
)
immediate = _string_set(
raw.get("open_immediately_error_codes") or [],
f"{name}.open_immediately_error_codes",
)
return RouteHealthPolicy(
_integer(
raw.get("failure_threshold", 3),
f"{name}.failure_threshold",
minimum=1,
maximum=20,
),
_integer(
raw.get("cooldown_seconds", 30),
f"{name}.cooldown_seconds",
minimum=1,
maximum=3600,
),
_integer(
raw.get("half_open_max_inflight", 1),
f"{name}.half_open_max_inflight",
minimum=1,
maximum=16,
),
counted,
immediate,
)
def _parse_v3_web_runtime(value: Any) -> WebRuntimePolicy:
raw = _mapping(value or {}, "web_runtime")
specs = {
"title_start_timeout_seconds": (30, 1, 300),
"prepare_ttl_seconds": (30, 5, 300),
"turn_lease_grace_seconds": (30, 1, 300),
"active_run_timeout_seconds": (1800, 60, 7200),
"max_run_journal_events": (10000, 100, 100000),
"max_run_journal_bytes": (16777216, 1048576, 268435456),
"max_prepared_runs_per_subject": (4, 1, 32),
"max_prepared_runs_total": (128, 1, 4096),
}
_strict_keys(raw, allowed=set(specs), name="web_runtime")
parsed = {
key: _integer(
raw.get(key, default), f"web_runtime.{key}", minimum=low, maximum=high
)
for key, (default, low, high) in specs.items()
}
if parsed["max_prepared_runs_total"] < parsed["max_prepared_runs_per_subject"]:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "prepared run limits are inconsistent"
)
return WebRuntimePolicy(**parsed)
def _parse_v3_config(
raw: Mapping[str, Any], *, require_evidence: bool
) -> EvoModelConfig:
_strict_keys(
raw,
allowed={
"schema_version",
"config_revision",
"config_identity_key_id",
"runtime_defaults",
"providers",
"aliases",
"purpose_defaults",
"purpose_routes",
"purpose_call_limits",
"health_policy",
"web_runtime",
"capability_evidence",
},
name="model_routes",
)
revision = _integer(raw.get("config_revision"), "config_revision", minimum=1)
identity_key_id = _text(raw.get("config_identity_key_id"), "config_identity_key_id")
runtime_defaults = _parse_v3_runtime_defaults(raw.get("runtime_defaults"))
registry = get_adapter_registry()
providers_raw = _sequence(raw.get("providers"), "providers", allow_empty=False)
providers: dict[str, ProviderConfig] = {}
for provider_index, provider_value in enumerate(providers_raw):
name = f"providers[{provider_index}]"
item = _mapping(provider_value, name)
_strict_keys(
item,
allowed={
"provider_id",
"display_name",
"adapter_id",
"adapter_revision",
"wire_protocol",
"enabled",
"connection",
"defaults",
"models",
},
name=name,
)
provider_id = _text(item.get("provider_id"), f"{name}.provider_id")
if provider_id in providers:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate provider_id"
)
adapter_id = _text(item.get("adapter_id"), f"{name}.adapter_id")
exact_revision = _text(item.get("adapter_revision"), f"{name}.adapter_revision")
if exact_revision in {"latest", "current"}:
raise EvoRuntimeError("MODEL_ADAPTER_UNAVAILABLE")
registration = registry.get(adapter_id, exact_revision)
if registration.lifecycle == "blocked":
raise EvoRuntimeError("MODEL_ADAPTER_BLOCKED")
wire_protocol = _text(item.get("wire_protocol"), f"{name}.wire_protocol")
if wire_protocol not in registration.supported_wire_protocols:
raise EvoRuntimeError("MODEL_WIRE_PROTOCOL_UNSUPPORTED")
connection = _mapping(item.get("connection"), f"{name}.connection")
_strict_keys(
connection,
allowed={"base_url", "credential_ref"},
name=f"{name}.connection",
)
base_url = _normalize_base_url(
connection.get("base_url"), f"{name}.connection.base_url"
)
legacy_ref = connection.get("credential_ref")
if legacy_ref is None:
credential_ref = f"provider://{provider_id}"
secret_version = 0
else:
credential_ref = _text(legacy_ref, f"{name}.connection.credential_ref")
expected_prefix = f"secret://model-providers/{provider_id}#"
if not credential_ref.startswith(expected_prefix):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"credential_ref must be scoped to its provider",
)
try:
secret_version = int(credential_ref.rsplit("#", 1)[1])
except ValueError as exc:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"credential_ref version is invalid",
) from exc
if secret_version < 1:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"credential_ref version is invalid",
)
provider_defaults = _parse_v3_provider_defaults(
item.get("defaults"), runtime_defaults
)
models: dict[str, ModelConfig] = {}
for model_index, model_value in enumerate(
_sequence(item.get("models"), f"{name}.models", allow_empty=False)
):
model_name = f"{name}.models[{model_index}]"
model_raw = _mapping(model_value, model_name)
_strict_keys(
model_raw,
allowed={
"model_key",
"provider_model_id",
"version_policy",
"resolved_model_revision",
"display_name",
"description",
"enabled",
"tags",
"invocation",
"capabilities",
"limits",
"parameters",
"access",
"billing",
},
name=model_name,
)
model_key = _text(model_raw.get("model_key"), f"{model_name}.model_key")
if model_key in models:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate model_key"
)
provider_model_id = _text(
model_raw.get("provider_model_id"), f"{model_name}.provider_model_id"
)
policy = _text(
model_raw.get("version_policy"), f"{model_name}.version_policy"
)
if policy not in {"pinned", "rolling"}:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid version_policy"
)
resolved = model_raw.get("resolved_model_revision")
if resolved is not None:
resolved = _text(resolved, f"{model_name}.resolved_model_revision")
if policy == "pinned" and resolved is None:
resolved = provider_model_id
capabilities_raw = _mapping(
model_raw.get("capabilities"), f"{model_name}.capabilities"
)
_strict_keys(
capabilities_raw,
allowed=set(_V3_CAPABILITIES),
name=f"{model_name}.capabilities",
)
capabilities = {
key: _bool(
capabilities_raw.get(key, key == "text"),
f"{model_name}.capabilities.{key}",
)
for key in _V3_CAPABILITIES
}
if not capabilities["text"]:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "text capability is required"
)
invocation = _mapping(
model_raw.get("invocation"), f"{model_name}.invocation"
)
_strict_keys(
invocation,
allowed={"api_mode", "tool_call_transport"},
name=f"{model_name}.invocation",
)
api_mode = _text(
invocation.get("api_mode"), f"{model_name}.invocation.api_mode"
)
transport = _text(
invocation.get("tool_call_transport"),
f"{model_name}.invocation.tool_call_transport",
)
if transport not in {"native", "prompt", "disabled"} or (
capabilities["tools"] and transport != "native"
):
raise EvoRuntimeError("MODEL_TOOL_TRANSPORT_UNSUPPORTED")
if not capabilities["tools"]:
# V4 save validation rejects this mismatch for new revisions.
# Keep old signed projections runnable while treating their
# stale native/prompt declaration as the only safe transport.
transport = "disabled"
limits = _mapping(model_raw.get("limits"), f"{model_name}.limits")
_strict_keys(
limits,
allowed={
"context_tokens",
"max_output_tokens",
"max_inflight_requests",
},
name=f"{model_name}.limits",
)
context_tokens = limits.get("context_tokens")
max_output_tokens = limits.get("max_output_tokens")
if context_tokens is not None:
context_tokens = _integer(
context_tokens, f"{model_name}.limits.context_tokens", minimum=1
)
if max_output_tokens is not None:
max_output_tokens = _integer(
max_output_tokens,
f"{model_name}.limits.max_output_tokens",
minimum=1,
)
descriptor = registration.resolve_model_descriptor(
provider_model_id,
api_mode,
context_tokens=context_tokens,
max_output_tokens=max_output_tokens,
declared_capabilities=capabilities,
)
max_inflight = limits.get("max_inflight_requests")
if max_inflight is not None:
max_inflight = _integer(
max_inflight,
f"{model_name}.limits.max_inflight_requests",
minimum=1,
maximum=provider_defaults["max_inflight_requests"],
)
parameters = _mapping(
model_raw.get("parameters") or {}, f"{model_name}.parameters"
)
_strict_keys(
parameters,
allowed={
"defaults",
"purpose_overrides",
"user_options",
"constraints",
"reasoning_policy",
},
name=f"{model_name}.parameters",
)
defaults = dict(
_mapping(
parameters.get("defaults") or {},
f"{model_name}.parameters.defaults",
)
)
registration.validate_parameters(
defaults, path=f"{model_name}.parameters.defaults"
)
overrides_raw = _mapping(
parameters.get("purpose_overrides") or {},
f"{model_name}.parameters.purpose_overrides",
)
if not set(overrides_raw) <= set(_PURPOSES):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unknown purpose override"
)
overrides = {}
for purpose, values in overrides_raw.items():
parsed_values = dict(
_mapping(
values, f"{model_name}.parameters.purpose_overrides.{purpose}"
)
)
registration.validate_parameters(
parsed_values,
path=f"{model_name}.parameters.purpose_overrides.{purpose}",
)
overrides[purpose] = parsed_values
user_options_raw = _mapping(
parameters.get("user_options") or {},
f"{model_name}.parameters.user_options",
)
unknown_options = set(user_options_raw) - set(
registration.all_parameter_schema
)
if unknown_options:
raise EvoRuntimeError(
"MODEL_PARAMETER_INVALID",
"user_options contains an unsupported parameter",
)
user_options = {}
for key, value in user_options_raw.items():
option_path = f"{model_name}.parameters.user_options.{key}"
option = dict(_mapping(value, option_path))
_strict_keys(
option,
allowed={
"default",
"applies_to",
"minimum",
"maximum",
"minimum_exclusive",
"maximum_exclusive",
"choices",
},
name=option_path,
)
rule = registration.all_parameter_schema[key]
option = dict(_normalize_v3_user_option(option, rule, option_path))
applies_to = _string_set(
option.get("applies_to") or ["main_agent"],
f"{option_path}.applies_to",
allowed=frozenset(_PURPOSES),
)
user_options[key] = {
**option,
"type": rule.kind,
"applies_to": applies_to,
}
constraints = tuple(
dict(_mapping(value, f"{model_name}.parameters.constraints"))
for value in _sequence(
parameters.get("constraints") or [],
f"{model_name}.parameters.constraints",
)
)
reasoning_policy = _mapping(
parameters.get("reasoning_policy") or {},
f"{model_name}.parameters.reasoning_policy",
)
_strict_keys(
reasoning_policy,
allowed={"mode", "allowed_efforts", "default_effort"},
name=f"{model_name}.parameters.reasoning_policy",
)
access = _parse_v3_access(model_raw.get("access"), f"{model_name}.access")
quote = _parse_v3_billing(
model_raw.get("billing"),
f"{model_name}.billing",
require_evidence=require_evidence,
)
# A declared model capability is usable only when the selected
# Adapter has a concrete wire-level reasoning implementation.
supports_reasoning = (
capabilities["thinking"] and registration.reasoning_mode != "none"
)
if reasoning_policy and not supports_reasoning:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
f"{model_name}.parameters.reasoning_policy requires reasoning capability",
)
reasoning_mode = registration.reasoning_mode
allowed_reasoning_efforts: tuple[str, ...] = ()
default_reasoning_effort: str | None = None
if supports_reasoning:
reasoning_mode = str(
reasoning_policy.get("mode") or registration.reasoning_mode
)
if reasoning_mode not in {"boolean", "effort"}:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
f"{model_name}.parameters.reasoning_policy.mode is invalid",
)
allowed_reasoning_efforts = _string_set(
reasoning_policy.get("allowed_efforts")
or ["low", "medium", "high"],
f"{model_name}.parameters.reasoning_policy.allowed_efforts",
allowed=_REASONING_EFFORTS - {"disabled"},
)
default_reasoning_effort = str(
reasoning_policy.get("default_effort") or "medium"
)
if default_reasoning_effort not in allowed_reasoning_efforts:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
f"{model_name}.parameters.reasoning_policy.default_effort is not allowed",
)
registration.validate_parameters(
{"reasoning": default_reasoning_effort},
path=f"{model_name}.parameters.reasoning_policy",
)
constraints = validate_parameter_constraints(
constraints,
allowed_names=set(user_options)
| ({"reasoning"} if supports_reasoning else set()),
)
models[model_key] = ModelConfig(
provider_model_id,
defaults,
capabilities["vision"],
supports_reasoning,
allowed_reasoning_efforts,
descriptor.context_tokens,
descriptor.max_output_tokens,
reasoning_mode,
{"reasoning": default_reasoning_effort}
if default_reasoning_effort
else {},
{"reasoning": "off"},
(),
tuple(access["roles"]),
quote,
model_key=model_key,
display_name=_text(
model_raw.get("display_name"), f"{model_name}.display_name"
),
description=str(model_raw.get("description") or ""),
tags=_string_set(model_raw.get("tags") or [], f"{model_name}.tags"),
enabled=_bool(model_raw.get("enabled"), f"{model_name}.enabled"),
version_policy=policy,
resolved_model_revision=resolved,
reproducible=policy == "pinned",
capabilities=capabilities,
purpose_overrides=overrides,
user_options=user_options,
parameter_constraints=constraints,
access=access,
max_inflight_requests=max_inflight,
descriptor_parameters=descriptor.parameters,
)
endpoint = EndpointConfig(
provider_id,
base_url,
SecretReference(credential_ref, secret_version),
{},
{},
{},
)
providers[provider_id] = ProviderConfig(
provider_id,
adapter_id,
{},
{provider_id: endpoint},
models,
display_name=_text(item.get("display_name"), f"{name}.display_name"),
adapter_id=adapter_id,
adapter_revision=exact_revision,
wire_protocol=wire_protocol,
enabled=_bool(item.get("enabled"), f"{name}.enabled"),
connection_defaults=provider_defaults,
implementation_fingerprint=registration.implementation_fingerprint,
)
aliases_raw = _sequence(raw.get("aliases"), "aliases", allow_empty=False)
aliases: dict[str, AliasConfig] = {}
selectors: dict[str, RouteSelector] = {}
selectable: dict[str, str] = {}
for index, alias_value in enumerate(aliases_raw):
name = f"aliases[{index}]"
item = _mapping(alias_value, name)
_strict_keys(
item,
allowed={
"alias",
"display_name",
"provider_ref",
"model_ref",
"enabled",
"access",
"defaults",
},
name=name,
)
alias = _text(item.get("alias"), f"{name}.alias")
if alias in aliases:
raise EvoRuntimeError("LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate alias")
provider_ref = _text(item.get("provider_ref"), f"{name}.provider_ref")
model_ref = _text(item.get("model_ref"), f"{name}.model_ref")
if (
provider_ref not in providers
or model_ref not in providers[provider_ref].models
):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "alias target does not exist"
)
access = _parse_v3_access(item.get("access"), f"{name}.access")
defaults = dict(_mapping(item.get("defaults") or {}, f"{name}.defaults"))
registration = registry.get(
providers[provider_ref].adapter_id, providers[provider_ref].adapter_revision
)
registration.validate_parameters(defaults, path=f"{name}.defaults")
model = providers[provider_ref].models[model_ref]
if not set(defaults) <= set(model.user_options):
raise EvoRuntimeError(
"MODEL_PARAMETER_INVALID", "alias defaults must use model user_options"
)
revision_key = hashlib.sha256(
str(model.resolved_model_revision or model.model_id).encode()
).hexdigest()[:12]
identity_selector = f"direct:{provider_ref}:{model_ref}:{revision_key}:{_find_model_api_mode(raw, provider_ref, model_ref)}:{_find_model_transport(raw, provider_ref, model_ref)}"
internal_selector = f"{identity_selector}:alias:{alias}"
selector = RouteSelector(
internal_selector,
provider_ref,
provider_ref,
None,
model_ref,
_find_model_api_mode(raw, provider_ref, model_ref),
_find_model_transport(raw, provider_ref, model_ref),
identity_selector_id=identity_selector,
alias=alias,
alias_defaults=defaults,
)
selectors[internal_selector] = selector
enabled = _bool(item.get("enabled"), f"{name}.enabled")
aliases[alias] = AliasConfig(
alias,
_text(item.get("display_name"), f"{name}.display_name"),
provider_ref,
model_ref,
enabled,
access,
defaults,
)
if enabled:
selectable[alias] = internal_selector
purpose_defaults_raw = _mapping(
raw.get("purpose_defaults") or {}, "purpose_defaults"
)
if set(purpose_defaults_raw) != set(_PURPOSES):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"purpose_defaults must define all purposes",
)
purpose_defaults = {
key: dict(_mapping(value, f"purpose_defaults.{key}"))
for key, value in purpose_defaults_raw.items()
}
purpose_routes = _mapping(raw.get("purpose_routes"), "purpose_routes")
if set(purpose_routes) != set(_PURPOSES):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"purpose_routes must define all purposes",
)
main_route = _mapping(purpose_routes["main_agent"], "purpose_routes.main_agent")
_strict_keys(
main_route, allowed={"default_alias"}, name="purpose_routes.main_agent"
)
default_alias = _text(
main_route.get("default_alias"), "purpose_routes.main_agent.default_alias"
)
if default_alias not in selectable:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "default alias is unavailable"
)
purpose_selector_ids: dict[str, str] = {}
for purpose in ("tool_selector", "deepagents_summarizer"):
route = purpose_routes[purpose]
if route != "inherit_main" and not (
isinstance(route, Mapping)
and set(route) == {"default_alias"}
and route["default_alias"] in selectable
):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{purpose} route is invalid"
)
if isinstance(route, Mapping):
purpose_selector_ids[purpose] = selectable[str(route["default_alias"])]
title_route = _mapping(purpose_routes["title"], "purpose_routes.title")
_strict_keys(title_route, allowed={"default_alias"}, name="purpose_routes.title")
title_alias = _text(
title_route.get("default_alias"), "purpose_routes.title.default_alias"
)
if title_alias not in selectable:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "title alias is unavailable"
)
limits_raw = _mapping(raw.get("purpose_call_limits"), "purpose_call_limits")
if set(limits_raw) != set(_PURPOSES):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"purpose_call_limits must define all purposes",
)
purpose_limits = {}
for purpose, value in limits_raw.items():
item = _mapping(value, f"purpose_call_limits.{purpose}")
_strict_keys(
item,
allowed={"max_output_tokens", "max_attempts_per_run"},
name=f"purpose_call_limits.{purpose}",
)
purpose_limits[purpose] = PurposeCallLimit(
_integer(
item.get("max_attempts_per_run"),
f"purpose_call_limits.{purpose}.max_attempts_per_run",
minimum=1,
maximum=8,
),
)
health = _mapping(raw.get("health_policy") or {}, "health_policy")
_strict_keys(
health, allowed={"provider_connection", "model_route"}, name="health_policy"
)
provider_health = _parse_v3_health(
health.get("provider_connection"), "health_policy.provider_connection"
)
model_health = _parse_v3_health(
health.get("model_route"), "health_policy.model_route"
)
config = EvoModelConfig(
revision,
identity_key_id,
runtime_defaults,
purpose_defaults,
providers,
{},
model_health,
selectors,
PurposeRoutes(default_alias, selectable),
selectable[title_alias],
purpose_limits,
_parse_v3_web_runtime(raw.get("web_runtime")),
{},
{},
dict(raw),
schema_version=3,
aliases=aliases,
provider_health=provider_health,
adapter_registry_revision=registry.registry_revision,
purpose_selector_ids=purpose_selector_ids,
)
config._validate(require_evidence=require_evidence)
return config
def _find_v3_model(
raw: Mapping[str, Any], provider_id: str, model_key: str
) -> Mapping[str, Any]:
for provider in raw.get("providers") or []:
if provider.get("provider_id") == provider_id:
for model in provider.get("models") or []:
if model.get("model_key") == model_key:
return model
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "model target does not exist"
)
def _find_model_api_mode(
raw: Mapping[str, Any], provider_id: str, model_key: str
) -> str:
return _text(
_mapping(
_find_v3_model(raw, provider_id, model_key).get("invocation"), "invocation"
).get("api_mode"),
"api_mode",
)
def _find_model_transport(
raw: Mapping[str, Any], provider_id: str, model_key: str
) -> str:
return _text(
_mapping(
_find_v3_model(raw, provider_id, model_key).get("invocation"), "invocation"
).get("tool_call_transport"),
"tool_call_transport",
)
def _parse_v3_evidence(
value: Any, config: EvoModelConfig
) -> Mapping[str, CapabilityEvidence]:
result: dict[str, CapabilityEvidence] = {}
for index, evidence_value in enumerate(_sequence(value, "capability_evidence")):
name = f"capability_evidence[{index}]"
raw = _mapping(evidence_value, name)
allowed = {
"provider_ref",
"model_ref",
"adapter_id",
"adapter_revision",
"implementation_fingerprint",
"wire_protocol",
"provider_model_id",
"resolved_model_revision",
"version_policy",
"reproducible",
"api_mode",
"tool_call_transport",
"base_url_fingerprint",
"secret_version",
"route_semantics_hash",
"fixture_digest",
"verified_at",
"evidence_expires_at",
"results",
}
_strict_keys(raw, allowed=allowed, name=name)
provider_ref = _text(raw.get("provider_ref"), f"{name}.provider_ref")
model_ref = _text(raw.get("model_ref"), f"{name}.model_ref")
matches = [
selector
for selector in config.route_selectors.values()
if selector.provider == provider_ref and selector.model == model_ref
]
if not matches:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "evidence target does not exist"
)
route = config.concrete_routes(matches[0].selector_id)[0]
results_raw = _mapping(raw.get("results"), f"{name}.results")
valid_statuses = {"supported", "failed", "not_verified", "not_declared"}
results = {}
for key, status in results_raw.items():
status_text = _text(status, f"{name}.results.{key}")
if status_text not in valid_statuses:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid evidence status"
)
results[key] = status_text
model = config.providers[provider_ref].models[model_ref]
evidence = CapabilityEvidence(
route,
results.get("connectivity", "failed"),
results.get("tools", "not_declared"),
_text(raw.get("route_semantics_hash"), f"{name}.route_semantics_hash"),
_text(raw.get("base_url_fingerprint"), f"{name}.base_url_fingerprint"),
config.config_identity_key_id,
_text(raw.get("adapter_revision"), f"{name}.adapter_revision"),
_text(raw.get("fixture_digest"), f"{name}.fixture_digest"),
_text(raw.get("verified_at"), f"{name}.verified_at"),
results=results,
evidence_expires_at=_text(
raw.get("evidence_expires_at"), f"{name}.evidence_expires_at"
),
implementation_fingerprint=_text(
raw.get("implementation_fingerprint"),
f"{name}.implementation_fingerprint",
),
resolved_model_revision=str(raw.get("resolved_model_revision") or "")
or None,
reproducible=_bool(raw.get("reproducible"), f"{name}.reproducible"),
)
if (
raw.get("provider_model_id") != model.model_id
or raw.get("adapter_id") != config.providers[provider_ref].adapter_id
):
raise EvoRuntimeError("CAPABILITY_EVIDENCE_STALE")
result[route.key()] = evidence
return result
def _parse_secret(value: Any, name: str) -> SecretReference:
raw = _mapping(value, name)
_strict_keys(raw, allowed={"ref", "revision"}, name=name)
ref = _text(raw.get("ref"), f"{name}.ref")
revision = _integer(raw.get("revision"), f"{name}.revision", minimum=1)
if ref.startswith("env://"):
if not _ENV_RE.fullmatch(ref[6:]):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name}.ref is invalid"
)
elif ref.startswith("secret://"):
if "#" not in ref[9:] or not ref.rsplit("#", 1)[1]:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name}.ref needs a version"
)
else:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name}.ref scheme is unsupported"
)
return SecretReference(ref, revision)
def _normalize_base_url(value: Any, name: str) -> str:
return str(value or "").strip().rstrip("/")
def _parse_providers(value: Any) -> Mapping[str, ProviderConfig]:
raw = _mapping(value, "providers")
if not raw:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "providers must not be empty"
)
result: dict[str, ProviderConfig] = {}
for provider_key, provider_value in raw.items():
key = _text(provider_key, "provider key")
provider = _mapping(provider_value, f"providers.{key}")
_strict_keys(
provider,
allowed={"protocol", "params", "endpoints", "models"},
name=f"providers.{key}",
)
protocol = _text(provider.get("protocol"), f"providers.{key}.protocol")
endpoints: dict[str, EndpointConfig] = {}
for index, value_item in enumerate(
_sequence(provider.get("endpoints"), "endpoints", allow_empty=False)
):
item = _mapping(value_item, f"providers.{key}.endpoints[{index}]")
_strict_keys(
item,
allowed={
"name",
"base_url",
"auth",
"headers",
"header_refs",
"params",
},
name=f"providers.{key}.endpoints[{index}]",
)
endpoint_name = _text(item.get("name"), "endpoint.name")
if endpoint_name in endpoints:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate endpoint"
)
headers = dict(_mapping(item.get("headers") or {}, "endpoint.headers"))
header_refs_raw = _mapping(
item.get("header_refs") or {}, "endpoint.header_refs"
)
parsed_headers: dict[str, str] = {}
for header, header_value in headers.items():
if (
not _HEADER_RE.fullmatch(header)
or header.lower() in _SENSITIVE_HEADERS
):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid static header"
)
text = _text(header_value, f"headers.{header}")
if any(char in text for char in "\r\n\0"):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid header value"
)
parsed_headers[header] = text
parsed_header_refs: dict[str, SecretReference] = {}
for header, reference in header_refs_raw.items():
if not _HEADER_RE.fullmatch(header):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid secret header"
)
if header.lower() in {name.lower() for name in parsed_headers}:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate header"
)
parsed_header_refs[header] = _parse_secret(
reference, f"header_refs.{header}"
)
endpoints[endpoint_name] = EndpointConfig(
endpoint_name,
_normalize_base_url(item.get("base_url"), "endpoint.base_url"),
_parse_secret(item.get("auth"), "endpoint.auth"),
parsed_headers,
parsed_header_refs,
dict(_mapping(item.get("params") or {}, "endpoint.params")),
)
models: dict[str, ModelConfig] = {}
for index, value_item in enumerate(
_sequence(provider.get("models"), "models", allow_empty=False)
):
item = _mapping(value_item, f"providers.{key}.models[{index}]")
_strict_keys(
item,
allowed={
"id",
"params",
"supports_vision",
"supports_reasoning",
"allowed_reasoning_efforts",
"context_window",
"max_output_tokens",
"reasoning",
"access",
"billing",
},
name=f"providers.{key}.models[{index}]",
)
model_id = _text(item.get("id"), "model.id")
if model_id in models:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate model id"
)
access = _mapping(item.get("access"), "model.access")
_strict_keys(
access, allowed={"allowed_plans", "allowed_roles"}, name="model.access"
)
billing = _mapping(item.get("billing"), "model.billing")
_strict_keys(
billing,
allowed={
"sku",
"pricing_revision",
"currency",
"unit_scale",
"input_microunits_per_million",
"output_microunits_per_million",
"cached_microunits_per_million",
"multiplier",
},
name="model.billing",
)
quote_payload = {
"billing_sku": _text(billing.get("sku"), "billing.sku"),
"pricing_revision": _text(
billing.get("pricing_revision"), "billing.pricing_revision"
),
"currency": _text(billing.get("currency"), "billing.currency"),
"unit_scale": _integer(
billing.get("unit_scale"), "billing.unit_scale", minimum=1
),
"input_microunits_per_million": _integer(
billing.get("input_microunits_per_million"), "billing.input"
),
"cached_input_microunits_per_million": _integer(
billing.get("cached_microunits_per_million"), "billing.cached"
),
"output_microunits_per_million": _integer(
billing.get("output_microunits_per_million"), "billing.output"
),
"multiplier": _positive_decimal(
billing.get("multiplier"), "billing.multiplier"
),
}
if (
quote_payload["currency"] != "CNY"
or quote_payload["unit_scale"] != 1_000_000
):
raise EvoRuntimeError("PRICING_DIMENSION_UNSUPPORTED")
quote = PricingQuote(**quote_payload, quote_id=sha256_id(quote_payload))
supports_reasoning = _bool(
item.get("supports_reasoning"), "model.supports_reasoning"
)
efforts = _string_set(
item.get("allowed_reasoning_efforts"),
"model.allowed_reasoning_efforts",
allowed=_REASONING_EFFORTS - {"disabled"},
)
if bool(efforts) != supports_reasoning:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"reasoning capability is inconsistent",
)
reasoning = _parse_model_reasoning(
item.get("reasoning"), supports_reasoning=supports_reasoning
)
context_window = _integer(
item.get("context_window"), "model.context_window", minimum=1
)
max_output_tokens = _integer(
item.get("max_output_tokens", context_window),
"model.max_output_tokens",
minimum=1,
maximum=context_window,
)
models[model_id] = ModelConfig(
model_id,
dict(_mapping(item.get("params") or {}, "model.params")),
_bool(item.get("supports_vision"), "model.supports_vision"),
supports_reasoning,
efforts,
context_window,
max_output_tokens,
reasoning["mode"],
reasoning["enabled_params"],
reasoning["disabled_params"],
_string_set(access.get("allowed_plans"), "allowed_plans"),
_string_set(access.get("allowed_roles"), "allowed_roles"),
quote,
)
result[key] = ProviderConfig(
key,
protocol,
dict(_mapping(provider.get("params") or {}, "provider.params")),
endpoints,
models,
)
return result
def _parse_model_reasoning(
value: Any, *, supports_reasoning: bool
) -> Mapping[str, Any]:
"""Parse provider-specific reasoning controls without exposing them to users."""
if value is None:
return {
"mode": "effort" if supports_reasoning else "boolean",
"enabled_params": {},
"disabled_params": {},
}
raw = _mapping(value, "model.reasoning")
_strict_keys(
raw,
allowed={"mode", "enabled_params", "disabled_params"},
name="model.reasoning",
)
mode = _text(raw.get("mode"), "model.reasoning.mode")
if mode not in _REASONING_MODES:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid reasoning mode"
)
if not supports_reasoning:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"non-reasoning model cannot define reasoning controls",
)
return {
"mode": mode,
"enabled_params": dict(
_mapping(raw.get("enabled_params") or {}, "model.reasoning.enabled_params")
),
"disabled_params": dict(
_mapping(
raw.get("disabled_params") or {}, "model.reasoning.disabled_params"
)
),
}
def _parse_endpoint_pools(
value: Any, providers: Mapping[str, ProviderConfig]
) -> Mapping[str, EndpointPool]:
raw = _mapping(value, "endpoint_pools")
result: dict[str, EndpointPool] = {}
for pool_key, pool_value in raw.items():
pool_id = _text(pool_key, "pool id")
pool = _mapping(pool_value, f"endpoint_pools.{pool_id}")
_strict_keys(
pool,
allowed={"provider", "strategy", "endpoints"},
name=f"endpoint_pools.{pool_id}",
)
provider = _text(pool.get("provider"), "pool.provider")
if provider not in providers:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unknown pool provider"
)
strategy = _text(pool.get("strategy"), "pool.strategy")
if strategy != "smooth_weighted_round_robin":
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unsupported pool strategy"
)
members: list[PoolEndpoint] = []
seen: set[str] = set()
for item_value in _sequence(
pool.get("endpoints"), "pool.endpoints", allow_empty=False
):
item = _mapping(item_value, "pool endpoint")
_strict_keys(item, allowed={"name", "weight"}, name="pool endpoint")
name = _text(item.get("name"), "pool endpoint.name")
if name in seen or name not in providers[provider].endpoints:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid pool endpoint"
)
seen.add(name)
members.append(
PoolEndpoint(
name,
_integer(item.get("weight"), "pool endpoint.weight", minimum=1),
)
)
result[pool_id] = EndpointPool(pool_id, provider, strategy, tuple(members))
return result
def _parse_selectors(
value: Any,
providers: Mapping[str, ProviderConfig],
pools: Mapping[str, EndpointPool],
) -> Mapping[str, RouteSelector]:
raw = _mapping(value, "route_selectors")
result: dict[str, RouteSelector] = {}
for selector_key, selector_value in raw.items():
selector_id = _text(selector_key, "selector id")
item = _mapping(selector_value, f"route_selectors.{selector_id}")
_strict_keys(
item,
allowed={
"provider",
"endpoint",
"endpoint_pool",
"model",
"api_mode",
"tool_call_transport",
},
name=f"route_selectors.{selector_id}",
)
provider_id = _text(item.get("provider"), "selector.provider")
provider = providers.get(provider_id)
if provider is None:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unknown selector provider"
)
endpoint = item.get("endpoint")
endpoint_pool = item.get("endpoint_pool")
if (endpoint is None) == (endpoint_pool is None):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"selector requires exactly one endpoint source",
)
endpoint_name = (
_text(endpoint, "selector.endpoint") if endpoint is not None else None
)
pool_id = (
_text(endpoint_pool, "selector.endpoint_pool")
if endpoint_pool is not None
else None
)
if endpoint_name is not None and endpoint_name not in provider.endpoints:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unknown selector endpoint"
)
if pool_id is not None and (
pool_id not in pools or pools[pool_id].provider != provider_id
):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "selector pool provider mismatch"
)
model_id = _text(item.get("model"), "selector.model")
if model_id not in provider.models:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unknown selector model"
)
api_mode = _text(item.get("api_mode"), "selector.api_mode")
allowed = _PARAM_ALLOWLIST.get((provider.protocol, api_mode))
if allowed is None:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "provider/api mode has no adapter"
)
transport = _text(
item.get("tool_call_transport"), "selector.tool_call_transport"
)
if transport != "non_streaming":
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"Web tool routes must be non_streaming",
)
_validate_params(
provider.params, name=f"providers.{provider_id}.params", allowed=allowed
)
for concrete_endpoint in provider.endpoints.values():
_validate_params(
concrete_endpoint.params, name="endpoint.params", allowed=allowed
)
_validate_params(
provider.models[model_id].params, name="model.params", allowed=allowed
)
result[selector_id] = RouteSelector(
selector_id,
provider_id,
endpoint_name,
pool_id,
model_id,
api_mode,
transport,
)
if not result:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "route_selectors must not be empty"
)
return result
def _parse_purpose_defaults(value: Any) -> Mapping[str, Mapping[str, str]]:
raw = _mapping(value, "purpose_defaults")
if set(raw) != set(_PURPOSES):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED",
"purpose_defaults must define all purposes",
)
result = {}
for purpose in _PURPOSES:
item = _mapping(raw[purpose], f"purpose_defaults.{purpose}")
_strict_keys(
item, allowed={"reasoning_effort"}, name=f"purpose_defaults.{purpose}"
)
effort = _text(item.get("reasoning_effort"), "reasoning_effort")
if effort not in _REASONING_EFFORTS:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid reasoning effort"
)
result[purpose] = {"reasoning_effort": effort}
return result
def _parse_purpose_routes(
value: Any, selectors: Mapping[str, RouteSelector]
) -> tuple[PurposeRoutes, str]:
raw = _mapping(value, "purpose_routes")
_strict_keys(raw, allowed={"main_agent", "title"}, name="purpose_routes")
main = _mapping(raw.get("main_agent"), "purpose_routes.main_agent")
_strict_keys(
main, allowed={"default_alias", "selectable"}, name="purpose_routes.main_agent"
)
selectable_raw = _mapping(main.get("selectable"), "main_agent.selectable")
selectable = {
_text(alias, "model alias"): _text(selector_id, "selector id")
for alias, selector_id in selectable_raw.items()
}
if not selectable or any(
selector not in selectors for selector in selectable.values()
):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid selectable route"
)
default_alias = _text(main.get("default_alias"), "main_agent.default_alias")
if default_alias not in selectable:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "default alias is not selectable"
)
title = _mapping(raw.get("title"), "purpose_routes.title")
_strict_keys(title, allowed={"default"}, name="purpose_routes.title")
title_selector = _text(title.get("default"), "title.default")
if title_selector not in selectors:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unknown title selector"
)
return PurposeRoutes(default_alias, selectable), title_selector
def _parse_call_limits(value: Any) -> Mapping[str, PurposeCallLimit]:
raw = _mapping(value, "purpose_call_limits")
if set(raw) != set(_PURPOSES):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "call limits must define all purposes"
)
result = {}
for purpose in _PURPOSES:
item = _mapping(raw[purpose], f"purpose_call_limits.{purpose}")
_strict_keys(
item,
allowed={"max_output_tokens", "max_attempts_per_run"},
name=f"purpose_call_limits.{purpose}",
)
result[purpose] = PurposeCallLimit(
_integer(
item.get("max_attempts_per_run"), "max_attempts_per_run", minimum=1
),
)
return result
def _parse_health(value: Any) -> RouteHealthPolicy:
raw = _mapping(value, "route_health")
_strict_keys(
raw,
allowed={
"failure_threshold",
"cooldown_seconds",
"half_open_max_inflight",
"counted_error_codes",
"open_immediately_error_codes",
},
name="route_health",
)
return RouteHealthPolicy(
_integer(raw.get("failure_threshold"), "failure_threshold", minimum=1),
_integer(raw.get("cooldown_seconds"), "cooldown_seconds", minimum=1),
_integer(
raw.get("half_open_max_inflight"), "half_open_max_inflight", minimum=1
),
_string_set(raw.get("counted_error_codes"), "counted_error_codes"),
_string_set(
raw.get("open_immediately_error_codes"), "open_immediately_error_codes"
),
)
def _parse_web_runtime(value: Any) -> WebRuntimePolicy:
raw = _mapping(value, "web_runtime")
names = {
"title_start_timeout_seconds",
"prepare_ttl_seconds",
"turn_lease_grace_seconds",
"active_run_timeout_seconds",
"max_run_journal_events",
"max_run_journal_bytes",
"max_prepared_runs_per_subject",
"max_prepared_runs_total",
}
_strict_keys(raw, allowed=names, name="web_runtime")
values = {
name: _integer(raw.get(name), f"web_runtime.{name}", minimum=1)
for name in names
}
if (
values["prepare_ttl_seconds"] > 120
or values["title_start_timeout_seconds"] > 300
):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "runtime TTL is out of range"
)
if values["max_prepared_runs_per_subject"] > values["max_prepared_runs_total"]:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "prepared run limits are inconsistent"
)
return WebRuntimePolicy(**values)
def _parse_fallbacks(
value: Any, selectors: Mapping[str, RouteSelector]
) -> Mapping[str, tuple[str, ...]]:
result: dict[str, tuple[str, ...]] = {}
for item_value in _sequence(value, "tool_protocol_fallbacks"):
item = _mapping(item_value, "tool_protocol_fallback")
_strict_keys(
item, allowed={"primary", "fallbacks"}, name="tool_protocol_fallback"
)
primary = _text(item.get("primary"), "fallback.primary")
fallbacks = tuple(
_text(entry, "fallback selector")
for entry in _sequence(item.get("fallbacks"), "fallbacks")
)
if (
primary in result
or primary not in selectors
or any(entry not in selectors for entry in fallbacks)
):
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid fallback selector"
)
result[primary] = fallbacks
return result
def _route_from_evidence(value: Any, config: EvoModelConfig) -> RouteRef:
raw = _mapping(value, "evidence.route")
_strict_keys(
raw,
allowed={"provider", "endpoint", "model", "api_mode", "tool_call_transport"},
name="evidence.route",
)
matches = [
route
for selector_id in config.route_selectors
for route in config.concrete_routes(selector_id)
if route.provider == raw.get("provider")
and route.endpoint == raw.get("endpoint")
and route.model == raw.get("model")
and route.api_mode == raw.get("api_mode")
and route.tool_call_transport == raw.get("tool_call_transport")
]
if len(matches) != 1:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "evidence route is ambiguous or unknown"
)
return matches[0]
def _parse_evidence(
value: Any, config: EvoModelConfig
) -> Mapping[str, CapabilityEvidence]:
result: dict[str, CapabilityEvidence] = {}
for item_value in _sequence(value, "capability_evidence"):
item = _mapping(item_value, "capability_evidence item")
_strict_keys(
item,
allowed={"route", "connectivity", "tool_capability", "probe"},
name="capability_evidence item",
)
route = _route_from_evidence(item.get("route"), config)
probe = _mapping(item.get("probe"), "capability_evidence.probe")
_strict_keys(
probe,
allowed={
"route_semantics_hash",
"endpoint_fingerprint",
"config_identity_key_id",
"adapter_revision",
"fixture_digest",
"verified_at",
},
name="capability_evidence.probe",
)
evidence = CapabilityEvidence(
route,
_text(item.get("connectivity"), "connectivity"),
_text(item.get("tool_capability"), "tool_capability"),
_text(probe.get("route_semantics_hash"), "route_semantics_hash"),
_text(probe.get("endpoint_fingerprint"), "endpoint_fingerprint"),
_text(probe.get("config_identity_key_id"), "config_identity_key_id"),
_text(probe.get("adapter_revision"), "adapter_revision"),
_text(probe.get("fixture_digest"), "fixture_digest"),
_text(probe.get("verified_at"), "verified_at"),
)
if route.key() in result:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate capability evidence"
)
result[route.key()] = evidence
return result
def route_semantics_payload(
config: EvoModelConfig, route: RouteRef
) -> Mapping[str, Any]:
provider = config.providers[route.provider]
endpoint = provider.endpoints[route.endpoint]
model = provider.models[route.model]
if config.schema_version == 3:
return {
"schema_version": 3,
"config_revision": config.config_revision,
"provider_id": provider.key,
"adapter_id": provider.adapter_id,
"adapter_revision": provider.adapter_revision,
"implementation_fingerprint": provider.implementation_fingerprint,
"wire_protocol": provider.wire_protocol,
"base_url": endpoint.base_url,
"credential_ref": endpoint.auth.ref,
"model_key": route.model,
"provider_model_id": model.model_id,
"version_policy": model.version_policy,
"resolved_model_revision": model.resolved_model_revision,
"api_mode": route.api_mode,
"tool_call_transport": route.tool_call_transport,
# Capability evidence is reusable by every alias that targets this
# provider/model. Alias, purpose and user parameters belong to the
# invocation fingerprint produced by the runtime.
"params": dict(model.params),
}
allowed = _PARAM_ALLOWLIST[(provider.protocol, route.api_mode)]
return {
"provider": provider.key,
"protocol": provider.protocol,
"endpoint": endpoint.name,
"base_url": endpoint.base_url,
"auth": asdict(endpoint.auth),
"headers": dict(endpoint.headers),
"header_refs": {
key: asdict(value) for key, value in endpoint.header_refs.items()
},
"model": model.model_id,
"model_capabilities": {
"context_window": model.context_window,
"max_output_tokens": model.max_output_tokens,
"supports_vision": model.supports_vision,
"supports_reasoning": model.supports_reasoning,
"allowed_reasoning_efforts": model.allowed_reasoning_efforts,
"reasoning_mode": model.reasoning_mode,
"reasoning_enabled_params": model.reasoning_enabled_params,
"reasoning_disabled_params": model.reasoning_disabled_params,
},
"params": {
"provider": _validate_params(
provider.params, name="provider.params", allowed=allowed
),
"endpoint": _validate_params(
endpoint.params, name="endpoint.params", allowed=allowed
),
"model": _validate_params(
model.params, name="model.params", allowed=allowed
),
},
"api_mode": route.api_mode,
"tool_call_transport": route.tool_call_transport,
"adapter_revision": adapter_revision(provider.protocol, route.api_mode),
}
def adapter_revision(protocol: str, api_mode: str) -> str:
return (
f"ai4sci-provider-adapter-v3:{protocol}:{api_mode}:"
"bounds=model-cap-v1:reasoning=declarative-v1:"
"media=descriptor-v1:margin=table-v1"
)
def route_semantics_hash(
config: EvoModelConfig, route: RouteRef, identity_key: bytes
) -> str:
return hmac_id(identity_key, route_semantics_payload(config, route))
def route_fingerprint(
config: EvoModelConfig, route: RouteRef, identity_key: bytes
) -> str:
return hmac_id(
identity_key, {"route": route_semantics_payload(config, route), "kind": "route"}
)
def invocation_fingerprint(
config: EvoModelConfig,
route: RouteRef,
purpose: str,
final_params: Mapping[str, Any],
identity_key: bytes,
) -> str:
"""Bind an invocation identity to alias, purpose, and final parameters."""
return hmac_id(
identity_key,
{
"route": route_semantics_payload(config, route),
"kind": "invocation",
"alias": route.selector_id,
"purpose": purpose,
"final_params": dict(final_params),
},
)
def endpoint_fingerprint(
config: EvoModelConfig, route: RouteRef, identity_key: bytes
) -> str:
endpoint = config.providers[route.provider].endpoints[route.endpoint]
if config.schema_version == 3:
provider = config.providers[route.provider]
return hmac_id(
identity_key,
{
"provider_id": route.provider,
"base_url": endpoint.base_url,
"credential_ref": endpoint.auth.ref,
"adapter_revision": provider.adapter_revision,
},
)
return hmac_id(
identity_key, {"provider": route.provider, "endpoint": asdict(endpoint)}
)
_PROCESS_SECRET_FINGERPRINT_KEY = secrets.token_bytes(32)
def resolve_secret(
reference: SecretReference, *, secret_resolver: SecretResolver | None = None
) -> ResolvedSecret:
if secret_resolver is not None:
resolved = secret_resolver(reference)
elif reference.ref.startswith("env://"):
value = os.environ.get(reference.ref[6:], "")
if not value:
raise EvoRuntimeError("ROUTE_SECRET_UNAVAILABLE")
fingerprint = hmac.new(
_PROCESS_SECRET_FINGERPRINT_KEY, value.encode(), hashlib.sha256
).hexdigest()
resolved = ResolvedSecret(value, reference.revision, None, fingerprint)
else:
raise EvoRuntimeError("ROUTE_SECRET_UNAVAILABLE")
if resolved.declared_revision != reference.revision:
raise EvoRuntimeError("ROUTE_SECRET_UNAVAILABLE")
if reference.ref.startswith("secret://"):
expected = reference.ref.rsplit("#", 1)[1]
if resolved.authoritative_version != expected:
raise EvoRuntimeError("ROUTE_SECRET_UNAVAILABLE")
if any(char in resolved.value for char in "\r\n\0"):
raise EvoRuntimeError("ROUTE_SECRET_UNAVAILABLE")
return resolved
@dataclass(frozen=True, slots=True)
class ConfigRevision:
config_revision: int
@dataclass(frozen=True, slots=True)
class SaveModelConfigCommand:
expected_revision: int
payload: Mapping[str, Any]
admin_grant: AdminConfigGrant
operation_id: str = ""
class FileEvoModelConfigStore:
"""Integrity-checked V2/V3 store with a durable committed-head journal."""
def __init__(
self,
path: Path | None = None,
*,
admin_verifier: AdminConfigGrantVerifier | None = None,
ops_path: Path | None = None,
) -> None:
self.path = path or (get_config_dir() / "model_routes.yaml")
self._lock = FileLock(str(self.path) + ".lock")
self._admin_verifier = admin_verifier
self.ops_path = ops_path or self.path.with_name("model_config_ops.sqlite")
self._recovery_conflict = False
self._init_ops()
self._recover_preparing_operations()
def _connect(self) -> sqlite3.Connection:
connection = sqlite3.connect(self.ops_path)
connection.row_factory = sqlite3.Row
connection.execute("PRAGMA foreign_keys=ON")
connection.execute("PRAGMA busy_timeout=5000")
connection.execute("PRAGMA synchronous=FULL")
connection.execute("PRAGMA journal_mode=WAL")
return connection
def _init_ops(self) -> None:
self.ops_path.parent.mkdir(parents=True, exist_ok=True)
with self._connect() as connection:
connection.executescript(
"""
CREATE TABLE IF NOT EXISTS committed_head (
singleton INTEGER PRIMARY KEY CHECK (singleton = 1),
config_revision INTEGER NOT NULL,
canonical_payload_hash TEXT NOT NULL,
config_identity_key_id TEXT NOT NULL,
commit_operation_id TEXT NOT NULL
);
CREATE TABLE IF NOT EXISTS config_operations (
subject_id TEXT NOT NULL,
action TEXT NOT NULL,
operation_id TEXT NOT NULL,
request_digest TEXT NOT NULL,
result_digest TEXT,
status TEXT NOT NULL,
expected_revision INTEGER,
target_revision INTEGER,
proposal_hash TEXT,
payload_hash TEXT,
created_at INTEGER NOT NULL,
updated_at INTEGER NOT NULL,
PRIMARY KEY (subject_id, action, operation_id)
);
"""
)
columns = {
str(row[1])
for row in connection.execute("PRAGMA table_info(config_operations)")
}
if "canonical_payload" not in columns:
connection.execute(
"ALTER TABLE config_operations ADD COLUMN canonical_payload TEXT"
)
try:
os.chmod(self.ops_path, stat.S_IRUSR | stat.S_IWUSR)
except OSError:
pass
def load(self) -> EvoModelConfig:
with self._lock:
return self._load_locked()
def _load_locked(self) -> EvoModelConfig:
if self._recovery_conflict:
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
with self._connect() as connection:
head = connection.execute(
"SELECT * FROM committed_head WHERE singleton = 1"
).fetchone()
if not self.path.exists():
if head is None:
raise EvoRuntimeError(
"LLM_ROUTE_CONFIGURATION_REQUIRED", "model routes are unconfigured"
)
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
if head is None:
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
config = EvoModelConfig.parse(
load_yaml_unique(self.path.read_text(encoding="utf-8"))
)
payload_hash = sha256_id(config.raw)
if (
config.config_revision != int(head["config_revision"])
or config.config_identity_key_id != str(head["config_identity_key_id"])
or payload_hash != str(head["canonical_payload_hash"])
):
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
return config
def bootstrap_for_development(
self, payload: Mapping[str, Any], *, operation_id: str = "bootstrap"
) -> ConfigRevision:
"""Explicitly initialize an empty development store; never overwrites a head."""
with self._lock:
with self._connect() as connection:
if connection.execute(
"SELECT 1 FROM committed_head WHERE singleton = 1"
).fetchone():
raise EvoRuntimeError("CONFIG_REVISION_CONFLICT")
if self.path.exists():
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
candidate = dict(payload)
candidate["config_revision"] = 1
config = EvoModelConfig.parse(candidate)
self._write_committed(config.raw, operation_id=operation_id)
return ConfigRevision(1)
def save(self, command: SaveModelConfigCommand) -> ConfigRevision:
if self._admin_verifier is None:
raise EvoRuntimeError("ADMIN_CONFIG_FORBIDDEN")
self._admin_verifier.require_admin(command.admin_grant)
if command.admin_grant.action != "model_config:commit":
raise EvoRuntimeError("ADMIN_CONFIG_FORBIDDEN")
operation_id = command.operation_id or command.admin_grant.operation_id
if operation_id != command.admin_grant.operation_id:
raise EvoRuntimeError("IDEMPOTENCY_CONFLICT")
with self._lock:
current_revision = 0
with self._connect() as connection:
head = connection.execute(
"SELECT * FROM committed_head WHERE singleton = 1"
).fetchone()
if head is not None:
current_revision = int(head["config_revision"])
if current_revision != command.expected_revision:
raise EvoRuntimeError("CONFIG_REVISION_CONFLICT")
candidate = dict(command.payload)
candidate["config_revision"] = current_revision + 1
config = EvoModelConfig.parse(candidate)
request_digest = sha256_id(
{
"expected_revision": command.expected_revision,
"payload": command.payload,
}
)
if request_digest != command.admin_grant.request_digest:
raise EvoRuntimeError("CONTRACT_SIGNATURE_INVALID")
self._write_committed(config.raw, operation_id=operation_id)
return ConfigRevision(config.config_revision)
def current_revision(self) -> int:
with self._connect() as connection:
row = connection.execute(
"SELECT config_revision FROM committed_head WHERE singleton=1"
).fetchone()
return int(row[0]) if row is not None else 0
def load_revision(self, revision: int) -> EvoModelConfig:
with self._connect() as connection:
row = connection.execute(
"""SELECT canonical_payload FROM config_operations
WHERE action='model_config:commit' AND status='COMMITTED'
AND target_revision=? AND canonical_payload IS NOT NULL
ORDER BY updated_at DESC LIMIT 1""",
(revision,),
).fetchone()
if row is None:
raise EvoRuntimeError("CONFIG_REVISION_NOT_FOUND")
return EvoModelConfig.parse(json.loads(str(row["canonical_payload"])))
def commit_validated(
self,
payload: Mapping[str, Any],
*,
expected_revision: int,
operation_id: str,
) -> ConfigRevision:
"""Commit a payload already authorized and evidenced by config admin."""
with self._lock:
target = expected_revision + 1
candidate = dict(payload)
candidate["config_revision"] = target
config = EvoModelConfig.parse(candidate)
payload_hash = sha256_id(config.raw)
replay = self._committed_operation_revision(
operation_id, payload_hash=payload_hash, target_revision=target
)
if replay is not None:
return ConfigRevision(replay)
current = self.current_revision()
if current != expected_revision:
raise EvoRuntimeError("CONFIG_REVISION_CONFLICT")
self._write_committed(config.raw, operation_id=operation_id)
return ConfigRevision(config.config_revision)
def activate_revision(
self,
target_revision: int,
*,
expected_revision: int,
operation_id: str,
) -> ConfigRevision:
"""Atomically point the active head at an immutable committed revision."""
with self._lock:
if self.current_revision() != expected_revision:
raise EvoRuntimeError("CONFIG_REVISION_CONFLICT")
with self._connect() as connection:
replay = connection.execute(
"""SELECT status, target_revision FROM config_operations
WHERE subject_id='admin' AND action='model_config:rollback'
AND operation_id=?""",
(operation_id,),
).fetchone()
if replay is not None:
if int(replay["target_revision"] or 0) != target_revision:
raise EvoRuntimeError("IDEMPOTENCY_CONFLICT")
if str(replay["status"]) == "COMMITTED":
return ConfigRevision(target_revision)
source = connection.execute(
"""SELECT canonical_payload FROM config_operations
WHERE action='model_config:commit' AND status='COMMITTED'
AND target_revision=? AND canonical_payload IS NOT NULL
ORDER BY updated_at DESC LIMIT 1""",
(target_revision,),
).fetchone()
if source is None:
raise EvoRuntimeError("CONFIG_REVISION_NOT_FOUND")
payload = json.loads(str(source["canonical_payload"]))
config = EvoModelConfig.parse(payload)
now = time.time_ns() // 1_000_000
payload_hash = sha256_id(config.raw)
with self._connect() as connection:
connection.execute(
"""INSERT INTO config_operations
(subject_id, action, operation_id, request_digest, result_digest,
status, expected_revision, target_revision, payload_hash,
canonical_payload, created_at, updated_at)
VALUES ('admin', 'model_config:rollback', ?, ?, ?, 'PREPARING',
?, ?, ?, ?, ?, ?)""",
(
operation_id,
sha256_id(
{
"expected_revision": expected_revision,
"target_revision": target_revision,
}
),
sha256_id({"config_revision": target_revision}),
expected_revision,
target_revision,
payload_hash,
canonical_json_v1(config.raw).decode(),
now,
now,
),
)
self._atomic_replace_payload(config.raw)
with self._connect() as connection:
connection.execute(
"""INSERT INTO committed_head
(singleton, config_revision, canonical_payload_hash,
config_identity_key_id, commit_operation_id)
VALUES (1, ?, ?, ?, ?)
ON CONFLICT(singleton) DO UPDATE SET
config_revision=excluded.config_revision,
canonical_payload_hash=excluded.canonical_payload_hash,
config_identity_key_id=excluded.config_identity_key_id,
commit_operation_id=excluded.commit_operation_id""",
(
target_revision,
payload_hash,
config.config_identity_key_id,
operation_id,
),
)
connection.execute(
"""UPDATE config_operations SET status='COMMITTED', updated_at=?
WHERE subject_id='admin' AND action='model_config:rollback'
AND operation_id=?""",
(now, operation_id),
)
return ConfigRevision(target_revision)
def _committed_operation_revision(
self, operation_id: str, *, payload_hash: str, target_revision: int
) -> int | None:
subject_id = "development" if operation_id == "bootstrap" else "admin"
with self._connect() as connection:
row = connection.execute(
"""SELECT status, payload_hash, target_revision
FROM config_operations
WHERE subject_id=? AND action='model_config:commit' AND operation_id=?""",
(subject_id, operation_id),
).fetchone()
if row is None:
return None
if (
str(row["payload_hash"]) != payload_hash
or int(row["target_revision"] or 0) != target_revision
):
raise EvoRuntimeError("IDEMPOTENCY_CONFLICT")
if row["status"] == "COMMITTED":
return target_revision
if row["status"] == "CONFLICT":
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
return None
def _write_committed(
self, payload: Mapping[str, Any], *, operation_id: str
) -> None:
payload_hash = sha256_id(payload)
revision = int(payload["config_revision"])
identity_key_id = str(payload["config_identity_key_id"])
now = time.time_ns() // 1_000_000
subject_id = "development" if operation_id == "bootstrap" else "admin"
self.path.parent.mkdir(parents=True, exist_ok=True)
canonical_payload = canonical_json_v1(payload).decode("utf-8")
with self._connect() as connection:
existing = connection.execute(
"""SELECT status, payload_hash, target_revision
FROM config_operations
WHERE subject_id=? AND action='model_config:commit' AND operation_id=?""",
(subject_id, operation_id),
).fetchone()
if existing is not None:
if (
str(existing["payload_hash"]) != payload_hash
or int(existing["target_revision"] or 0) != revision
):
raise EvoRuntimeError("IDEMPOTENCY_CONFLICT")
if existing["status"] == "COMMITTED":
return
if existing["status"] == "CONFLICT":
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
else:
connection.execute(
"""INSERT INTO config_operations
(subject_id, action, operation_id, request_digest, status, expected_revision,
target_revision, payload_hash, canonical_payload, created_at, updated_at)
VALUES (?, 'model_config:commit', ?, ?, 'PREPARING', ?, ?, ?, ?, ?, ?)""",
(
subject_id,
operation_id,
payload_hash,
revision - 1,
revision,
payload_hash,
canonical_payload,
now,
now,
),
)
self._atomic_replace_payload(payload)
with self._connect() as connection:
connection.execute(
"""INSERT INTO committed_head
(singleton, config_revision, canonical_payload_hash, config_identity_key_id, commit_operation_id)
VALUES (1, ?, ?, ?, ?)
ON CONFLICT(singleton) DO UPDATE SET
config_revision=excluded.config_revision,
canonical_payload_hash=excluded.canonical_payload_hash,
config_identity_key_id=excluded.config_identity_key_id,
commit_operation_id=excluded.commit_operation_id""",
(revision, payload_hash, identity_key_id, operation_id),
)
connection.execute(
"""UPDATE config_operations
SET status='COMMITTED', result_digest=?, updated_at=?
WHERE subject_id=? AND action='model_config:commit' AND operation_id=?""",
(
sha256_id({"config_revision": revision}),
now,
subject_id,
operation_id,
),
)
def _atomic_replace_payload(self, payload: Mapping[str, Any]) -> None:
rendered = yaml.safe_dump(dict(payload), allow_unicode=True, sort_keys=False)
fd, temporary_name = tempfile.mkstemp(
prefix=".model_routes.", suffix=".tmp", dir=self.path.parent
)
temporary = Path(temporary_name)
try:
with os.fdopen(fd, "w", encoding="utf-8") as handle:
handle.write(rendered)
handle.flush()
os.fsync(handle.fileno())
os.chmod(temporary, stat.S_IRUSR | stat.S_IWUSR)
os.replace(temporary, self.path)
directory_fd = os.open(self.path.parent, os.O_RDONLY)
try:
os.fsync(directory_fd)
finally:
os.close(directory_fd)
finally:
temporary.unlink(missing_ok=True)
def _recover_preparing_operations(self) -> None:
"""Complete or safely replay interrupted file/head commits."""
with self._lock:
with self._connect() as connection:
rows = connection.execute(
"""SELECT * FROM config_operations
WHERE action IN ('model_config:commit', 'model_config:rollback')
AND status='PREPARING'
ORDER BY created_at"""
).fetchall()
for row in rows:
try:
payload_text = str(row["canonical_payload"] or "")
if not payload_text:
raise ValueError("missing recovery payload")
payload = json.loads(payload_text)
target = int(row["target_revision"])
payload_hash = str(row["payload_hash"])
if int(payload.get("config_revision", 0)) != target:
raise ValueError("recovery revision mismatch")
file_matches_target = False
if self.path.exists():
current_payload = load_yaml_unique(
self.path.read_text(encoding="utf-8")
)
file_matches_target = sha256_id(current_payload) == payload_hash
if not file_matches_target:
head_revision = self.current_revision()
if head_revision != int(row["expected_revision"] or 0):
raise ValueError("recovery head moved")
self._atomic_replace_payload(payload)
now = time.time_ns() // 1_000_000
with self._connect() as connection:
connection.execute(
"""INSERT INTO committed_head
(singleton, config_revision, canonical_payload_hash,
config_identity_key_id, commit_operation_id)
VALUES (1, ?, ?, ?, ?)
ON CONFLICT(singleton) DO UPDATE SET
config_revision=excluded.config_revision,
canonical_payload_hash=excluded.canonical_payload_hash,
config_identity_key_id=excluded.config_identity_key_id,
commit_operation_id=excluded.commit_operation_id""",
(
target,
payload_hash,
str(payload["config_identity_key_id"]),
str(row["operation_id"]),
),
)
connection.execute(
"""UPDATE config_operations
SET status='COMMITTED', result_digest=?, updated_at=?
WHERE subject_id=? AND action=? AND operation_id=?""",
(
sha256_id({"config_revision": target}),
now,
row["subject_id"],
row["action"],
row["operation_id"],
),
)
except Exception:
self._recovery_conflict = True
with self._connect() as connection:
connection.execute(
"""UPDATE config_operations SET status='CONFLICT', updated_at=?
WHERE subject_id=? AND action=? AND operation_id=?""",
(
time.time_ns() // 1_000_000,
row["subject_id"],
row["action"],
row["operation_id"],
),
)
def proposal_payload(
payload: Mapping[str, Any], *, target_revision: int
) -> Mapping[str, Any]:
candidate = dict(payload)
candidate["config_revision"] = target_revision
candidate.pop("capability_evidence", None)
return candidate
def proposal_hash(
payload: Mapping[str, Any], *, target_revision: int, identity_key: bytes
) -> str:
return hmac_id(
identity_key, proposal_payload(payload, target_revision=target_revision)
)
@dataclass(frozen=True, slots=True)
class V2MigrationReport:
converted_providers: tuple[str, ...]
converted_models: tuple[str, ...]
converted_aliases: tuple[str, ...]
blocking_issues: tuple[Mapping[str, str], ...]
warnings: tuple[Mapping[str, str], ...]
required_secret_mappings: tuple[Mapping[str, str], ...]
def convert_v2_to_v3_draft(
payload: Mapping[str, Any],
*,
target_revision: int,
config_identity_key_id: str,
) -> tuple[Mapping[str, Any], V2MigrationReport]:
"""Convert unambiguous V2 routes to a V3 Draft without guessing endpoints."""
source = EvoModelConfig.parse(payload, require_evidence=False)
if source.schema_version != 2:
raise EvoRuntimeError("LLM_ROUTE_CONFIGURATION_REQUIRED", "source must be V2")
adapter_map = {
"openai": ("openai", "openai-v1", "openai_native"),
"anthropic": ("anthropic", "anthropic-v1", "anthropic_native"),
"custom-openai": (
"generic-openai-compatible",
"generic-openai-compatible-v1",
"openai_compatible",
),
}
providers: list[Mapping[str, Any]] = []
aliases: list[Mapping[str, Any]] = []
converted_models: list[str] = []
blocking: list[Mapping[str, str]] = []
warnings: list[Mapping[str, str]] = []
secret_mappings: list[Mapping[str, str]] = []
model_keys: dict[tuple[str, str], str] = {}
for provider_id, provider in source.providers.items():
referenced = {
route.endpoint
for selector_id in source.route_selectors
for route in source.concrete_routes(selector_id)
if route.provider == provider_id
}
if len(referenced) != 1:
blocking.append(
{
"code": "MULTIPLE_ENDPOINTS_REQUIRE_SPLIT",
"provider_id": provider_id,
"detail": ",".join(sorted(referenced)),
}
)
continue
mapped = adapter_map.get(provider.protocol)
if mapped is None:
blocking.append(
{
"code": "ADAPTER_MAPPING_REQUIRED",
"provider_id": provider_id,
"detail": provider.protocol,
}
)
continue
adapter_id, revision, wire = mapped
if provider.protocol == "custom-openai":
warnings.append(
{
"code": "GENERIC_ADAPTER_REQUIRES_CONFIRMATION",
"provider_id": provider_id,
"detail": "Base URL was not used to infer a vendor adapter",
}
)
endpoint = provider.endpoints[next(iter(referenced))]
credential_ref = (
f"secret://model-providers/{provider_id}#{endpoint.auth.revision}"
)
secret_mappings.append(
{
"provider_id": provider_id,
"old_ref": endpoint.auth.ref,
"new_ref": credential_ref,
}
)
models: list[Mapping[str, Any]] = []
for index, (legacy_key, model) in enumerate(provider.models.items()):
model_key = re.sub(r"[^A-Za-z0-9._-]+", "-", legacy_key).strip("-")
model_key = model_key or f"model-{index + 1}"
if any(item["model_key"] == model_key for item in models):
model_key = f"{model_key}-{index + 1}"
model_keys[(provider_id, legacy_key)] = model_key
converted_models.append(f"{provider_id}/{model_key}")
selector = next(
(
item
for item in source.route_selectors.values()
if item.provider == provider_id and item.model == legacy_key
),
None,
)
models.append(
{
"model_key": model_key,
"provider_model_id": model.model_id,
"version_policy": "rolling",
"resolved_model_revision": None,
"display_name": model.model_id,
"description": "Migrated from model routes V2",
"enabled": True,
"tags": ["migrated-v2"],
"invocation": {
"api_mode": selector.api_mode
if selector
else "chat_completions",
"tool_call_transport": "native",
},
"capabilities": {
"text": True,
"vision": model.supports_vision,
"video": False,
"documents": False,
"tools": True,
"structured_output": False,
"thinking": model.supports_reasoning,
},
"limits": {
"context_tokens": model.context_window,
"max_output_tokens": model.max_output_tokens,
},
"parameters": {
"defaults": dict(model.params),
"purpose_overrides": {},
"user_options": {},
"constraints": [],
},
"access": {
"visibility": "role_based"
if model.allowed_roles
else "authenticated",
"roles": list(model.allowed_roles),
"groups": [],
"users": [],
},
"billing": {
"sku": model.quote.billing_sku,
"pricing_revision": model.quote.pricing_revision,
"currency": model.quote.currency,
"unit_scale": model.quote.unit_scale,
"input_microunits_per_million": model.quote.input_microunits_per_million,
"output_microunits_per_million": model.quote.output_microunits_per_million,
"cached_microunits_per_million": model.quote.cached_input_microunits_per_million,
"multiplier": model.quote.multiplier,
},
}
)
providers.append(
{
"provider_id": provider_id,
"display_name": provider_id,
"adapter_id": adapter_id,
"adapter_revision": revision,
"wire_protocol": wire,
"enabled": True,
"connection": {
"base_url": endpoint.base_url,
"credential_ref": credential_ref,
},
"defaults": {},
"models": models,
}
)
for alias, selector_id in source.main_routes.selectable.items():
selector = source.route_selectors[selector_id]
key = model_keys.get((selector.provider, selector.model))
if key is None:
continue
aliases.append(
{
"alias": alias,
"display_name": alias,
"provider_ref": selector.provider,
"model_ref": key,
"enabled": True,
"access": {
"visibility": "authenticated",
"roles": [],
"groups": [],
"users": [],
},
"defaults": {},
}
)
default_alias = (
source.main_routes.default_alias
if any(item["alias"] == source.main_routes.default_alias for item in aliases)
else (str(aliases[0]["alias"]) if aliases else "")
)
title_route = source.concrete_routes(source.title_selector_id)[0]
title_alias = next(
(
str(item["alias"])
for item in aliases
if item["provider_ref"] == title_route.provider
and item["model_ref"]
== model_keys.get((title_route.provider, title_route.model))
),
default_alias,
)
draft = {
"schema_version": 3,
"config_revision": target_revision,
"config_identity_key_id": config_identity_key_id,
"runtime_defaults": {},
"providers": providers,
"aliases": aliases,
"purpose_defaults": {purpose: {} for purpose in _PURPOSES},
"purpose_routes": {
"main_agent": {"default_alias": default_alias},
"tool_selector": "inherit_main",
"deepagents_summarizer": "inherit_main",
"title": {"default_alias": title_alias},
},
"purpose_call_limits": {
purpose: asdict(limit)
for purpose, limit in source.purpose_call_limits.items()
},
"health_policy": {
"provider_connection": asdict(source.route_health),
"model_route": asdict(source.route_health),
},
"web_runtime": asdict(source.web_runtime),
"capability_evidence": [],
}
return draft, V2MigrationReport(
converted_providers=tuple(str(item["provider_id"]) for item in providers),
converted_models=tuple(converted_models),
converted_aliases=tuple(str(item["alias"]) for item in aliases),
blocking_issues=tuple(blocking),
warnings=tuple(warnings),
required_secret_mappings=tuple(secret_mappings),
)