3310 lines
128 KiB
Python
3310 lines
128 KiB
Python
"""Strict Evo-owned V2/V3 model-route configuration."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import hmac
|
|
import json
|
|
import math
|
|
import os
|
|
import re
|
|
import secrets
|
|
import sqlite3
|
|
import stat
|
|
import tempfile
|
|
import time
|
|
from collections.abc import Mapping, Sequence
|
|
from dataclasses import asdict, dataclass, field
|
|
from decimal import Decimal, InvalidOperation
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import yaml
|
|
from filelock import FileLock
|
|
|
|
from ..config.settings import get_config_dir
|
|
from .adapter_registry import get_adapter_registry
|
|
from .configuration import (
|
|
EndpointConfig,
|
|
ModelConfig,
|
|
ProviderConfig,
|
|
ResolvedSecret,
|
|
SecretReference,
|
|
SecretResolver,
|
|
)
|
|
from .contracts import (
|
|
AdminConfigGrant,
|
|
AdminConfigGrantVerifier,
|
|
EvoRuntimeError,
|
|
PricingQuote,
|
|
)
|
|
from .crypto import canonical_json_v1, hmac_id, sha256_id
|
|
from .user_options import (
|
|
project_user_options_for_purpose,
|
|
validate_parameter_constraints,
|
|
)
|
|
|
|
_PURPOSES = ("main_agent", "tool_selector", "deepagents_summarizer", "title")
|
|
_REASONING_EFFORTS = frozenset({"disabled", "low", "medium", "high", "max"})
|
|
_REASONING_MODES = frozenset({"effort", "boolean"})
|
|
_BLOCKED_PARAM_KEYS = frozenset(
|
|
{
|
|
"api_key",
|
|
"base_url",
|
|
"max_tokens",
|
|
"max_output_tokens",
|
|
"max_retries",
|
|
"streaming",
|
|
"disable_streaming",
|
|
"use_responses_api",
|
|
"reasoning",
|
|
"token",
|
|
"secret",
|
|
"password",
|
|
"authorization",
|
|
}
|
|
)
|
|
_SENSITIVE_HEADERS = frozenset(
|
|
{"authorization", "proxy-authorization", "cookie", "set-cookie", "x-api-key"}
|
|
)
|
|
_HEADER_RE = re.compile(r"^[!#$%&'*+.^_`|~0-9A-Za-z-]+$")
|
|
_ENV_RE = re.compile(r"^[A-Z_][A-Z0-9_]*$")
|
|
_BIGINT_MAX = 2**63 - 1
|
|
_PARAM_ALLOWLIST: Mapping[tuple[str, str], frozenset[str]] = {
|
|
("custom-openai", "chat_completions"): frozenset(
|
|
{
|
|
"temperature",
|
|
"top_p",
|
|
"seed",
|
|
"frequency_penalty",
|
|
"presence_penalty",
|
|
"timeout",
|
|
"extra_body",
|
|
"output_token_limit",
|
|
}
|
|
),
|
|
("custom-openai", "responses"): frozenset(
|
|
{"temperature", "top_p", "seed", "timeout", "extra_body", "output_token_limit"}
|
|
),
|
|
("openai", "chat_completions"): frozenset(
|
|
{
|
|
"temperature",
|
|
"top_p",
|
|
"seed",
|
|
"frequency_penalty",
|
|
"presence_penalty",
|
|
"timeout",
|
|
"extra_body",
|
|
"output_token_limit",
|
|
}
|
|
),
|
|
("openai", "responses"): frozenset(
|
|
{"temperature", "top_p", "seed", "timeout", "extra_body", "output_token_limit"}
|
|
),
|
|
("anthropic", "messages"): frozenset(
|
|
{"temperature", "top_p", "top_k", "timeout", "output_token_limit"}
|
|
),
|
|
}
|
|
|
|
|
|
class _UniqueKeyLoader(yaml.SafeLoader):
|
|
pass
|
|
|
|
|
|
def _construct_mapping(
|
|
loader: yaml.SafeLoader, node: yaml.MappingNode, deep: bool = False
|
|
) -> Any:
|
|
loader.flatten_mapping(node)
|
|
result: dict[Any, Any] = {}
|
|
for key_node, value_node in node.value:
|
|
key = loader.construct_object(key_node, deep=deep)
|
|
if key in result:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"duplicate YAML key: {key}"
|
|
)
|
|
result[key] = loader.construct_object(value_node, deep=deep)
|
|
return result
|
|
|
|
|
|
_UniqueKeyLoader.add_constructor(
|
|
yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG,
|
|
_construct_mapping,
|
|
)
|
|
|
|
|
|
def load_yaml_unique(text: str) -> Any:
|
|
try:
|
|
return yaml.load(text, Loader=_UniqueKeyLoader)
|
|
except EvoRuntimeError:
|
|
raise
|
|
except yaml.YAMLError as exc:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid model route YAML"
|
|
) from exc
|
|
|
|
|
|
def _mapping(value: Any, name: str) -> Mapping[str, Any]:
|
|
if not isinstance(value, Mapping):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be a mapping"
|
|
)
|
|
if not all(isinstance(key, str) for key in value):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} keys must be strings"
|
|
)
|
|
return value
|
|
|
|
|
|
def _sequence(value: Any, name: str, *, allow_empty: bool = True) -> Sequence[Any]:
|
|
if not isinstance(value, list | tuple):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be a list"
|
|
)
|
|
if not allow_empty and not value:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must not be empty"
|
|
)
|
|
return value
|
|
|
|
|
|
def _strict_keys(value: Mapping[str, Any], *, allowed: set[str], name: str) -> None:
|
|
unknown = set(value) - allowed
|
|
if unknown:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
f"{name} has unknown fields: {', '.join(sorted(unknown))}",
|
|
)
|
|
|
|
|
|
def _text(value: Any, name: str) -> str:
|
|
if not isinstance(value, str) or not value.strip():
|
|
raise EvoRuntimeError("LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} is required")
|
|
return value.strip()
|
|
|
|
|
|
def _integer(
|
|
value: Any, name: str, *, minimum: int = 0, maximum: int = _BIGINT_MAX
|
|
) -> int:
|
|
if isinstance(value, bool) or not isinstance(value, int):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be an integer"
|
|
)
|
|
if value < minimum or value > maximum:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} is out of range"
|
|
)
|
|
return value
|
|
|
|
|
|
def _positive_decimal(value: Any, name: str, *, default: str = "1") -> str:
|
|
if value is None:
|
|
value = default
|
|
if isinstance(value, bool):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be a positive number"
|
|
)
|
|
try:
|
|
parsed = Decimal(str(value))
|
|
except (InvalidOperation, TypeError, ValueError) as exc:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be a positive number"
|
|
) from exc
|
|
if not parsed.is_finite() or parsed <= 0:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be greater than zero"
|
|
)
|
|
return format(parsed.normalize(), "f")
|
|
|
|
|
|
def _bool(value: Any, name: str) -> bool:
|
|
if not isinstance(value, bool):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} must be boolean"
|
|
)
|
|
return value
|
|
|
|
|
|
def _string_set(
|
|
value: Any, name: str, *, allowed: frozenset[str] | None = None
|
|
) -> tuple[str, ...]:
|
|
items = _sequence(value, name)
|
|
normalized = tuple(sorted({_text(item, name) for item in items}))
|
|
if allowed is not None and not set(normalized) <= allowed:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} contains unsupported values"
|
|
)
|
|
return normalized
|
|
|
|
|
|
def _validate_params(
|
|
value: Any, *, name: str, allowed: frozenset[str]
|
|
) -> Mapping[str, Any]:
|
|
params = dict(_mapping(value or {}, name))
|
|
unknown = set(params) - allowed
|
|
if unknown:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
f"{name} has unsupported parameters: {', '.join(sorted(unknown))}",
|
|
)
|
|
|
|
def visit(item: Any, path: str) -> None:
|
|
if item is None or isinstance(item, bool | int | float | str):
|
|
canonical_json_v1(item)
|
|
return
|
|
if isinstance(item, Mapping):
|
|
for key, child in item.items():
|
|
if not isinstance(key, str):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{path} key must be text"
|
|
)
|
|
if key.lower() in _BLOCKED_PARAM_KEYS:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{path}.{key} is forbidden"
|
|
)
|
|
visit(child, f"{path}.{key}")
|
|
return
|
|
if isinstance(item, list | tuple):
|
|
for index, child in enumerate(item):
|
|
visit(child, f"{path}[{index}]")
|
|
return
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{path} is not JSON data"
|
|
)
|
|
|
|
visit(params, name)
|
|
return params
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class AliasConfig:
|
|
alias: str
|
|
display_name: str
|
|
provider_ref: str
|
|
model_ref: str
|
|
enabled: bool
|
|
access: Mapping[str, Any]
|
|
defaults: Mapping[str, Any]
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class PoolEndpoint:
|
|
name: str
|
|
weight: int
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class EndpointPool:
|
|
pool_id: str
|
|
provider: str
|
|
strategy: LiteralStrategy
|
|
endpoints: tuple[PoolEndpoint, ...]
|
|
|
|
|
|
LiteralStrategy = str
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class RouteSelector:
|
|
selector_id: str
|
|
provider: str
|
|
endpoint: str | None
|
|
endpoint_pool: str | None
|
|
model: str
|
|
api_mode: str
|
|
tool_call_transport: str
|
|
identity_selector_id: str = ""
|
|
alias: str = ""
|
|
alias_defaults: Mapping[str, Any] = field(default_factory=dict)
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class RouteRef:
|
|
selector_id: str
|
|
provider: str
|
|
endpoint: str
|
|
model: str
|
|
api_mode: str
|
|
tool_call_transport: str
|
|
|
|
def key(self) -> str:
|
|
return ":".join(
|
|
(
|
|
self.provider,
|
|
self.endpoint,
|
|
self.model,
|
|
self.api_mode,
|
|
self.tool_call_transport,
|
|
)
|
|
)
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class PurposeRoutes:
|
|
default_alias: str
|
|
selectable: Mapping[str, str]
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class PurposeCallLimit:
|
|
max_attempts_per_run: int
|
|
|
|
|
|
def _effective_output_limit(model: ModelConfig, params: Mapping[str, Any]) -> int:
|
|
override = params.get("output_token_limit")
|
|
return model.max_output_tokens if override is None else int(override)
|
|
|
|
|
|
def _validate_model_output_limits(model: ModelConfig) -> None:
|
|
candidates = [model.params.get("output_token_limit")]
|
|
candidates.extend(
|
|
values.get("output_token_limit")
|
|
for values in model.purpose_overrides.values()
|
|
)
|
|
for value in candidates:
|
|
if value is not None and int(value) > model.max_output_tokens:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"model output_token_limit exceeds model capability",
|
|
)
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class RouteHealthPolicy:
|
|
failure_threshold: int
|
|
cooldown_seconds: int
|
|
half_open_max_inflight: int
|
|
counted_error_codes: tuple[str, ...]
|
|
open_immediately_error_codes: tuple[str, ...]
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class WebRuntimePolicy:
|
|
title_start_timeout_seconds: int
|
|
prepare_ttl_seconds: int
|
|
turn_lease_grace_seconds: int
|
|
active_run_timeout_seconds: int
|
|
max_run_journal_events: int
|
|
max_run_journal_bytes: int
|
|
max_prepared_runs_per_subject: int
|
|
max_prepared_runs_total: int
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class CapabilityEvidence:
|
|
route: RouteRef
|
|
connectivity: str
|
|
tool_capability: str
|
|
route_semantics_hash: str
|
|
endpoint_fingerprint: str
|
|
config_identity_key_id: str
|
|
adapter_revision: str
|
|
fixture_digest: str
|
|
verified_at: str
|
|
results: Mapping[str, str] = field(default_factory=dict)
|
|
evidence_expires_at: str = ""
|
|
implementation_fingerprint: str = ""
|
|
resolved_model_revision: str | None = None
|
|
reproducible: bool = False
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class EvoModelConfig:
|
|
config_revision: int
|
|
config_identity_key_id: str
|
|
runtime_defaults: Mapping[str, Any]
|
|
purpose_defaults: Mapping[str, Mapping[str, str]]
|
|
providers: Mapping[str, ProviderConfig]
|
|
endpoint_pools: Mapping[str, EndpointPool]
|
|
route_health: RouteHealthPolicy
|
|
route_selectors: Mapping[str, RouteSelector]
|
|
main_routes: PurposeRoutes
|
|
title_selector_id: str
|
|
purpose_call_limits: Mapping[str, PurposeCallLimit]
|
|
web_runtime: WebRuntimePolicy
|
|
capability_evidence: Mapping[str, CapabilityEvidence]
|
|
tool_protocol_fallbacks: Mapping[str, tuple[str, ...]]
|
|
raw: Mapping[str, Any]
|
|
schema_version: int = 2
|
|
aliases: Mapping[str, AliasConfig] = field(default_factory=dict)
|
|
provider_health: RouteHealthPolicy | None = None
|
|
adapter_registry_revision: str = ""
|
|
purpose_selector_ids: Mapping[str, str] = field(default_factory=dict)
|
|
|
|
@classmethod
|
|
def parse(cls, payload: Any, *, require_evidence: bool = True) -> EvoModelConfig:
|
|
# ``require_evidence`` is retained for callers of the legacy schema API.
|
|
# Capability evidence is historical audit data, not a publication or
|
|
# invocation prerequisite.
|
|
raw = _mapping(payload, "model_routes")
|
|
if raw.get("schema_version") == 3:
|
|
return _parse_v3_config(raw, require_evidence=require_evidence)
|
|
_strict_keys(
|
|
raw,
|
|
allowed={
|
|
"schema_version",
|
|
"config_revision",
|
|
"config_identity_key_id",
|
|
"runtime_defaults",
|
|
"purpose_defaults",
|
|
"providers",
|
|
"endpoint_pools",
|
|
"route_health",
|
|
"route_selectors",
|
|
"purpose_routes",
|
|
"purpose_call_limits",
|
|
"web_runtime",
|
|
"capability_evidence",
|
|
"tool_protocol_fallbacks",
|
|
},
|
|
name="model_routes",
|
|
)
|
|
if raw.get("schema_version") != 2:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "schema_version must be 2"
|
|
)
|
|
revision = _integer(raw.get("config_revision"), "config_revision", minimum=1)
|
|
identity_key_id = _text(
|
|
raw.get("config_identity_key_id"), "config_identity_key_id"
|
|
)
|
|
runtime_defaults = _mapping(raw.get("runtime_defaults"), "runtime_defaults")
|
|
_strict_keys(runtime_defaults, allowed={"max_retries"}, name="runtime_defaults")
|
|
if runtime_defaults.get("max_retries") != 0:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "runtime max_retries must be 0"
|
|
)
|
|
purpose_defaults = _parse_purpose_defaults(raw.get("purpose_defaults"))
|
|
providers = _parse_providers(raw.get("providers"))
|
|
pools = _parse_endpoint_pools(raw.get("endpoint_pools"), providers)
|
|
health = _parse_health(raw.get("route_health"))
|
|
selectors = _parse_selectors(raw.get("route_selectors"), providers, pools)
|
|
main_routes, title_selector_id = _parse_purpose_routes(
|
|
raw.get("purpose_routes"), selectors
|
|
)
|
|
limits = _parse_call_limits(raw.get("purpose_call_limits"))
|
|
web_runtime = _parse_web_runtime(raw.get("web_runtime"))
|
|
fallbacks = _parse_fallbacks(raw.get("tool_protocol_fallbacks"), selectors)
|
|
config = cls(
|
|
config_revision=revision,
|
|
config_identity_key_id=identity_key_id,
|
|
runtime_defaults=dict(runtime_defaults),
|
|
purpose_defaults=purpose_defaults,
|
|
providers=providers,
|
|
endpoint_pools=pools,
|
|
route_health=health,
|
|
route_selectors=selectors,
|
|
main_routes=main_routes,
|
|
title_selector_id=title_selector_id,
|
|
purpose_call_limits=limits,
|
|
web_runtime=web_runtime,
|
|
capability_evidence={},
|
|
tool_protocol_fallbacks=fallbacks,
|
|
raw=dict(raw),
|
|
)
|
|
config._validate(require_evidence=require_evidence)
|
|
return config
|
|
|
|
@property
|
|
def catalog_revision(self) -> int:
|
|
return self.config_revision
|
|
|
|
@property
|
|
def title_start_timeout_seconds(self) -> int:
|
|
return self.web_runtime.title_start_timeout_seconds
|
|
|
|
def resolve_main_selector(self, alias: str | None) -> RouteSelector:
|
|
selected_alias = str(alias or "").strip() or self.main_routes.default_alias
|
|
selector_id = self.main_routes.selectable.get(selected_alias)
|
|
if selector_id is None:
|
|
raise EvoRuntimeError("MODEL_ACCESS_DENIED")
|
|
return self.route_selectors[selector_id]
|
|
|
|
def concrete_routes(self, selector_id: str) -> tuple[RouteRef, ...]:
|
|
selector = self.route_selectors[selector_id]
|
|
if selector.endpoint is not None:
|
|
endpoints = (selector.endpoint,)
|
|
else:
|
|
assert selector.endpoint_pool is not None
|
|
endpoints = tuple(
|
|
item.name
|
|
for item in self.endpoint_pools[selector.endpoint_pool].endpoints
|
|
)
|
|
return tuple(
|
|
RouteRef(
|
|
selector_id=selector_id,
|
|
provider=selector.provider,
|
|
endpoint=endpoint,
|
|
model=selector.model,
|
|
api_mode=selector.api_mode,
|
|
tool_call_transport=selector.tool_call_transport,
|
|
)
|
|
for endpoint in endpoints
|
|
)
|
|
|
|
def fallback_selectors(self, primary_selector_id: str) -> tuple[str, ...]:
|
|
return self.tool_protocol_fallbacks.get(primary_selector_id, ())
|
|
|
|
def route_model(self, route: RouteRef) -> ModelConfig:
|
|
return self.providers[route.provider].models[route.model]
|
|
|
|
def required_concrete_routes(self) -> Mapping[str, tuple[str, ...]]:
|
|
if self.schema_version == 3:
|
|
required: dict[str, tuple[str, ...]] = {}
|
|
selector_ids = set(self.main_routes.selectable.values()) | {
|
|
self.title_selector_id
|
|
}
|
|
for selector_id in selector_ids:
|
|
route = self.concrete_routes(selector_id)[0]
|
|
model = self.route_model(route)
|
|
probes = ["connectivity"]
|
|
probes.extend(
|
|
key
|
|
for key, enabled in model.capabilities.items()
|
|
if enabled and key != "text"
|
|
)
|
|
required[route.key()] = tuple(dict.fromkeys(probes))
|
|
return required
|
|
main_selector_ids = set(self.main_routes.selectable.values())
|
|
reachable = set(main_selector_ids)
|
|
for selector_id in tuple(main_selector_ids):
|
|
reachable.update(self.fallback_selectors(selector_id))
|
|
result: dict[str, tuple[str, ...]] = {}
|
|
for selector_id in reachable:
|
|
for route in self.concrete_routes(selector_id):
|
|
result[route.key()] = ("connectivity", "tool_protocol")
|
|
for route in self.concrete_routes(self.title_selector_id):
|
|
result.setdefault(route.key(), ("connectivity",))
|
|
return result
|
|
|
|
def _validate(self, *, require_evidence: bool = True) -> None:
|
|
if self.schema_version == 3:
|
|
self._validate_v3(require_evidence=require_evidence)
|
|
return
|
|
main_limit = self.purpose_call_limits["main_agent"].max_attempts_per_run
|
|
for selector in self.route_selectors.values():
|
|
model = self.providers[selector.provider].models[selector.model]
|
|
_validate_model_output_limits(model)
|
|
for primary, fallbacks in self.tool_protocol_fallbacks.items():
|
|
chain = (primary, *fallbacks)
|
|
if len(chain) != len(set(chain)) or len(chain) > main_limit:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid fallback chain"
|
|
)
|
|
primary_models = [
|
|
self.route_model(route) for route in self.concrete_routes(primary)
|
|
]
|
|
base = primary_models[0]
|
|
for selector_id in fallbacks:
|
|
for candidate in self.concrete_routes(selector_id):
|
|
model = self.route_model(candidate)
|
|
if model.model_id != base.model_id or model.quote != base.quote:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"fallback billing differs",
|
|
)
|
|
|
|
def _validate_v3(self, *, require_evidence: bool) -> None:
|
|
if self.endpoint_pools or self.tool_protocol_fallbacks:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"V3 direct routes cannot contain pools or fallback chains",
|
|
)
|
|
if not self.aliases or not self.main_routes.selectable:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "aliases are required"
|
|
)
|
|
total_attempts = sum(
|
|
value.max_attempts_per_run for value in self.purpose_call_limits.values()
|
|
)
|
|
if total_attempts > 16:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"purpose attempts exceed the run limit",
|
|
)
|
|
for alias, selector_id in self.main_routes.selectable.items():
|
|
selector = self.route_selectors[selector_id]
|
|
model = self.providers[selector.provider].models[selector.model]
|
|
registration = get_adapter_registry().get(
|
|
self.providers[selector.provider].adapter_id,
|
|
self.providers[selector.provider].adapter_revision,
|
|
)
|
|
if not model.enabled or not self.providers[selector.provider].enabled:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
f"enabled alias {alias} has a disabled target",
|
|
)
|
|
_validate_model_output_limits(model)
|
|
alias_config = self.aliases[alias]
|
|
applicable_purposes = {"main_agent"}
|
|
for purpose in ("tool_selector", "deepagents_summarizer"):
|
|
explicit_selector = self.purpose_selector_ids.get(purpose)
|
|
if explicit_selector in {None, selector_id}:
|
|
applicable_purposes.add(purpose)
|
|
if self.title_selector_id == selector_id:
|
|
applicable_purposes.add("title")
|
|
for purpose in applicable_purposes:
|
|
params = dict(self.purpose_defaults[purpose])
|
|
params.update(
|
|
project_user_options_for_purpose(
|
|
values=model.params,
|
|
user_options=model.user_options,
|
|
purpose=purpose,
|
|
)
|
|
)
|
|
for name, option in model.user_options.items():
|
|
if "default" in option and purpose in set(
|
|
option.get("applies_to") or ("main_agent",)
|
|
):
|
|
params.setdefault(name, option["default"])
|
|
params.update(
|
|
project_user_options_for_purpose(
|
|
values=alias_config.defaults,
|
|
user_options=model.user_options,
|
|
purpose=purpose,
|
|
)
|
|
)
|
|
params.update(model.purpose_overrides.get(purpose, {}))
|
|
registration.validate_parameters(
|
|
params, path=f"aliases.{alias}.merged.{purpose}"
|
|
)
|
|
registration.compile_runtime_parameters(
|
|
selector.api_mode,
|
|
params,
|
|
_effective_output_limit(model, params),
|
|
provider_model_id=model.model_id,
|
|
)
|
|
|
|
|
|
_V3_RUNTIME_DEFAULTS: Mapping[str, tuple[int, int, int]] = {
|
|
"sdk_max_retries": (0, 0, 0),
|
|
"connect_timeout_seconds": (10, 1, 60),
|
|
"first_event_timeout_seconds": (60, 1, 300),
|
|
"stream_idle_timeout_seconds": (60, 1, 300),
|
|
"attempt_timeout_seconds": (600, 1, 3600),
|
|
"max_sse_event_bytes": (1_048_576, 4_096, 4_194_304),
|
|
"max_content_block_bytes": (4_194_304, 4_096, 16_777_216),
|
|
"max_output_bytes": (16_777_216, 65_536, 67_108_864),
|
|
"max_opaque_state_bytes": (8_388_608, 65_536, 33_554_432),
|
|
"stream_buffer_max_events": (256, 1, 1_024),
|
|
"stream_buffer_max_bytes": (2_097_152, 65_536, 8_388_608),
|
|
"max_tool_schema_bytes": (262_144, 4_096, 1_048_576),
|
|
"max_tool_arguments_bytes": (1_048_576, 4_096, 4_194_304),
|
|
"max_tool_schema_depth": (16, 1, 32),
|
|
"max_tool_argument_depth": (32, 1, 64),
|
|
}
|
|
_V3_CAPABILITIES = (
|
|
"text",
|
|
"vision",
|
|
"video",
|
|
"documents",
|
|
"tools",
|
|
"structured_output",
|
|
"thinking",
|
|
)
|
|
_V3_ACCESS_KEYS = {"visibility", "roles", "groups", "users"}
|
|
|
|
|
|
def _parse_v3_access(value: Any, name: str) -> Mapping[str, Any]:
|
|
raw = _mapping(value, name)
|
|
_strict_keys(raw, allowed=_V3_ACCESS_KEYS, name=name)
|
|
visibility = _text(raw.get("visibility"), f"{name}.visibility")
|
|
if visibility not in {"authenticated", "role_based", "private"}:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name}.visibility is invalid"
|
|
)
|
|
result = {
|
|
"visibility": visibility,
|
|
"roles": _string_set(raw.get("roles") or [], f"{name}.roles"),
|
|
"groups": _string_set(raw.get("groups") or [], f"{name}.groups"),
|
|
"users": _string_set(raw.get("users") or [], f"{name}.users"),
|
|
}
|
|
if visibility == "authenticated" and any(
|
|
result[key] for key in ("roles", "groups", "users")
|
|
):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
f"{name} authenticated access must be unscoped",
|
|
)
|
|
if visibility == "role_based" and not (result["roles"] or result["groups"]):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} role access is empty"
|
|
)
|
|
if visibility == "private" and not result["users"]:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name} private access is empty"
|
|
)
|
|
return result
|
|
|
|
|
|
def _normalize_v3_user_option(
|
|
value: Mapping[str, Any], rule: Any, path: str
|
|
) -> Mapping[str, Any]:
|
|
option = dict(value)
|
|
numeric = rule.kind in {"integer", "number"}
|
|
bound_names = (
|
|
"minimum",
|
|
"minimum_exclusive",
|
|
"maximum",
|
|
"maximum_exclusive",
|
|
)
|
|
if any(name in option for name in bound_names) and not numeric:
|
|
raise EvoRuntimeError(
|
|
"MODEL_PARAMETER_INVALID", f"{path} bounds require a numeric parameter"
|
|
)
|
|
if "minimum" in option and "minimum_exclusive" in option:
|
|
raise EvoRuntimeError("MODEL_PARAMETER_INVALID", f"{path} has two lower bounds")
|
|
if "maximum" in option and "maximum_exclusive" in option:
|
|
raise EvoRuntimeError("MODEL_PARAMETER_INVALID", f"{path} has two upper bounds")
|
|
for name in bound_names:
|
|
if name not in option:
|
|
continue
|
|
bound = option[name]
|
|
if (
|
|
isinstance(bound, bool)
|
|
or not isinstance(bound, int | float)
|
|
or not math.isfinite(float(bound))
|
|
):
|
|
raise EvoRuntimeError(
|
|
"MODEL_PARAMETER_INVALID", f"{path}.{name} must be finite"
|
|
)
|
|
if numeric and rule.minimum is not None:
|
|
default_name = "minimum_exclusive" if rule.minimum_exclusive else "minimum"
|
|
option.setdefault(default_name, rule.minimum)
|
|
if numeric and rule.maximum is not None:
|
|
default_name = "maximum_exclusive" if rule.maximum_exclusive else "maximum"
|
|
option.setdefault(default_name, rule.maximum)
|
|
|
|
lower_name = next(
|
|
(name for name in ("minimum", "minimum_exclusive") if name in option), None
|
|
)
|
|
upper_name = next(
|
|
(name for name in ("maximum", "maximum_exclusive") if name in option), None
|
|
)
|
|
if lower_name and rule.minimum is not None:
|
|
lower = float(option[lower_name])
|
|
if lower < rule.minimum or (
|
|
lower == rule.minimum and rule.minimum_exclusive and lower_name == "minimum"
|
|
):
|
|
raise EvoRuntimeError(
|
|
"MODEL_PARAMETER_INVALID", f"{path} loosens the adapter minimum"
|
|
)
|
|
if upper_name and rule.maximum is not None:
|
|
upper = float(option[upper_name])
|
|
if upper > rule.maximum or (
|
|
upper == rule.maximum and rule.maximum_exclusive and upper_name == "maximum"
|
|
):
|
|
raise EvoRuntimeError(
|
|
"MODEL_PARAMETER_INVALID", f"{path} loosens the adapter maximum"
|
|
)
|
|
if lower_name and upper_name:
|
|
lower = float(option[lower_name])
|
|
upper = float(option[upper_name])
|
|
if lower > upper or (
|
|
lower == upper
|
|
and (lower_name == "minimum_exclusive" or upper_name == "maximum_exclusive")
|
|
):
|
|
raise EvoRuntimeError(
|
|
"MODEL_PARAMETER_INVALID", f"{path} has an empty range"
|
|
)
|
|
|
|
if "choices" in option:
|
|
choices = list(
|
|
_sequence(option["choices"], f"{path}.choices", allow_empty=False)
|
|
)
|
|
for index, choice in enumerate(choices):
|
|
rule.validate(choice, f"{path}.choices[{index}]")
|
|
if len({canonical_json_v1(choice) for choice in choices}) != len(choices):
|
|
raise EvoRuntimeError(
|
|
"MODEL_PARAMETER_INVALID", f"{path}.choices contains duplicates"
|
|
)
|
|
option["choices"] = choices
|
|
elif rule.kind == "enum":
|
|
option["choices"] = list(rule.choices)
|
|
|
|
if "default" in option:
|
|
default = option["default"]
|
|
rule.validate(default, f"{path}.default")
|
|
if lower_name and (
|
|
default < option[lower_name]
|
|
or (lower_name == "minimum_exclusive" and default == option[lower_name])
|
|
):
|
|
raise EvoRuntimeError(
|
|
"MODEL_PARAMETER_INVALID", f"{path}.default is below its range"
|
|
)
|
|
if upper_name and (
|
|
default > option[upper_name]
|
|
or (upper_name == "maximum_exclusive" and default == option[upper_name])
|
|
):
|
|
raise EvoRuntimeError(
|
|
"MODEL_PARAMETER_INVALID", f"{path}.default exceeds its range"
|
|
)
|
|
if option.get("choices") and default not in option["choices"]:
|
|
raise EvoRuntimeError(
|
|
"MODEL_PARAMETER_INVALID", f"{path}.default is not an allowed choice"
|
|
)
|
|
return option
|
|
|
|
|
|
def _parse_v3_billing(value: Any, name: str, *, require_evidence: bool) -> PricingQuote:
|
|
raw = _mapping(value, name)
|
|
_strict_keys(
|
|
raw,
|
|
allowed={
|
|
"sku",
|
|
"pricing_revision",
|
|
"currency",
|
|
"unit_scale",
|
|
"input_microunits_per_million",
|
|
"output_microunits_per_million",
|
|
"cached_microunits_per_million",
|
|
"multiplier",
|
|
},
|
|
name=name,
|
|
)
|
|
payload = {
|
|
"billing_sku": _text(raw.get("sku"), f"{name}.sku"),
|
|
"pricing_revision": _text(
|
|
raw.get("pricing_revision"), f"{name}.pricing_revision"
|
|
),
|
|
"currency": _text(raw.get("currency"), f"{name}.currency"),
|
|
"unit_scale": _integer(raw.get("unit_scale"), f"{name}.unit_scale", minimum=1),
|
|
"input_microunits_per_million": _integer(
|
|
raw.get("input_microunits_per_million"), f"{name}.input", minimum=0
|
|
),
|
|
"output_microunits_per_million": _integer(
|
|
raw.get("output_microunits_per_million"), f"{name}.output", minimum=0
|
|
),
|
|
"cached_input_microunits_per_million": _integer(
|
|
raw.get("cached_microunits_per_million"), f"{name}.cached", minimum=0
|
|
),
|
|
"multiplier": _positive_decimal(raw.get("multiplier"), f"{name}.multiplier"),
|
|
}
|
|
if payload["currency"] != "CNY" or payload["unit_scale"] != 1_000_000:
|
|
raise EvoRuntimeError("PRICING_DIMENSION_UNSUPPORTED")
|
|
if require_evidence and payload["pricing_revision"].startswith("draft-"):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "active pricing is not approved"
|
|
)
|
|
return PricingQuote(**payload, quote_id=sha256_id(payload))
|
|
|
|
|
|
def _parse_v3_runtime_defaults(value: Any) -> Mapping[str, int]:
|
|
raw = _mapping(value or {}, "runtime_defaults")
|
|
_strict_keys(raw, allowed=set(_V3_RUNTIME_DEFAULTS), name="runtime_defaults")
|
|
result = {
|
|
key: _integer(
|
|
raw.get(key, default),
|
|
f"runtime_defaults.{key}",
|
|
minimum=minimum,
|
|
maximum=maximum,
|
|
)
|
|
for key, (default, minimum, maximum) in _V3_RUNTIME_DEFAULTS.items()
|
|
}
|
|
if result["sdk_max_retries"] != 0:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "sdk_max_retries must be 0"
|
|
)
|
|
if result["attempt_timeout_seconds"] < max(
|
|
result["connect_timeout_seconds"],
|
|
result["first_event_timeout_seconds"],
|
|
result["stream_idle_timeout_seconds"],
|
|
):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "attempt timeout is too small"
|
|
)
|
|
return result
|
|
|
|
|
|
def _parse_v3_provider_defaults(
|
|
value: Any, runtime: Mapping[str, int]
|
|
) -> Mapping[str, int]:
|
|
raw = _mapping(value or {}, "provider.defaults")
|
|
allowed = {
|
|
"connect_timeout_seconds",
|
|
"first_event_timeout_seconds",
|
|
"stream_idle_timeout_seconds",
|
|
"attempt_timeout_seconds",
|
|
"max_inflight_requests",
|
|
"queue_timeout_seconds",
|
|
}
|
|
_strict_keys(raw, allowed=allowed, name="provider.defaults")
|
|
result: dict[str, int] = {}
|
|
for key in allowed - {"max_inflight_requests", "queue_timeout_seconds"}:
|
|
result[key] = _integer(
|
|
raw.get(key, runtime[key]),
|
|
f"provider.defaults.{key}",
|
|
minimum=1,
|
|
maximum=_V3_RUNTIME_DEFAULTS[key][2],
|
|
)
|
|
result["max_inflight_requests"] = _integer(
|
|
raw.get("max_inflight_requests", 16),
|
|
"provider.defaults.max_inflight_requests",
|
|
minimum=1,
|
|
maximum=256,
|
|
)
|
|
result["queue_timeout_seconds"] = _integer(
|
|
raw.get("queue_timeout_seconds", 5),
|
|
"provider.defaults.queue_timeout_seconds",
|
|
minimum=1,
|
|
maximum=60,
|
|
)
|
|
if result["attempt_timeout_seconds"] < max(
|
|
result["connect_timeout_seconds"],
|
|
result["first_event_timeout_seconds"],
|
|
result["stream_idle_timeout_seconds"],
|
|
):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "provider attempt timeout is too small"
|
|
)
|
|
return result
|
|
|
|
|
|
def _parse_v3_health(value: Any, name: str) -> RouteHealthPolicy:
|
|
raw = _mapping(value or {}, name)
|
|
_strict_keys(
|
|
raw,
|
|
allowed={
|
|
"failure_threshold",
|
|
"cooldown_seconds",
|
|
"half_open_max_inflight",
|
|
"counted_error_codes",
|
|
"open_immediately_error_codes",
|
|
},
|
|
name=name,
|
|
)
|
|
counted = _string_set(
|
|
raw.get("counted_error_codes") or [], f"{name}.counted_error_codes"
|
|
)
|
|
immediate = _string_set(
|
|
raw.get("open_immediately_error_codes") or [],
|
|
f"{name}.open_immediately_error_codes",
|
|
)
|
|
return RouteHealthPolicy(
|
|
_integer(
|
|
raw.get("failure_threshold", 3),
|
|
f"{name}.failure_threshold",
|
|
minimum=1,
|
|
maximum=20,
|
|
),
|
|
_integer(
|
|
raw.get("cooldown_seconds", 30),
|
|
f"{name}.cooldown_seconds",
|
|
minimum=1,
|
|
maximum=3600,
|
|
),
|
|
_integer(
|
|
raw.get("half_open_max_inflight", 1),
|
|
f"{name}.half_open_max_inflight",
|
|
minimum=1,
|
|
maximum=16,
|
|
),
|
|
counted,
|
|
immediate,
|
|
)
|
|
|
|
|
|
def _parse_v3_web_runtime(value: Any) -> WebRuntimePolicy:
|
|
raw = _mapping(value or {}, "web_runtime")
|
|
specs = {
|
|
"title_start_timeout_seconds": (30, 1, 300),
|
|
"prepare_ttl_seconds": (30, 5, 300),
|
|
"turn_lease_grace_seconds": (30, 1, 300),
|
|
"active_run_timeout_seconds": (1800, 60, 7200),
|
|
"max_run_journal_events": (10000, 100, 100000),
|
|
"max_run_journal_bytes": (16777216, 1048576, 268435456),
|
|
"max_prepared_runs_per_subject": (4, 1, 32),
|
|
"max_prepared_runs_total": (128, 1, 4096),
|
|
}
|
|
_strict_keys(raw, allowed=set(specs), name="web_runtime")
|
|
parsed = {
|
|
key: _integer(
|
|
raw.get(key, default), f"web_runtime.{key}", minimum=low, maximum=high
|
|
)
|
|
for key, (default, low, high) in specs.items()
|
|
}
|
|
if parsed["max_prepared_runs_total"] < parsed["max_prepared_runs_per_subject"]:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "prepared run limits are inconsistent"
|
|
)
|
|
return WebRuntimePolicy(**parsed)
|
|
|
|
|
|
def _parse_v3_config(
|
|
raw: Mapping[str, Any], *, require_evidence: bool
|
|
) -> EvoModelConfig:
|
|
_strict_keys(
|
|
raw,
|
|
allowed={
|
|
"schema_version",
|
|
"config_revision",
|
|
"config_identity_key_id",
|
|
"runtime_defaults",
|
|
"providers",
|
|
"aliases",
|
|
"purpose_defaults",
|
|
"purpose_routes",
|
|
"purpose_call_limits",
|
|
"health_policy",
|
|
"web_runtime",
|
|
"capability_evidence",
|
|
},
|
|
name="model_routes",
|
|
)
|
|
revision = _integer(raw.get("config_revision"), "config_revision", minimum=1)
|
|
identity_key_id = _text(raw.get("config_identity_key_id"), "config_identity_key_id")
|
|
runtime_defaults = _parse_v3_runtime_defaults(raw.get("runtime_defaults"))
|
|
registry = get_adapter_registry()
|
|
providers_raw = _sequence(raw.get("providers"), "providers", allow_empty=False)
|
|
providers: dict[str, ProviderConfig] = {}
|
|
for provider_index, provider_value in enumerate(providers_raw):
|
|
name = f"providers[{provider_index}]"
|
|
item = _mapping(provider_value, name)
|
|
_strict_keys(
|
|
item,
|
|
allowed={
|
|
"provider_id",
|
|
"display_name",
|
|
"adapter_id",
|
|
"adapter_revision",
|
|
"wire_protocol",
|
|
"enabled",
|
|
"connection",
|
|
"defaults",
|
|
"models",
|
|
},
|
|
name=name,
|
|
)
|
|
provider_id = _text(item.get("provider_id"), f"{name}.provider_id")
|
|
if provider_id in providers:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate provider_id"
|
|
)
|
|
adapter_id = _text(item.get("adapter_id"), f"{name}.adapter_id")
|
|
exact_revision = _text(item.get("adapter_revision"), f"{name}.adapter_revision")
|
|
if exact_revision in {"latest", "current"}:
|
|
raise EvoRuntimeError("MODEL_ADAPTER_UNAVAILABLE")
|
|
registration = registry.get(adapter_id, exact_revision)
|
|
if registration.lifecycle == "blocked":
|
|
raise EvoRuntimeError("MODEL_ADAPTER_BLOCKED")
|
|
wire_protocol = _text(item.get("wire_protocol"), f"{name}.wire_protocol")
|
|
if wire_protocol not in registration.supported_wire_protocols:
|
|
raise EvoRuntimeError("MODEL_WIRE_PROTOCOL_UNSUPPORTED")
|
|
connection = _mapping(item.get("connection"), f"{name}.connection")
|
|
_strict_keys(
|
|
connection,
|
|
allowed={"base_url", "credential_ref"},
|
|
name=f"{name}.connection",
|
|
)
|
|
base_url = _normalize_base_url(
|
|
connection.get("base_url"), f"{name}.connection.base_url"
|
|
)
|
|
legacy_ref = connection.get("credential_ref")
|
|
if legacy_ref is None:
|
|
credential_ref = f"provider://{provider_id}"
|
|
secret_version = 0
|
|
else:
|
|
credential_ref = _text(legacy_ref, f"{name}.connection.credential_ref")
|
|
expected_prefix = f"secret://model-providers/{provider_id}#"
|
|
if not credential_ref.startswith(expected_prefix):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"credential_ref must be scoped to its provider",
|
|
)
|
|
try:
|
|
secret_version = int(credential_ref.rsplit("#", 1)[1])
|
|
except ValueError as exc:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"credential_ref version is invalid",
|
|
) from exc
|
|
if secret_version < 1:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"credential_ref version is invalid",
|
|
)
|
|
provider_defaults = _parse_v3_provider_defaults(
|
|
item.get("defaults"), runtime_defaults
|
|
)
|
|
models: dict[str, ModelConfig] = {}
|
|
for model_index, model_value in enumerate(
|
|
_sequence(item.get("models"), f"{name}.models", allow_empty=False)
|
|
):
|
|
model_name = f"{name}.models[{model_index}]"
|
|
model_raw = _mapping(model_value, model_name)
|
|
_strict_keys(
|
|
model_raw,
|
|
allowed={
|
|
"model_key",
|
|
"provider_model_id",
|
|
"version_policy",
|
|
"resolved_model_revision",
|
|
"display_name",
|
|
"description",
|
|
"enabled",
|
|
"tags",
|
|
"invocation",
|
|
"capabilities",
|
|
"limits",
|
|
"parameters",
|
|
"access",
|
|
"billing",
|
|
},
|
|
name=model_name,
|
|
)
|
|
model_key = _text(model_raw.get("model_key"), f"{model_name}.model_key")
|
|
if model_key in models:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate model_key"
|
|
)
|
|
provider_model_id = _text(
|
|
model_raw.get("provider_model_id"), f"{model_name}.provider_model_id"
|
|
)
|
|
policy = _text(
|
|
model_raw.get("version_policy"), f"{model_name}.version_policy"
|
|
)
|
|
if policy not in {"pinned", "rolling"}:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid version_policy"
|
|
)
|
|
resolved = model_raw.get("resolved_model_revision")
|
|
if resolved is not None:
|
|
resolved = _text(resolved, f"{model_name}.resolved_model_revision")
|
|
if policy == "pinned" and resolved is None:
|
|
resolved = provider_model_id
|
|
capabilities_raw = _mapping(
|
|
model_raw.get("capabilities"), f"{model_name}.capabilities"
|
|
)
|
|
_strict_keys(
|
|
capabilities_raw,
|
|
allowed=set(_V3_CAPABILITIES),
|
|
name=f"{model_name}.capabilities",
|
|
)
|
|
capabilities = {
|
|
key: _bool(
|
|
capabilities_raw.get(key, key == "text"),
|
|
f"{model_name}.capabilities.{key}",
|
|
)
|
|
for key in _V3_CAPABILITIES
|
|
}
|
|
if not capabilities["text"]:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "text capability is required"
|
|
)
|
|
invocation = _mapping(
|
|
model_raw.get("invocation"), f"{model_name}.invocation"
|
|
)
|
|
_strict_keys(
|
|
invocation,
|
|
allowed={"api_mode", "tool_call_transport"},
|
|
name=f"{model_name}.invocation",
|
|
)
|
|
api_mode = _text(
|
|
invocation.get("api_mode"), f"{model_name}.invocation.api_mode"
|
|
)
|
|
transport = _text(
|
|
invocation.get("tool_call_transport"),
|
|
f"{model_name}.invocation.tool_call_transport",
|
|
)
|
|
if transport not in {"native", "prompt", "disabled"} or (
|
|
capabilities["tools"] and transport != "native"
|
|
):
|
|
raise EvoRuntimeError("MODEL_TOOL_TRANSPORT_UNSUPPORTED")
|
|
if not capabilities["tools"]:
|
|
# V4 save validation rejects this mismatch for new revisions.
|
|
# Keep old signed projections runnable while treating their
|
|
# stale native/prompt declaration as the only safe transport.
|
|
transport = "disabled"
|
|
limits = _mapping(model_raw.get("limits"), f"{model_name}.limits")
|
|
_strict_keys(
|
|
limits,
|
|
allowed={
|
|
"context_tokens",
|
|
"max_output_tokens",
|
|
"max_inflight_requests",
|
|
},
|
|
name=f"{model_name}.limits",
|
|
)
|
|
context_tokens = limits.get("context_tokens")
|
|
max_output_tokens = limits.get("max_output_tokens")
|
|
if context_tokens is not None:
|
|
context_tokens = _integer(
|
|
context_tokens, f"{model_name}.limits.context_tokens", minimum=1
|
|
)
|
|
if max_output_tokens is not None:
|
|
max_output_tokens = _integer(
|
|
max_output_tokens,
|
|
f"{model_name}.limits.max_output_tokens",
|
|
minimum=1,
|
|
)
|
|
descriptor = registration.resolve_model_descriptor(
|
|
provider_model_id,
|
|
api_mode,
|
|
context_tokens=context_tokens,
|
|
max_output_tokens=max_output_tokens,
|
|
declared_capabilities=capabilities,
|
|
)
|
|
max_inflight = limits.get("max_inflight_requests")
|
|
if max_inflight is not None:
|
|
max_inflight = _integer(
|
|
max_inflight,
|
|
f"{model_name}.limits.max_inflight_requests",
|
|
minimum=1,
|
|
maximum=provider_defaults["max_inflight_requests"],
|
|
)
|
|
parameters = _mapping(
|
|
model_raw.get("parameters") or {}, f"{model_name}.parameters"
|
|
)
|
|
_strict_keys(
|
|
parameters,
|
|
allowed={
|
|
"defaults",
|
|
"purpose_overrides",
|
|
"user_options",
|
|
"constraints",
|
|
"reasoning_policy",
|
|
},
|
|
name=f"{model_name}.parameters",
|
|
)
|
|
defaults = dict(
|
|
_mapping(
|
|
parameters.get("defaults") or {},
|
|
f"{model_name}.parameters.defaults",
|
|
)
|
|
)
|
|
registration.validate_parameters(
|
|
defaults, path=f"{model_name}.parameters.defaults"
|
|
)
|
|
overrides_raw = _mapping(
|
|
parameters.get("purpose_overrides") or {},
|
|
f"{model_name}.parameters.purpose_overrides",
|
|
)
|
|
if not set(overrides_raw) <= set(_PURPOSES):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unknown purpose override"
|
|
)
|
|
overrides = {}
|
|
for purpose, values in overrides_raw.items():
|
|
parsed_values = dict(
|
|
_mapping(
|
|
values, f"{model_name}.parameters.purpose_overrides.{purpose}"
|
|
)
|
|
)
|
|
registration.validate_parameters(
|
|
parsed_values,
|
|
path=f"{model_name}.parameters.purpose_overrides.{purpose}",
|
|
)
|
|
overrides[purpose] = parsed_values
|
|
user_options_raw = _mapping(
|
|
parameters.get("user_options") or {},
|
|
f"{model_name}.parameters.user_options",
|
|
)
|
|
unknown_options = set(user_options_raw) - set(
|
|
registration.all_parameter_schema
|
|
)
|
|
if unknown_options:
|
|
raise EvoRuntimeError(
|
|
"MODEL_PARAMETER_INVALID",
|
|
"user_options contains an unsupported parameter",
|
|
)
|
|
user_options = {}
|
|
for key, value in user_options_raw.items():
|
|
option_path = f"{model_name}.parameters.user_options.{key}"
|
|
option = dict(_mapping(value, option_path))
|
|
_strict_keys(
|
|
option,
|
|
allowed={
|
|
"default",
|
|
"applies_to",
|
|
"minimum",
|
|
"maximum",
|
|
"minimum_exclusive",
|
|
"maximum_exclusive",
|
|
"choices",
|
|
},
|
|
name=option_path,
|
|
)
|
|
rule = registration.all_parameter_schema[key]
|
|
option = dict(_normalize_v3_user_option(option, rule, option_path))
|
|
applies_to = _string_set(
|
|
option.get("applies_to") or ["main_agent"],
|
|
f"{option_path}.applies_to",
|
|
allowed=frozenset(_PURPOSES),
|
|
)
|
|
user_options[key] = {
|
|
**option,
|
|
"type": rule.kind,
|
|
"applies_to": applies_to,
|
|
}
|
|
constraints = tuple(
|
|
dict(_mapping(value, f"{model_name}.parameters.constraints"))
|
|
for value in _sequence(
|
|
parameters.get("constraints") or [],
|
|
f"{model_name}.parameters.constraints",
|
|
)
|
|
)
|
|
reasoning_policy = _mapping(
|
|
parameters.get("reasoning_policy") or {},
|
|
f"{model_name}.parameters.reasoning_policy",
|
|
)
|
|
_strict_keys(
|
|
reasoning_policy,
|
|
allowed={"mode", "allowed_efforts", "default_effort"},
|
|
name=f"{model_name}.parameters.reasoning_policy",
|
|
)
|
|
access = _parse_v3_access(model_raw.get("access"), f"{model_name}.access")
|
|
quote = _parse_v3_billing(
|
|
model_raw.get("billing"),
|
|
f"{model_name}.billing",
|
|
require_evidence=require_evidence,
|
|
)
|
|
# A declared model capability is usable only when the selected
|
|
# Adapter has a concrete wire-level reasoning implementation.
|
|
supports_reasoning = (
|
|
capabilities["thinking"] and registration.reasoning_mode != "none"
|
|
)
|
|
if reasoning_policy and not supports_reasoning:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
f"{model_name}.parameters.reasoning_policy requires reasoning capability",
|
|
)
|
|
reasoning_mode = registration.reasoning_mode
|
|
allowed_reasoning_efforts: tuple[str, ...] = ()
|
|
default_reasoning_effort: str | None = None
|
|
if supports_reasoning:
|
|
reasoning_mode = str(
|
|
reasoning_policy.get("mode") or registration.reasoning_mode
|
|
)
|
|
if reasoning_mode not in {"boolean", "effort"}:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
f"{model_name}.parameters.reasoning_policy.mode is invalid",
|
|
)
|
|
allowed_reasoning_efforts = _string_set(
|
|
reasoning_policy.get("allowed_efforts")
|
|
or ["low", "medium", "high"],
|
|
f"{model_name}.parameters.reasoning_policy.allowed_efforts",
|
|
allowed=_REASONING_EFFORTS - {"disabled"},
|
|
)
|
|
default_reasoning_effort = str(
|
|
reasoning_policy.get("default_effort") or "medium"
|
|
)
|
|
if default_reasoning_effort not in allowed_reasoning_efforts:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
f"{model_name}.parameters.reasoning_policy.default_effort is not allowed",
|
|
)
|
|
registration.validate_parameters(
|
|
{"reasoning": default_reasoning_effort},
|
|
path=f"{model_name}.parameters.reasoning_policy",
|
|
)
|
|
constraints = validate_parameter_constraints(
|
|
constraints,
|
|
allowed_names=set(user_options)
|
|
| ({"reasoning"} if supports_reasoning else set()),
|
|
)
|
|
models[model_key] = ModelConfig(
|
|
provider_model_id,
|
|
defaults,
|
|
capabilities["vision"],
|
|
supports_reasoning,
|
|
allowed_reasoning_efforts,
|
|
descriptor.context_tokens,
|
|
descriptor.max_output_tokens,
|
|
reasoning_mode,
|
|
{"reasoning": default_reasoning_effort}
|
|
if default_reasoning_effort
|
|
else {},
|
|
{"reasoning": "off"},
|
|
(),
|
|
tuple(access["roles"]),
|
|
quote,
|
|
model_key=model_key,
|
|
display_name=_text(
|
|
model_raw.get("display_name"), f"{model_name}.display_name"
|
|
),
|
|
description=str(model_raw.get("description") or ""),
|
|
tags=_string_set(model_raw.get("tags") or [], f"{model_name}.tags"),
|
|
enabled=_bool(model_raw.get("enabled"), f"{model_name}.enabled"),
|
|
version_policy=policy,
|
|
resolved_model_revision=resolved,
|
|
reproducible=policy == "pinned",
|
|
capabilities=capabilities,
|
|
purpose_overrides=overrides,
|
|
user_options=user_options,
|
|
parameter_constraints=constraints,
|
|
access=access,
|
|
max_inflight_requests=max_inflight,
|
|
descriptor_parameters=descriptor.parameters,
|
|
)
|
|
endpoint = EndpointConfig(
|
|
provider_id,
|
|
base_url,
|
|
SecretReference(credential_ref, secret_version),
|
|
{},
|
|
{},
|
|
{},
|
|
)
|
|
providers[provider_id] = ProviderConfig(
|
|
provider_id,
|
|
adapter_id,
|
|
{},
|
|
{provider_id: endpoint},
|
|
models,
|
|
display_name=_text(item.get("display_name"), f"{name}.display_name"),
|
|
adapter_id=adapter_id,
|
|
adapter_revision=exact_revision,
|
|
wire_protocol=wire_protocol,
|
|
enabled=_bool(item.get("enabled"), f"{name}.enabled"),
|
|
connection_defaults=provider_defaults,
|
|
implementation_fingerprint=registration.implementation_fingerprint,
|
|
)
|
|
aliases_raw = _sequence(raw.get("aliases"), "aliases", allow_empty=False)
|
|
aliases: dict[str, AliasConfig] = {}
|
|
selectors: dict[str, RouteSelector] = {}
|
|
selectable: dict[str, str] = {}
|
|
for index, alias_value in enumerate(aliases_raw):
|
|
name = f"aliases[{index}]"
|
|
item = _mapping(alias_value, name)
|
|
_strict_keys(
|
|
item,
|
|
allowed={
|
|
"alias",
|
|
"display_name",
|
|
"provider_ref",
|
|
"model_ref",
|
|
"enabled",
|
|
"access",
|
|
"defaults",
|
|
},
|
|
name=name,
|
|
)
|
|
alias = _text(item.get("alias"), f"{name}.alias")
|
|
if alias in aliases:
|
|
raise EvoRuntimeError("LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate alias")
|
|
provider_ref = _text(item.get("provider_ref"), f"{name}.provider_ref")
|
|
model_ref = _text(item.get("model_ref"), f"{name}.model_ref")
|
|
if (
|
|
provider_ref not in providers
|
|
or model_ref not in providers[provider_ref].models
|
|
):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "alias target does not exist"
|
|
)
|
|
access = _parse_v3_access(item.get("access"), f"{name}.access")
|
|
defaults = dict(_mapping(item.get("defaults") or {}, f"{name}.defaults"))
|
|
registration = registry.get(
|
|
providers[provider_ref].adapter_id, providers[provider_ref].adapter_revision
|
|
)
|
|
registration.validate_parameters(defaults, path=f"{name}.defaults")
|
|
model = providers[provider_ref].models[model_ref]
|
|
if not set(defaults) <= set(model.user_options):
|
|
raise EvoRuntimeError(
|
|
"MODEL_PARAMETER_INVALID", "alias defaults must use model user_options"
|
|
)
|
|
revision_key = hashlib.sha256(
|
|
str(model.resolved_model_revision or model.model_id).encode()
|
|
).hexdigest()[:12]
|
|
identity_selector = f"direct:{provider_ref}:{model_ref}:{revision_key}:{_find_model_api_mode(raw, provider_ref, model_ref)}:{_find_model_transport(raw, provider_ref, model_ref)}"
|
|
internal_selector = f"{identity_selector}:alias:{alias}"
|
|
selector = RouteSelector(
|
|
internal_selector,
|
|
provider_ref,
|
|
provider_ref,
|
|
None,
|
|
model_ref,
|
|
_find_model_api_mode(raw, provider_ref, model_ref),
|
|
_find_model_transport(raw, provider_ref, model_ref),
|
|
identity_selector_id=identity_selector,
|
|
alias=alias,
|
|
alias_defaults=defaults,
|
|
)
|
|
selectors[internal_selector] = selector
|
|
enabled = _bool(item.get("enabled"), f"{name}.enabled")
|
|
aliases[alias] = AliasConfig(
|
|
alias,
|
|
_text(item.get("display_name"), f"{name}.display_name"),
|
|
provider_ref,
|
|
model_ref,
|
|
enabled,
|
|
access,
|
|
defaults,
|
|
)
|
|
if enabled:
|
|
selectable[alias] = internal_selector
|
|
purpose_defaults_raw = _mapping(
|
|
raw.get("purpose_defaults") or {}, "purpose_defaults"
|
|
)
|
|
if set(purpose_defaults_raw) != set(_PURPOSES):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"purpose_defaults must define all purposes",
|
|
)
|
|
purpose_defaults = {
|
|
key: dict(_mapping(value, f"purpose_defaults.{key}"))
|
|
for key, value in purpose_defaults_raw.items()
|
|
}
|
|
purpose_routes = _mapping(raw.get("purpose_routes"), "purpose_routes")
|
|
if set(purpose_routes) != set(_PURPOSES):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"purpose_routes must define all purposes",
|
|
)
|
|
main_route = _mapping(purpose_routes["main_agent"], "purpose_routes.main_agent")
|
|
_strict_keys(
|
|
main_route, allowed={"default_alias"}, name="purpose_routes.main_agent"
|
|
)
|
|
default_alias = _text(
|
|
main_route.get("default_alias"), "purpose_routes.main_agent.default_alias"
|
|
)
|
|
if default_alias not in selectable:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "default alias is unavailable"
|
|
)
|
|
purpose_selector_ids: dict[str, str] = {}
|
|
for purpose in ("tool_selector", "deepagents_summarizer"):
|
|
route = purpose_routes[purpose]
|
|
if route != "inherit_main" and not (
|
|
isinstance(route, Mapping)
|
|
and set(route) == {"default_alias"}
|
|
and route["default_alias"] in selectable
|
|
):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{purpose} route is invalid"
|
|
)
|
|
if isinstance(route, Mapping):
|
|
purpose_selector_ids[purpose] = selectable[str(route["default_alias"])]
|
|
title_route = _mapping(purpose_routes["title"], "purpose_routes.title")
|
|
_strict_keys(title_route, allowed={"default_alias"}, name="purpose_routes.title")
|
|
title_alias = _text(
|
|
title_route.get("default_alias"), "purpose_routes.title.default_alias"
|
|
)
|
|
if title_alias not in selectable:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "title alias is unavailable"
|
|
)
|
|
limits_raw = _mapping(raw.get("purpose_call_limits"), "purpose_call_limits")
|
|
if set(limits_raw) != set(_PURPOSES):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"purpose_call_limits must define all purposes",
|
|
)
|
|
purpose_limits = {}
|
|
for purpose, value in limits_raw.items():
|
|
item = _mapping(value, f"purpose_call_limits.{purpose}")
|
|
_strict_keys(
|
|
item,
|
|
allowed={"max_output_tokens", "max_attempts_per_run"},
|
|
name=f"purpose_call_limits.{purpose}",
|
|
)
|
|
purpose_limits[purpose] = PurposeCallLimit(
|
|
_integer(
|
|
item.get("max_attempts_per_run"),
|
|
f"purpose_call_limits.{purpose}.max_attempts_per_run",
|
|
minimum=1,
|
|
maximum=8,
|
|
),
|
|
)
|
|
health = _mapping(raw.get("health_policy") or {}, "health_policy")
|
|
_strict_keys(
|
|
health, allowed={"provider_connection", "model_route"}, name="health_policy"
|
|
)
|
|
provider_health = _parse_v3_health(
|
|
health.get("provider_connection"), "health_policy.provider_connection"
|
|
)
|
|
model_health = _parse_v3_health(
|
|
health.get("model_route"), "health_policy.model_route"
|
|
)
|
|
config = EvoModelConfig(
|
|
revision,
|
|
identity_key_id,
|
|
runtime_defaults,
|
|
purpose_defaults,
|
|
providers,
|
|
{},
|
|
model_health,
|
|
selectors,
|
|
PurposeRoutes(default_alias, selectable),
|
|
selectable[title_alias],
|
|
purpose_limits,
|
|
_parse_v3_web_runtime(raw.get("web_runtime")),
|
|
{},
|
|
{},
|
|
dict(raw),
|
|
schema_version=3,
|
|
aliases=aliases,
|
|
provider_health=provider_health,
|
|
adapter_registry_revision=registry.registry_revision,
|
|
purpose_selector_ids=purpose_selector_ids,
|
|
)
|
|
config._validate(require_evidence=require_evidence)
|
|
return config
|
|
|
|
|
|
def _find_v3_model(
|
|
raw: Mapping[str, Any], provider_id: str, model_key: str
|
|
) -> Mapping[str, Any]:
|
|
for provider in raw.get("providers") or []:
|
|
if provider.get("provider_id") == provider_id:
|
|
for model in provider.get("models") or []:
|
|
if model.get("model_key") == model_key:
|
|
return model
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "model target does not exist"
|
|
)
|
|
|
|
|
|
def _find_model_api_mode(
|
|
raw: Mapping[str, Any], provider_id: str, model_key: str
|
|
) -> str:
|
|
return _text(
|
|
_mapping(
|
|
_find_v3_model(raw, provider_id, model_key).get("invocation"), "invocation"
|
|
).get("api_mode"),
|
|
"api_mode",
|
|
)
|
|
|
|
|
|
def _find_model_transport(
|
|
raw: Mapping[str, Any], provider_id: str, model_key: str
|
|
) -> str:
|
|
return _text(
|
|
_mapping(
|
|
_find_v3_model(raw, provider_id, model_key).get("invocation"), "invocation"
|
|
).get("tool_call_transport"),
|
|
"tool_call_transport",
|
|
)
|
|
|
|
|
|
def _parse_v3_evidence(
|
|
value: Any, config: EvoModelConfig
|
|
) -> Mapping[str, CapabilityEvidence]:
|
|
result: dict[str, CapabilityEvidence] = {}
|
|
for index, evidence_value in enumerate(_sequence(value, "capability_evidence")):
|
|
name = f"capability_evidence[{index}]"
|
|
raw = _mapping(evidence_value, name)
|
|
allowed = {
|
|
"provider_ref",
|
|
"model_ref",
|
|
"adapter_id",
|
|
"adapter_revision",
|
|
"implementation_fingerprint",
|
|
"wire_protocol",
|
|
"provider_model_id",
|
|
"resolved_model_revision",
|
|
"version_policy",
|
|
"reproducible",
|
|
"api_mode",
|
|
"tool_call_transport",
|
|
"base_url_fingerprint",
|
|
"secret_version",
|
|
"route_semantics_hash",
|
|
"fixture_digest",
|
|
"verified_at",
|
|
"evidence_expires_at",
|
|
"results",
|
|
}
|
|
_strict_keys(raw, allowed=allowed, name=name)
|
|
provider_ref = _text(raw.get("provider_ref"), f"{name}.provider_ref")
|
|
model_ref = _text(raw.get("model_ref"), f"{name}.model_ref")
|
|
matches = [
|
|
selector
|
|
for selector in config.route_selectors.values()
|
|
if selector.provider == provider_ref and selector.model == model_ref
|
|
]
|
|
if not matches:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "evidence target does not exist"
|
|
)
|
|
route = config.concrete_routes(matches[0].selector_id)[0]
|
|
results_raw = _mapping(raw.get("results"), f"{name}.results")
|
|
valid_statuses = {"supported", "failed", "not_verified", "not_declared"}
|
|
results = {}
|
|
for key, status in results_raw.items():
|
|
status_text = _text(status, f"{name}.results.{key}")
|
|
if status_text not in valid_statuses:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid evidence status"
|
|
)
|
|
results[key] = status_text
|
|
model = config.providers[provider_ref].models[model_ref]
|
|
evidence = CapabilityEvidence(
|
|
route,
|
|
results.get("connectivity", "failed"),
|
|
results.get("tools", "not_declared"),
|
|
_text(raw.get("route_semantics_hash"), f"{name}.route_semantics_hash"),
|
|
_text(raw.get("base_url_fingerprint"), f"{name}.base_url_fingerprint"),
|
|
config.config_identity_key_id,
|
|
_text(raw.get("adapter_revision"), f"{name}.adapter_revision"),
|
|
_text(raw.get("fixture_digest"), f"{name}.fixture_digest"),
|
|
_text(raw.get("verified_at"), f"{name}.verified_at"),
|
|
results=results,
|
|
evidence_expires_at=_text(
|
|
raw.get("evidence_expires_at"), f"{name}.evidence_expires_at"
|
|
),
|
|
implementation_fingerprint=_text(
|
|
raw.get("implementation_fingerprint"),
|
|
f"{name}.implementation_fingerprint",
|
|
),
|
|
resolved_model_revision=str(raw.get("resolved_model_revision") or "")
|
|
or None,
|
|
reproducible=_bool(raw.get("reproducible"), f"{name}.reproducible"),
|
|
)
|
|
if (
|
|
raw.get("provider_model_id") != model.model_id
|
|
or raw.get("adapter_id") != config.providers[provider_ref].adapter_id
|
|
):
|
|
raise EvoRuntimeError("CAPABILITY_EVIDENCE_STALE")
|
|
result[route.key()] = evidence
|
|
return result
|
|
|
|
|
|
def _parse_secret(value: Any, name: str) -> SecretReference:
|
|
raw = _mapping(value, name)
|
|
_strict_keys(raw, allowed={"ref", "revision"}, name=name)
|
|
ref = _text(raw.get("ref"), f"{name}.ref")
|
|
revision = _integer(raw.get("revision"), f"{name}.revision", minimum=1)
|
|
if ref.startswith("env://"):
|
|
if not _ENV_RE.fullmatch(ref[6:]):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name}.ref is invalid"
|
|
)
|
|
elif ref.startswith("secret://"):
|
|
if "#" not in ref[9:] or not ref.rsplit("#", 1)[1]:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name}.ref needs a version"
|
|
)
|
|
else:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", f"{name}.ref scheme is unsupported"
|
|
)
|
|
return SecretReference(ref, revision)
|
|
|
|
|
|
def _normalize_base_url(value: Any, name: str) -> str:
|
|
return str(value or "").strip().rstrip("/")
|
|
|
|
|
|
def _parse_providers(value: Any) -> Mapping[str, ProviderConfig]:
|
|
raw = _mapping(value, "providers")
|
|
if not raw:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "providers must not be empty"
|
|
)
|
|
result: dict[str, ProviderConfig] = {}
|
|
for provider_key, provider_value in raw.items():
|
|
key = _text(provider_key, "provider key")
|
|
provider = _mapping(provider_value, f"providers.{key}")
|
|
_strict_keys(
|
|
provider,
|
|
allowed={"protocol", "params", "endpoints", "models"},
|
|
name=f"providers.{key}",
|
|
)
|
|
protocol = _text(provider.get("protocol"), f"providers.{key}.protocol")
|
|
endpoints: dict[str, EndpointConfig] = {}
|
|
for index, value_item in enumerate(
|
|
_sequence(provider.get("endpoints"), "endpoints", allow_empty=False)
|
|
):
|
|
item = _mapping(value_item, f"providers.{key}.endpoints[{index}]")
|
|
_strict_keys(
|
|
item,
|
|
allowed={
|
|
"name",
|
|
"base_url",
|
|
"auth",
|
|
"headers",
|
|
"header_refs",
|
|
"params",
|
|
},
|
|
name=f"providers.{key}.endpoints[{index}]",
|
|
)
|
|
endpoint_name = _text(item.get("name"), "endpoint.name")
|
|
if endpoint_name in endpoints:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate endpoint"
|
|
)
|
|
headers = dict(_mapping(item.get("headers") or {}, "endpoint.headers"))
|
|
header_refs_raw = _mapping(
|
|
item.get("header_refs") or {}, "endpoint.header_refs"
|
|
)
|
|
parsed_headers: dict[str, str] = {}
|
|
for header, header_value in headers.items():
|
|
if (
|
|
not _HEADER_RE.fullmatch(header)
|
|
or header.lower() in _SENSITIVE_HEADERS
|
|
):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid static header"
|
|
)
|
|
text = _text(header_value, f"headers.{header}")
|
|
if any(char in text for char in "\r\n\0"):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid header value"
|
|
)
|
|
parsed_headers[header] = text
|
|
parsed_header_refs: dict[str, SecretReference] = {}
|
|
for header, reference in header_refs_raw.items():
|
|
if not _HEADER_RE.fullmatch(header):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid secret header"
|
|
)
|
|
if header.lower() in {name.lower() for name in parsed_headers}:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate header"
|
|
)
|
|
parsed_header_refs[header] = _parse_secret(
|
|
reference, f"header_refs.{header}"
|
|
)
|
|
endpoints[endpoint_name] = EndpointConfig(
|
|
endpoint_name,
|
|
_normalize_base_url(item.get("base_url"), "endpoint.base_url"),
|
|
_parse_secret(item.get("auth"), "endpoint.auth"),
|
|
parsed_headers,
|
|
parsed_header_refs,
|
|
dict(_mapping(item.get("params") or {}, "endpoint.params")),
|
|
)
|
|
models: dict[str, ModelConfig] = {}
|
|
for index, value_item in enumerate(
|
|
_sequence(provider.get("models"), "models", allow_empty=False)
|
|
):
|
|
item = _mapping(value_item, f"providers.{key}.models[{index}]")
|
|
_strict_keys(
|
|
item,
|
|
allowed={
|
|
"id",
|
|
"params",
|
|
"supports_vision",
|
|
"supports_reasoning",
|
|
"allowed_reasoning_efforts",
|
|
"context_window",
|
|
"max_output_tokens",
|
|
"reasoning",
|
|
"access",
|
|
"billing",
|
|
},
|
|
name=f"providers.{key}.models[{index}]",
|
|
)
|
|
model_id = _text(item.get("id"), "model.id")
|
|
if model_id in models:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate model id"
|
|
)
|
|
access = _mapping(item.get("access"), "model.access")
|
|
_strict_keys(
|
|
access, allowed={"allowed_plans", "allowed_roles"}, name="model.access"
|
|
)
|
|
billing = _mapping(item.get("billing"), "model.billing")
|
|
_strict_keys(
|
|
billing,
|
|
allowed={
|
|
"sku",
|
|
"pricing_revision",
|
|
"currency",
|
|
"unit_scale",
|
|
"input_microunits_per_million",
|
|
"output_microunits_per_million",
|
|
"cached_microunits_per_million",
|
|
"multiplier",
|
|
},
|
|
name="model.billing",
|
|
)
|
|
quote_payload = {
|
|
"billing_sku": _text(billing.get("sku"), "billing.sku"),
|
|
"pricing_revision": _text(
|
|
billing.get("pricing_revision"), "billing.pricing_revision"
|
|
),
|
|
"currency": _text(billing.get("currency"), "billing.currency"),
|
|
"unit_scale": _integer(
|
|
billing.get("unit_scale"), "billing.unit_scale", minimum=1
|
|
),
|
|
"input_microunits_per_million": _integer(
|
|
billing.get("input_microunits_per_million"), "billing.input"
|
|
),
|
|
"cached_input_microunits_per_million": _integer(
|
|
billing.get("cached_microunits_per_million"), "billing.cached"
|
|
),
|
|
"output_microunits_per_million": _integer(
|
|
billing.get("output_microunits_per_million"), "billing.output"
|
|
),
|
|
"multiplier": _positive_decimal(
|
|
billing.get("multiplier"), "billing.multiplier"
|
|
),
|
|
}
|
|
if (
|
|
quote_payload["currency"] != "CNY"
|
|
or quote_payload["unit_scale"] != 1_000_000
|
|
):
|
|
raise EvoRuntimeError("PRICING_DIMENSION_UNSUPPORTED")
|
|
quote = PricingQuote(**quote_payload, quote_id=sha256_id(quote_payload))
|
|
supports_reasoning = _bool(
|
|
item.get("supports_reasoning"), "model.supports_reasoning"
|
|
)
|
|
efforts = _string_set(
|
|
item.get("allowed_reasoning_efforts"),
|
|
"model.allowed_reasoning_efforts",
|
|
allowed=_REASONING_EFFORTS - {"disabled"},
|
|
)
|
|
if bool(efforts) != supports_reasoning:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"reasoning capability is inconsistent",
|
|
)
|
|
reasoning = _parse_model_reasoning(
|
|
item.get("reasoning"), supports_reasoning=supports_reasoning
|
|
)
|
|
context_window = _integer(
|
|
item.get("context_window"), "model.context_window", minimum=1
|
|
)
|
|
max_output_tokens = _integer(
|
|
item.get("max_output_tokens", context_window),
|
|
"model.max_output_tokens",
|
|
minimum=1,
|
|
maximum=context_window,
|
|
)
|
|
models[model_id] = ModelConfig(
|
|
model_id,
|
|
dict(_mapping(item.get("params") or {}, "model.params")),
|
|
_bool(item.get("supports_vision"), "model.supports_vision"),
|
|
supports_reasoning,
|
|
efforts,
|
|
context_window,
|
|
max_output_tokens,
|
|
reasoning["mode"],
|
|
reasoning["enabled_params"],
|
|
reasoning["disabled_params"],
|
|
_string_set(access.get("allowed_plans"), "allowed_plans"),
|
|
_string_set(access.get("allowed_roles"), "allowed_roles"),
|
|
quote,
|
|
)
|
|
result[key] = ProviderConfig(
|
|
key,
|
|
protocol,
|
|
dict(_mapping(provider.get("params") or {}, "provider.params")),
|
|
endpoints,
|
|
models,
|
|
)
|
|
return result
|
|
|
|
|
|
def _parse_model_reasoning(
|
|
value: Any, *, supports_reasoning: bool
|
|
) -> Mapping[str, Any]:
|
|
"""Parse provider-specific reasoning controls without exposing them to users."""
|
|
|
|
if value is None:
|
|
return {
|
|
"mode": "effort" if supports_reasoning else "boolean",
|
|
"enabled_params": {},
|
|
"disabled_params": {},
|
|
}
|
|
raw = _mapping(value, "model.reasoning")
|
|
_strict_keys(
|
|
raw,
|
|
allowed={"mode", "enabled_params", "disabled_params"},
|
|
name="model.reasoning",
|
|
)
|
|
mode = _text(raw.get("mode"), "model.reasoning.mode")
|
|
if mode not in _REASONING_MODES:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid reasoning mode"
|
|
)
|
|
if not supports_reasoning:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"non-reasoning model cannot define reasoning controls",
|
|
)
|
|
return {
|
|
"mode": mode,
|
|
"enabled_params": dict(
|
|
_mapping(raw.get("enabled_params") or {}, "model.reasoning.enabled_params")
|
|
),
|
|
"disabled_params": dict(
|
|
_mapping(
|
|
raw.get("disabled_params") or {}, "model.reasoning.disabled_params"
|
|
)
|
|
),
|
|
}
|
|
|
|
|
|
def _parse_endpoint_pools(
|
|
value: Any, providers: Mapping[str, ProviderConfig]
|
|
) -> Mapping[str, EndpointPool]:
|
|
raw = _mapping(value, "endpoint_pools")
|
|
result: dict[str, EndpointPool] = {}
|
|
for pool_key, pool_value in raw.items():
|
|
pool_id = _text(pool_key, "pool id")
|
|
pool = _mapping(pool_value, f"endpoint_pools.{pool_id}")
|
|
_strict_keys(
|
|
pool,
|
|
allowed={"provider", "strategy", "endpoints"},
|
|
name=f"endpoint_pools.{pool_id}",
|
|
)
|
|
provider = _text(pool.get("provider"), "pool.provider")
|
|
if provider not in providers:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unknown pool provider"
|
|
)
|
|
strategy = _text(pool.get("strategy"), "pool.strategy")
|
|
if strategy != "smooth_weighted_round_robin":
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unsupported pool strategy"
|
|
)
|
|
members: list[PoolEndpoint] = []
|
|
seen: set[str] = set()
|
|
for item_value in _sequence(
|
|
pool.get("endpoints"), "pool.endpoints", allow_empty=False
|
|
):
|
|
item = _mapping(item_value, "pool endpoint")
|
|
_strict_keys(item, allowed={"name", "weight"}, name="pool endpoint")
|
|
name = _text(item.get("name"), "pool endpoint.name")
|
|
if name in seen or name not in providers[provider].endpoints:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid pool endpoint"
|
|
)
|
|
seen.add(name)
|
|
members.append(
|
|
PoolEndpoint(
|
|
name,
|
|
_integer(item.get("weight"), "pool endpoint.weight", minimum=1),
|
|
)
|
|
)
|
|
result[pool_id] = EndpointPool(pool_id, provider, strategy, tuple(members))
|
|
return result
|
|
|
|
|
|
def _parse_selectors(
|
|
value: Any,
|
|
providers: Mapping[str, ProviderConfig],
|
|
pools: Mapping[str, EndpointPool],
|
|
) -> Mapping[str, RouteSelector]:
|
|
raw = _mapping(value, "route_selectors")
|
|
result: dict[str, RouteSelector] = {}
|
|
for selector_key, selector_value in raw.items():
|
|
selector_id = _text(selector_key, "selector id")
|
|
item = _mapping(selector_value, f"route_selectors.{selector_id}")
|
|
_strict_keys(
|
|
item,
|
|
allowed={
|
|
"provider",
|
|
"endpoint",
|
|
"endpoint_pool",
|
|
"model",
|
|
"api_mode",
|
|
"tool_call_transport",
|
|
},
|
|
name=f"route_selectors.{selector_id}",
|
|
)
|
|
provider_id = _text(item.get("provider"), "selector.provider")
|
|
provider = providers.get(provider_id)
|
|
if provider is None:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unknown selector provider"
|
|
)
|
|
endpoint = item.get("endpoint")
|
|
endpoint_pool = item.get("endpoint_pool")
|
|
if (endpoint is None) == (endpoint_pool is None):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"selector requires exactly one endpoint source",
|
|
)
|
|
endpoint_name = (
|
|
_text(endpoint, "selector.endpoint") if endpoint is not None else None
|
|
)
|
|
pool_id = (
|
|
_text(endpoint_pool, "selector.endpoint_pool")
|
|
if endpoint_pool is not None
|
|
else None
|
|
)
|
|
if endpoint_name is not None and endpoint_name not in provider.endpoints:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unknown selector endpoint"
|
|
)
|
|
if pool_id is not None and (
|
|
pool_id not in pools or pools[pool_id].provider != provider_id
|
|
):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "selector pool provider mismatch"
|
|
)
|
|
model_id = _text(item.get("model"), "selector.model")
|
|
if model_id not in provider.models:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unknown selector model"
|
|
)
|
|
api_mode = _text(item.get("api_mode"), "selector.api_mode")
|
|
allowed = _PARAM_ALLOWLIST.get((provider.protocol, api_mode))
|
|
if allowed is None:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "provider/api mode has no adapter"
|
|
)
|
|
transport = _text(
|
|
item.get("tool_call_transport"), "selector.tool_call_transport"
|
|
)
|
|
if transport != "non_streaming":
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"Web tool routes must be non_streaming",
|
|
)
|
|
_validate_params(
|
|
provider.params, name=f"providers.{provider_id}.params", allowed=allowed
|
|
)
|
|
for concrete_endpoint in provider.endpoints.values():
|
|
_validate_params(
|
|
concrete_endpoint.params, name="endpoint.params", allowed=allowed
|
|
)
|
|
_validate_params(
|
|
provider.models[model_id].params, name="model.params", allowed=allowed
|
|
)
|
|
result[selector_id] = RouteSelector(
|
|
selector_id,
|
|
provider_id,
|
|
endpoint_name,
|
|
pool_id,
|
|
model_id,
|
|
api_mode,
|
|
transport,
|
|
)
|
|
if not result:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "route_selectors must not be empty"
|
|
)
|
|
return result
|
|
|
|
|
|
def _parse_purpose_defaults(value: Any) -> Mapping[str, Mapping[str, str]]:
|
|
raw = _mapping(value, "purpose_defaults")
|
|
if set(raw) != set(_PURPOSES):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED",
|
|
"purpose_defaults must define all purposes",
|
|
)
|
|
result = {}
|
|
for purpose in _PURPOSES:
|
|
item = _mapping(raw[purpose], f"purpose_defaults.{purpose}")
|
|
_strict_keys(
|
|
item, allowed={"reasoning_effort"}, name=f"purpose_defaults.{purpose}"
|
|
)
|
|
effort = _text(item.get("reasoning_effort"), "reasoning_effort")
|
|
if effort not in _REASONING_EFFORTS:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid reasoning effort"
|
|
)
|
|
result[purpose] = {"reasoning_effort": effort}
|
|
return result
|
|
|
|
|
|
def _parse_purpose_routes(
|
|
value: Any, selectors: Mapping[str, RouteSelector]
|
|
) -> tuple[PurposeRoutes, str]:
|
|
raw = _mapping(value, "purpose_routes")
|
|
_strict_keys(raw, allowed={"main_agent", "title"}, name="purpose_routes")
|
|
main = _mapping(raw.get("main_agent"), "purpose_routes.main_agent")
|
|
_strict_keys(
|
|
main, allowed={"default_alias", "selectable"}, name="purpose_routes.main_agent"
|
|
)
|
|
selectable_raw = _mapping(main.get("selectable"), "main_agent.selectable")
|
|
selectable = {
|
|
_text(alias, "model alias"): _text(selector_id, "selector id")
|
|
for alias, selector_id in selectable_raw.items()
|
|
}
|
|
if not selectable or any(
|
|
selector not in selectors for selector in selectable.values()
|
|
):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid selectable route"
|
|
)
|
|
default_alias = _text(main.get("default_alias"), "main_agent.default_alias")
|
|
if default_alias not in selectable:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "default alias is not selectable"
|
|
)
|
|
title = _mapping(raw.get("title"), "purpose_routes.title")
|
|
_strict_keys(title, allowed={"default"}, name="purpose_routes.title")
|
|
title_selector = _text(title.get("default"), "title.default")
|
|
if title_selector not in selectors:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "unknown title selector"
|
|
)
|
|
return PurposeRoutes(default_alias, selectable), title_selector
|
|
|
|
|
|
def _parse_call_limits(value: Any) -> Mapping[str, PurposeCallLimit]:
|
|
raw = _mapping(value, "purpose_call_limits")
|
|
if set(raw) != set(_PURPOSES):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "call limits must define all purposes"
|
|
)
|
|
result = {}
|
|
for purpose in _PURPOSES:
|
|
item = _mapping(raw[purpose], f"purpose_call_limits.{purpose}")
|
|
_strict_keys(
|
|
item,
|
|
allowed={"max_output_tokens", "max_attempts_per_run"},
|
|
name=f"purpose_call_limits.{purpose}",
|
|
)
|
|
result[purpose] = PurposeCallLimit(
|
|
_integer(
|
|
item.get("max_attempts_per_run"), "max_attempts_per_run", minimum=1
|
|
),
|
|
)
|
|
return result
|
|
|
|
|
|
def _parse_health(value: Any) -> RouteHealthPolicy:
|
|
raw = _mapping(value, "route_health")
|
|
_strict_keys(
|
|
raw,
|
|
allowed={
|
|
"failure_threshold",
|
|
"cooldown_seconds",
|
|
"half_open_max_inflight",
|
|
"counted_error_codes",
|
|
"open_immediately_error_codes",
|
|
},
|
|
name="route_health",
|
|
)
|
|
return RouteHealthPolicy(
|
|
_integer(raw.get("failure_threshold"), "failure_threshold", minimum=1),
|
|
_integer(raw.get("cooldown_seconds"), "cooldown_seconds", minimum=1),
|
|
_integer(
|
|
raw.get("half_open_max_inflight"), "half_open_max_inflight", minimum=1
|
|
),
|
|
_string_set(raw.get("counted_error_codes"), "counted_error_codes"),
|
|
_string_set(
|
|
raw.get("open_immediately_error_codes"), "open_immediately_error_codes"
|
|
),
|
|
)
|
|
|
|
|
|
def _parse_web_runtime(value: Any) -> WebRuntimePolicy:
|
|
raw = _mapping(value, "web_runtime")
|
|
names = {
|
|
"title_start_timeout_seconds",
|
|
"prepare_ttl_seconds",
|
|
"turn_lease_grace_seconds",
|
|
"active_run_timeout_seconds",
|
|
"max_run_journal_events",
|
|
"max_run_journal_bytes",
|
|
"max_prepared_runs_per_subject",
|
|
"max_prepared_runs_total",
|
|
}
|
|
_strict_keys(raw, allowed=names, name="web_runtime")
|
|
values = {
|
|
name: _integer(raw.get(name), f"web_runtime.{name}", minimum=1)
|
|
for name in names
|
|
}
|
|
if (
|
|
values["prepare_ttl_seconds"] > 120
|
|
or values["title_start_timeout_seconds"] > 300
|
|
):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "runtime TTL is out of range"
|
|
)
|
|
if values["max_prepared_runs_per_subject"] > values["max_prepared_runs_total"]:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "prepared run limits are inconsistent"
|
|
)
|
|
return WebRuntimePolicy(**values)
|
|
|
|
|
|
def _parse_fallbacks(
|
|
value: Any, selectors: Mapping[str, RouteSelector]
|
|
) -> Mapping[str, tuple[str, ...]]:
|
|
result: dict[str, tuple[str, ...]] = {}
|
|
for item_value in _sequence(value, "tool_protocol_fallbacks"):
|
|
item = _mapping(item_value, "tool_protocol_fallback")
|
|
_strict_keys(
|
|
item, allowed={"primary", "fallbacks"}, name="tool_protocol_fallback"
|
|
)
|
|
primary = _text(item.get("primary"), "fallback.primary")
|
|
fallbacks = tuple(
|
|
_text(entry, "fallback selector")
|
|
for entry in _sequence(item.get("fallbacks"), "fallbacks")
|
|
)
|
|
if (
|
|
primary in result
|
|
or primary not in selectors
|
|
or any(entry not in selectors for entry in fallbacks)
|
|
):
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "invalid fallback selector"
|
|
)
|
|
result[primary] = fallbacks
|
|
return result
|
|
|
|
|
|
def _route_from_evidence(value: Any, config: EvoModelConfig) -> RouteRef:
|
|
raw = _mapping(value, "evidence.route")
|
|
_strict_keys(
|
|
raw,
|
|
allowed={"provider", "endpoint", "model", "api_mode", "tool_call_transport"},
|
|
name="evidence.route",
|
|
)
|
|
matches = [
|
|
route
|
|
for selector_id in config.route_selectors
|
|
for route in config.concrete_routes(selector_id)
|
|
if route.provider == raw.get("provider")
|
|
and route.endpoint == raw.get("endpoint")
|
|
and route.model == raw.get("model")
|
|
and route.api_mode == raw.get("api_mode")
|
|
and route.tool_call_transport == raw.get("tool_call_transport")
|
|
]
|
|
if len(matches) != 1:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "evidence route is ambiguous or unknown"
|
|
)
|
|
return matches[0]
|
|
|
|
|
|
def _parse_evidence(
|
|
value: Any, config: EvoModelConfig
|
|
) -> Mapping[str, CapabilityEvidence]:
|
|
result: dict[str, CapabilityEvidence] = {}
|
|
for item_value in _sequence(value, "capability_evidence"):
|
|
item = _mapping(item_value, "capability_evidence item")
|
|
_strict_keys(
|
|
item,
|
|
allowed={"route", "connectivity", "tool_capability", "probe"},
|
|
name="capability_evidence item",
|
|
)
|
|
route = _route_from_evidence(item.get("route"), config)
|
|
probe = _mapping(item.get("probe"), "capability_evidence.probe")
|
|
_strict_keys(
|
|
probe,
|
|
allowed={
|
|
"route_semantics_hash",
|
|
"endpoint_fingerprint",
|
|
"config_identity_key_id",
|
|
"adapter_revision",
|
|
"fixture_digest",
|
|
"verified_at",
|
|
},
|
|
name="capability_evidence.probe",
|
|
)
|
|
evidence = CapabilityEvidence(
|
|
route,
|
|
_text(item.get("connectivity"), "connectivity"),
|
|
_text(item.get("tool_capability"), "tool_capability"),
|
|
_text(probe.get("route_semantics_hash"), "route_semantics_hash"),
|
|
_text(probe.get("endpoint_fingerprint"), "endpoint_fingerprint"),
|
|
_text(probe.get("config_identity_key_id"), "config_identity_key_id"),
|
|
_text(probe.get("adapter_revision"), "adapter_revision"),
|
|
_text(probe.get("fixture_digest"), "fixture_digest"),
|
|
_text(probe.get("verified_at"), "verified_at"),
|
|
)
|
|
if route.key() in result:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "duplicate capability evidence"
|
|
)
|
|
result[route.key()] = evidence
|
|
return result
|
|
|
|
|
|
def route_semantics_payload(
|
|
config: EvoModelConfig, route: RouteRef
|
|
) -> Mapping[str, Any]:
|
|
provider = config.providers[route.provider]
|
|
endpoint = provider.endpoints[route.endpoint]
|
|
model = provider.models[route.model]
|
|
if config.schema_version == 3:
|
|
return {
|
|
"schema_version": 3,
|
|
"config_revision": config.config_revision,
|
|
"provider_id": provider.key,
|
|
"adapter_id": provider.adapter_id,
|
|
"adapter_revision": provider.adapter_revision,
|
|
"implementation_fingerprint": provider.implementation_fingerprint,
|
|
"wire_protocol": provider.wire_protocol,
|
|
"base_url": endpoint.base_url,
|
|
"credential_ref": endpoint.auth.ref,
|
|
"model_key": route.model,
|
|
"provider_model_id": model.model_id,
|
|
"version_policy": model.version_policy,
|
|
"resolved_model_revision": model.resolved_model_revision,
|
|
"api_mode": route.api_mode,
|
|
"tool_call_transport": route.tool_call_transport,
|
|
# Capability evidence is reusable by every alias that targets this
|
|
# provider/model. Alias, purpose and user parameters belong to the
|
|
# invocation fingerprint produced by the runtime.
|
|
"params": dict(model.params),
|
|
}
|
|
allowed = _PARAM_ALLOWLIST[(provider.protocol, route.api_mode)]
|
|
return {
|
|
"provider": provider.key,
|
|
"protocol": provider.protocol,
|
|
"endpoint": endpoint.name,
|
|
"base_url": endpoint.base_url,
|
|
"auth": asdict(endpoint.auth),
|
|
"headers": dict(endpoint.headers),
|
|
"header_refs": {
|
|
key: asdict(value) for key, value in endpoint.header_refs.items()
|
|
},
|
|
"model": model.model_id,
|
|
"model_capabilities": {
|
|
"context_window": model.context_window,
|
|
"max_output_tokens": model.max_output_tokens,
|
|
"supports_vision": model.supports_vision,
|
|
"supports_reasoning": model.supports_reasoning,
|
|
"allowed_reasoning_efforts": model.allowed_reasoning_efforts,
|
|
"reasoning_mode": model.reasoning_mode,
|
|
"reasoning_enabled_params": model.reasoning_enabled_params,
|
|
"reasoning_disabled_params": model.reasoning_disabled_params,
|
|
},
|
|
"params": {
|
|
"provider": _validate_params(
|
|
provider.params, name="provider.params", allowed=allowed
|
|
),
|
|
"endpoint": _validate_params(
|
|
endpoint.params, name="endpoint.params", allowed=allowed
|
|
),
|
|
"model": _validate_params(
|
|
model.params, name="model.params", allowed=allowed
|
|
),
|
|
},
|
|
"api_mode": route.api_mode,
|
|
"tool_call_transport": route.tool_call_transport,
|
|
"adapter_revision": adapter_revision(provider.protocol, route.api_mode),
|
|
}
|
|
|
|
|
|
def adapter_revision(protocol: str, api_mode: str) -> str:
|
|
return (
|
|
f"ai4sci-provider-adapter-v3:{protocol}:{api_mode}:"
|
|
"bounds=model-cap-v1:reasoning=declarative-v1:"
|
|
"media=descriptor-v1:margin=table-v1"
|
|
)
|
|
|
|
|
|
def route_semantics_hash(
|
|
config: EvoModelConfig, route: RouteRef, identity_key: bytes
|
|
) -> str:
|
|
return hmac_id(identity_key, route_semantics_payload(config, route))
|
|
|
|
|
|
def route_fingerprint(
|
|
config: EvoModelConfig, route: RouteRef, identity_key: bytes
|
|
) -> str:
|
|
return hmac_id(
|
|
identity_key, {"route": route_semantics_payload(config, route), "kind": "route"}
|
|
)
|
|
|
|
|
|
def invocation_fingerprint(
|
|
config: EvoModelConfig,
|
|
route: RouteRef,
|
|
purpose: str,
|
|
final_params: Mapping[str, Any],
|
|
identity_key: bytes,
|
|
) -> str:
|
|
"""Bind an invocation identity to alias, purpose, and final parameters."""
|
|
|
|
return hmac_id(
|
|
identity_key,
|
|
{
|
|
"route": route_semantics_payload(config, route),
|
|
"kind": "invocation",
|
|
"alias": route.selector_id,
|
|
"purpose": purpose,
|
|
"final_params": dict(final_params),
|
|
},
|
|
)
|
|
|
|
|
|
def endpoint_fingerprint(
|
|
config: EvoModelConfig, route: RouteRef, identity_key: bytes
|
|
) -> str:
|
|
endpoint = config.providers[route.provider].endpoints[route.endpoint]
|
|
if config.schema_version == 3:
|
|
provider = config.providers[route.provider]
|
|
return hmac_id(
|
|
identity_key,
|
|
{
|
|
"provider_id": route.provider,
|
|
"base_url": endpoint.base_url,
|
|
"credential_ref": endpoint.auth.ref,
|
|
"adapter_revision": provider.adapter_revision,
|
|
},
|
|
)
|
|
return hmac_id(
|
|
identity_key, {"provider": route.provider, "endpoint": asdict(endpoint)}
|
|
)
|
|
|
|
|
|
_PROCESS_SECRET_FINGERPRINT_KEY = secrets.token_bytes(32)
|
|
|
|
|
|
def resolve_secret(
|
|
reference: SecretReference, *, secret_resolver: SecretResolver | None = None
|
|
) -> ResolvedSecret:
|
|
if secret_resolver is not None:
|
|
resolved = secret_resolver(reference)
|
|
elif reference.ref.startswith("env://"):
|
|
value = os.environ.get(reference.ref[6:], "")
|
|
if not value:
|
|
raise EvoRuntimeError("ROUTE_SECRET_UNAVAILABLE")
|
|
fingerprint = hmac.new(
|
|
_PROCESS_SECRET_FINGERPRINT_KEY, value.encode(), hashlib.sha256
|
|
).hexdigest()
|
|
resolved = ResolvedSecret(value, reference.revision, None, fingerprint)
|
|
else:
|
|
raise EvoRuntimeError("ROUTE_SECRET_UNAVAILABLE")
|
|
if resolved.declared_revision != reference.revision:
|
|
raise EvoRuntimeError("ROUTE_SECRET_UNAVAILABLE")
|
|
if reference.ref.startswith("secret://"):
|
|
expected = reference.ref.rsplit("#", 1)[1]
|
|
if resolved.authoritative_version != expected:
|
|
raise EvoRuntimeError("ROUTE_SECRET_UNAVAILABLE")
|
|
if any(char in resolved.value for char in "\r\n\0"):
|
|
raise EvoRuntimeError("ROUTE_SECRET_UNAVAILABLE")
|
|
return resolved
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class ConfigRevision:
|
|
config_revision: int
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class SaveModelConfigCommand:
|
|
expected_revision: int
|
|
payload: Mapping[str, Any]
|
|
admin_grant: AdminConfigGrant
|
|
operation_id: str = ""
|
|
|
|
|
|
class FileEvoModelConfigStore:
|
|
"""Integrity-checked V2/V3 store with a durable committed-head journal."""
|
|
|
|
def __init__(
|
|
self,
|
|
path: Path | None = None,
|
|
*,
|
|
admin_verifier: AdminConfigGrantVerifier | None = None,
|
|
ops_path: Path | None = None,
|
|
) -> None:
|
|
self.path = path or (get_config_dir() / "model_routes.yaml")
|
|
self._lock = FileLock(str(self.path) + ".lock")
|
|
self._admin_verifier = admin_verifier
|
|
self.ops_path = ops_path or self.path.with_name("model_config_ops.sqlite")
|
|
self._recovery_conflict = False
|
|
self._init_ops()
|
|
self._recover_preparing_operations()
|
|
|
|
def _connect(self) -> sqlite3.Connection:
|
|
connection = sqlite3.connect(self.ops_path)
|
|
connection.row_factory = sqlite3.Row
|
|
connection.execute("PRAGMA foreign_keys=ON")
|
|
connection.execute("PRAGMA busy_timeout=5000")
|
|
connection.execute("PRAGMA synchronous=FULL")
|
|
connection.execute("PRAGMA journal_mode=WAL")
|
|
return connection
|
|
|
|
def _init_ops(self) -> None:
|
|
self.ops_path.parent.mkdir(parents=True, exist_ok=True)
|
|
with self._connect() as connection:
|
|
connection.executescript(
|
|
"""
|
|
CREATE TABLE IF NOT EXISTS committed_head (
|
|
singleton INTEGER PRIMARY KEY CHECK (singleton = 1),
|
|
config_revision INTEGER NOT NULL,
|
|
canonical_payload_hash TEXT NOT NULL,
|
|
config_identity_key_id TEXT NOT NULL,
|
|
commit_operation_id TEXT NOT NULL
|
|
);
|
|
CREATE TABLE IF NOT EXISTS config_operations (
|
|
subject_id TEXT NOT NULL,
|
|
action TEXT NOT NULL,
|
|
operation_id TEXT NOT NULL,
|
|
request_digest TEXT NOT NULL,
|
|
result_digest TEXT,
|
|
status TEXT NOT NULL,
|
|
expected_revision INTEGER,
|
|
target_revision INTEGER,
|
|
proposal_hash TEXT,
|
|
payload_hash TEXT,
|
|
created_at INTEGER NOT NULL,
|
|
updated_at INTEGER NOT NULL,
|
|
PRIMARY KEY (subject_id, action, operation_id)
|
|
);
|
|
"""
|
|
)
|
|
columns = {
|
|
str(row[1])
|
|
for row in connection.execute("PRAGMA table_info(config_operations)")
|
|
}
|
|
if "canonical_payload" not in columns:
|
|
connection.execute(
|
|
"ALTER TABLE config_operations ADD COLUMN canonical_payload TEXT"
|
|
)
|
|
try:
|
|
os.chmod(self.ops_path, stat.S_IRUSR | stat.S_IWUSR)
|
|
except OSError:
|
|
pass
|
|
|
|
def load(self) -> EvoModelConfig:
|
|
with self._lock:
|
|
return self._load_locked()
|
|
|
|
def _load_locked(self) -> EvoModelConfig:
|
|
if self._recovery_conflict:
|
|
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
|
|
with self._connect() as connection:
|
|
head = connection.execute(
|
|
"SELECT * FROM committed_head WHERE singleton = 1"
|
|
).fetchone()
|
|
if not self.path.exists():
|
|
if head is None:
|
|
raise EvoRuntimeError(
|
|
"LLM_ROUTE_CONFIGURATION_REQUIRED", "model routes are unconfigured"
|
|
)
|
|
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
|
|
if head is None:
|
|
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
|
|
config = EvoModelConfig.parse(
|
|
load_yaml_unique(self.path.read_text(encoding="utf-8"))
|
|
)
|
|
payload_hash = sha256_id(config.raw)
|
|
if (
|
|
config.config_revision != int(head["config_revision"])
|
|
or config.config_identity_key_id != str(head["config_identity_key_id"])
|
|
or payload_hash != str(head["canonical_payload_hash"])
|
|
):
|
|
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
|
|
return config
|
|
|
|
def bootstrap_for_development(
|
|
self, payload: Mapping[str, Any], *, operation_id: str = "bootstrap"
|
|
) -> ConfigRevision:
|
|
"""Explicitly initialize an empty development store; never overwrites a head."""
|
|
|
|
with self._lock:
|
|
with self._connect() as connection:
|
|
if connection.execute(
|
|
"SELECT 1 FROM committed_head WHERE singleton = 1"
|
|
).fetchone():
|
|
raise EvoRuntimeError("CONFIG_REVISION_CONFLICT")
|
|
if self.path.exists():
|
|
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
|
|
candidate = dict(payload)
|
|
candidate["config_revision"] = 1
|
|
config = EvoModelConfig.parse(candidate)
|
|
self._write_committed(config.raw, operation_id=operation_id)
|
|
return ConfigRevision(1)
|
|
|
|
def save(self, command: SaveModelConfigCommand) -> ConfigRevision:
|
|
if self._admin_verifier is None:
|
|
raise EvoRuntimeError("ADMIN_CONFIG_FORBIDDEN")
|
|
self._admin_verifier.require_admin(command.admin_grant)
|
|
if command.admin_grant.action != "model_config:commit":
|
|
raise EvoRuntimeError("ADMIN_CONFIG_FORBIDDEN")
|
|
operation_id = command.operation_id or command.admin_grant.operation_id
|
|
if operation_id != command.admin_grant.operation_id:
|
|
raise EvoRuntimeError("IDEMPOTENCY_CONFLICT")
|
|
with self._lock:
|
|
current_revision = 0
|
|
with self._connect() as connection:
|
|
head = connection.execute(
|
|
"SELECT * FROM committed_head WHERE singleton = 1"
|
|
).fetchone()
|
|
if head is not None:
|
|
current_revision = int(head["config_revision"])
|
|
if current_revision != command.expected_revision:
|
|
raise EvoRuntimeError("CONFIG_REVISION_CONFLICT")
|
|
candidate = dict(command.payload)
|
|
candidate["config_revision"] = current_revision + 1
|
|
config = EvoModelConfig.parse(candidate)
|
|
request_digest = sha256_id(
|
|
{
|
|
"expected_revision": command.expected_revision,
|
|
"payload": command.payload,
|
|
}
|
|
)
|
|
if request_digest != command.admin_grant.request_digest:
|
|
raise EvoRuntimeError("CONTRACT_SIGNATURE_INVALID")
|
|
self._write_committed(config.raw, operation_id=operation_id)
|
|
return ConfigRevision(config.config_revision)
|
|
|
|
def current_revision(self) -> int:
|
|
with self._connect() as connection:
|
|
row = connection.execute(
|
|
"SELECT config_revision FROM committed_head WHERE singleton=1"
|
|
).fetchone()
|
|
return int(row[0]) if row is not None else 0
|
|
|
|
def load_revision(self, revision: int) -> EvoModelConfig:
|
|
with self._connect() as connection:
|
|
row = connection.execute(
|
|
"""SELECT canonical_payload FROM config_operations
|
|
WHERE action='model_config:commit' AND status='COMMITTED'
|
|
AND target_revision=? AND canonical_payload IS NOT NULL
|
|
ORDER BY updated_at DESC LIMIT 1""",
|
|
(revision,),
|
|
).fetchone()
|
|
if row is None:
|
|
raise EvoRuntimeError("CONFIG_REVISION_NOT_FOUND")
|
|
return EvoModelConfig.parse(json.loads(str(row["canonical_payload"])))
|
|
|
|
def commit_validated(
|
|
self,
|
|
payload: Mapping[str, Any],
|
|
*,
|
|
expected_revision: int,
|
|
operation_id: str,
|
|
) -> ConfigRevision:
|
|
"""Commit a payload already authorized and evidenced by config admin."""
|
|
|
|
with self._lock:
|
|
target = expected_revision + 1
|
|
candidate = dict(payload)
|
|
candidate["config_revision"] = target
|
|
config = EvoModelConfig.parse(candidate)
|
|
payload_hash = sha256_id(config.raw)
|
|
replay = self._committed_operation_revision(
|
|
operation_id, payload_hash=payload_hash, target_revision=target
|
|
)
|
|
if replay is not None:
|
|
return ConfigRevision(replay)
|
|
current = self.current_revision()
|
|
if current != expected_revision:
|
|
raise EvoRuntimeError("CONFIG_REVISION_CONFLICT")
|
|
self._write_committed(config.raw, operation_id=operation_id)
|
|
return ConfigRevision(config.config_revision)
|
|
|
|
def activate_revision(
|
|
self,
|
|
target_revision: int,
|
|
*,
|
|
expected_revision: int,
|
|
operation_id: str,
|
|
) -> ConfigRevision:
|
|
"""Atomically point the active head at an immutable committed revision."""
|
|
|
|
with self._lock:
|
|
if self.current_revision() != expected_revision:
|
|
raise EvoRuntimeError("CONFIG_REVISION_CONFLICT")
|
|
with self._connect() as connection:
|
|
replay = connection.execute(
|
|
"""SELECT status, target_revision FROM config_operations
|
|
WHERE subject_id='admin' AND action='model_config:rollback'
|
|
AND operation_id=?""",
|
|
(operation_id,),
|
|
).fetchone()
|
|
if replay is not None:
|
|
if int(replay["target_revision"] or 0) != target_revision:
|
|
raise EvoRuntimeError("IDEMPOTENCY_CONFLICT")
|
|
if str(replay["status"]) == "COMMITTED":
|
|
return ConfigRevision(target_revision)
|
|
source = connection.execute(
|
|
"""SELECT canonical_payload FROM config_operations
|
|
WHERE action='model_config:commit' AND status='COMMITTED'
|
|
AND target_revision=? AND canonical_payload IS NOT NULL
|
|
ORDER BY updated_at DESC LIMIT 1""",
|
|
(target_revision,),
|
|
).fetchone()
|
|
if source is None:
|
|
raise EvoRuntimeError("CONFIG_REVISION_NOT_FOUND")
|
|
payload = json.loads(str(source["canonical_payload"]))
|
|
config = EvoModelConfig.parse(payload)
|
|
now = time.time_ns() // 1_000_000
|
|
payload_hash = sha256_id(config.raw)
|
|
with self._connect() as connection:
|
|
connection.execute(
|
|
"""INSERT INTO config_operations
|
|
(subject_id, action, operation_id, request_digest, result_digest,
|
|
status, expected_revision, target_revision, payload_hash,
|
|
canonical_payload, created_at, updated_at)
|
|
VALUES ('admin', 'model_config:rollback', ?, ?, ?, 'PREPARING',
|
|
?, ?, ?, ?, ?, ?)""",
|
|
(
|
|
operation_id,
|
|
sha256_id(
|
|
{
|
|
"expected_revision": expected_revision,
|
|
"target_revision": target_revision,
|
|
}
|
|
),
|
|
sha256_id({"config_revision": target_revision}),
|
|
expected_revision,
|
|
target_revision,
|
|
payload_hash,
|
|
canonical_json_v1(config.raw).decode(),
|
|
now,
|
|
now,
|
|
),
|
|
)
|
|
self._atomic_replace_payload(config.raw)
|
|
with self._connect() as connection:
|
|
connection.execute(
|
|
"""INSERT INTO committed_head
|
|
(singleton, config_revision, canonical_payload_hash,
|
|
config_identity_key_id, commit_operation_id)
|
|
VALUES (1, ?, ?, ?, ?)
|
|
ON CONFLICT(singleton) DO UPDATE SET
|
|
config_revision=excluded.config_revision,
|
|
canonical_payload_hash=excluded.canonical_payload_hash,
|
|
config_identity_key_id=excluded.config_identity_key_id,
|
|
commit_operation_id=excluded.commit_operation_id""",
|
|
(
|
|
target_revision,
|
|
payload_hash,
|
|
config.config_identity_key_id,
|
|
operation_id,
|
|
),
|
|
)
|
|
connection.execute(
|
|
"""UPDATE config_operations SET status='COMMITTED', updated_at=?
|
|
WHERE subject_id='admin' AND action='model_config:rollback'
|
|
AND operation_id=?""",
|
|
(now, operation_id),
|
|
)
|
|
return ConfigRevision(target_revision)
|
|
|
|
def _committed_operation_revision(
|
|
self, operation_id: str, *, payload_hash: str, target_revision: int
|
|
) -> int | None:
|
|
subject_id = "development" if operation_id == "bootstrap" else "admin"
|
|
with self._connect() as connection:
|
|
row = connection.execute(
|
|
"""SELECT status, payload_hash, target_revision
|
|
FROM config_operations
|
|
WHERE subject_id=? AND action='model_config:commit' AND operation_id=?""",
|
|
(subject_id, operation_id),
|
|
).fetchone()
|
|
if row is None:
|
|
return None
|
|
if (
|
|
str(row["payload_hash"]) != payload_hash
|
|
or int(row["target_revision"] or 0) != target_revision
|
|
):
|
|
raise EvoRuntimeError("IDEMPOTENCY_CONFLICT")
|
|
if row["status"] == "COMMITTED":
|
|
return target_revision
|
|
if row["status"] == "CONFLICT":
|
|
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
|
|
return None
|
|
|
|
def _write_committed(
|
|
self, payload: Mapping[str, Any], *, operation_id: str
|
|
) -> None:
|
|
payload_hash = sha256_id(payload)
|
|
revision = int(payload["config_revision"])
|
|
identity_key_id = str(payload["config_identity_key_id"])
|
|
now = time.time_ns() // 1_000_000
|
|
subject_id = "development" if operation_id == "bootstrap" else "admin"
|
|
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
canonical_payload = canonical_json_v1(payload).decode("utf-8")
|
|
with self._connect() as connection:
|
|
existing = connection.execute(
|
|
"""SELECT status, payload_hash, target_revision
|
|
FROM config_operations
|
|
WHERE subject_id=? AND action='model_config:commit' AND operation_id=?""",
|
|
(subject_id, operation_id),
|
|
).fetchone()
|
|
if existing is not None:
|
|
if (
|
|
str(existing["payload_hash"]) != payload_hash
|
|
or int(existing["target_revision"] or 0) != revision
|
|
):
|
|
raise EvoRuntimeError("IDEMPOTENCY_CONFLICT")
|
|
if existing["status"] == "COMMITTED":
|
|
return
|
|
if existing["status"] == "CONFLICT":
|
|
raise EvoRuntimeError("CONFIG_INTEGRITY_MISMATCH")
|
|
else:
|
|
connection.execute(
|
|
"""INSERT INTO config_operations
|
|
(subject_id, action, operation_id, request_digest, status, expected_revision,
|
|
target_revision, payload_hash, canonical_payload, created_at, updated_at)
|
|
VALUES (?, 'model_config:commit', ?, ?, 'PREPARING', ?, ?, ?, ?, ?, ?)""",
|
|
(
|
|
subject_id,
|
|
operation_id,
|
|
payload_hash,
|
|
revision - 1,
|
|
revision,
|
|
payload_hash,
|
|
canonical_payload,
|
|
now,
|
|
now,
|
|
),
|
|
)
|
|
self._atomic_replace_payload(payload)
|
|
with self._connect() as connection:
|
|
connection.execute(
|
|
"""INSERT INTO committed_head
|
|
(singleton, config_revision, canonical_payload_hash, config_identity_key_id, commit_operation_id)
|
|
VALUES (1, ?, ?, ?, ?)
|
|
ON CONFLICT(singleton) DO UPDATE SET
|
|
config_revision=excluded.config_revision,
|
|
canonical_payload_hash=excluded.canonical_payload_hash,
|
|
config_identity_key_id=excluded.config_identity_key_id,
|
|
commit_operation_id=excluded.commit_operation_id""",
|
|
(revision, payload_hash, identity_key_id, operation_id),
|
|
)
|
|
connection.execute(
|
|
"""UPDATE config_operations
|
|
SET status='COMMITTED', result_digest=?, updated_at=?
|
|
WHERE subject_id=? AND action='model_config:commit' AND operation_id=?""",
|
|
(
|
|
sha256_id({"config_revision": revision}),
|
|
now,
|
|
subject_id,
|
|
operation_id,
|
|
),
|
|
)
|
|
|
|
def _atomic_replace_payload(self, payload: Mapping[str, Any]) -> None:
|
|
rendered = yaml.safe_dump(dict(payload), allow_unicode=True, sort_keys=False)
|
|
fd, temporary_name = tempfile.mkstemp(
|
|
prefix=".model_routes.", suffix=".tmp", dir=self.path.parent
|
|
)
|
|
temporary = Path(temporary_name)
|
|
try:
|
|
with os.fdopen(fd, "w", encoding="utf-8") as handle:
|
|
handle.write(rendered)
|
|
handle.flush()
|
|
os.fsync(handle.fileno())
|
|
os.chmod(temporary, stat.S_IRUSR | stat.S_IWUSR)
|
|
os.replace(temporary, self.path)
|
|
directory_fd = os.open(self.path.parent, os.O_RDONLY)
|
|
try:
|
|
os.fsync(directory_fd)
|
|
finally:
|
|
os.close(directory_fd)
|
|
finally:
|
|
temporary.unlink(missing_ok=True)
|
|
|
|
def _recover_preparing_operations(self) -> None:
|
|
"""Complete or safely replay interrupted file/head commits."""
|
|
|
|
with self._lock:
|
|
with self._connect() as connection:
|
|
rows = connection.execute(
|
|
"""SELECT * FROM config_operations
|
|
WHERE action IN ('model_config:commit', 'model_config:rollback')
|
|
AND status='PREPARING'
|
|
ORDER BY created_at"""
|
|
).fetchall()
|
|
for row in rows:
|
|
try:
|
|
payload_text = str(row["canonical_payload"] or "")
|
|
if not payload_text:
|
|
raise ValueError("missing recovery payload")
|
|
payload = json.loads(payload_text)
|
|
target = int(row["target_revision"])
|
|
payload_hash = str(row["payload_hash"])
|
|
if int(payload.get("config_revision", 0)) != target:
|
|
raise ValueError("recovery revision mismatch")
|
|
file_matches_target = False
|
|
if self.path.exists():
|
|
current_payload = load_yaml_unique(
|
|
self.path.read_text(encoding="utf-8")
|
|
)
|
|
file_matches_target = sha256_id(current_payload) == payload_hash
|
|
if not file_matches_target:
|
|
head_revision = self.current_revision()
|
|
if head_revision != int(row["expected_revision"] or 0):
|
|
raise ValueError("recovery head moved")
|
|
self._atomic_replace_payload(payload)
|
|
now = time.time_ns() // 1_000_000
|
|
with self._connect() as connection:
|
|
connection.execute(
|
|
"""INSERT INTO committed_head
|
|
(singleton, config_revision, canonical_payload_hash,
|
|
config_identity_key_id, commit_operation_id)
|
|
VALUES (1, ?, ?, ?, ?)
|
|
ON CONFLICT(singleton) DO UPDATE SET
|
|
config_revision=excluded.config_revision,
|
|
canonical_payload_hash=excluded.canonical_payload_hash,
|
|
config_identity_key_id=excluded.config_identity_key_id,
|
|
commit_operation_id=excluded.commit_operation_id""",
|
|
(
|
|
target,
|
|
payload_hash,
|
|
str(payload["config_identity_key_id"]),
|
|
str(row["operation_id"]),
|
|
),
|
|
)
|
|
connection.execute(
|
|
"""UPDATE config_operations
|
|
SET status='COMMITTED', result_digest=?, updated_at=?
|
|
WHERE subject_id=? AND action=? AND operation_id=?""",
|
|
(
|
|
sha256_id({"config_revision": target}),
|
|
now,
|
|
row["subject_id"],
|
|
row["action"],
|
|
row["operation_id"],
|
|
),
|
|
)
|
|
except Exception:
|
|
self._recovery_conflict = True
|
|
with self._connect() as connection:
|
|
connection.execute(
|
|
"""UPDATE config_operations SET status='CONFLICT', updated_at=?
|
|
WHERE subject_id=? AND action=? AND operation_id=?""",
|
|
(
|
|
time.time_ns() // 1_000_000,
|
|
row["subject_id"],
|
|
row["action"],
|
|
row["operation_id"],
|
|
),
|
|
)
|
|
|
|
|
|
def proposal_payload(
|
|
payload: Mapping[str, Any], *, target_revision: int
|
|
) -> Mapping[str, Any]:
|
|
candidate = dict(payload)
|
|
candidate["config_revision"] = target_revision
|
|
candidate.pop("capability_evidence", None)
|
|
return candidate
|
|
|
|
|
|
def proposal_hash(
|
|
payload: Mapping[str, Any], *, target_revision: int, identity_key: bytes
|
|
) -> str:
|
|
return hmac_id(
|
|
identity_key, proposal_payload(payload, target_revision=target_revision)
|
|
)
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class V2MigrationReport:
|
|
converted_providers: tuple[str, ...]
|
|
converted_models: tuple[str, ...]
|
|
converted_aliases: tuple[str, ...]
|
|
blocking_issues: tuple[Mapping[str, str], ...]
|
|
warnings: tuple[Mapping[str, str], ...]
|
|
required_secret_mappings: tuple[Mapping[str, str], ...]
|
|
|
|
|
|
def convert_v2_to_v3_draft(
|
|
payload: Mapping[str, Any],
|
|
*,
|
|
target_revision: int,
|
|
config_identity_key_id: str,
|
|
) -> tuple[Mapping[str, Any], V2MigrationReport]:
|
|
"""Convert unambiguous V2 routes to a V3 Draft without guessing endpoints."""
|
|
|
|
source = EvoModelConfig.parse(payload, require_evidence=False)
|
|
if source.schema_version != 2:
|
|
raise EvoRuntimeError("LLM_ROUTE_CONFIGURATION_REQUIRED", "source must be V2")
|
|
adapter_map = {
|
|
"openai": ("openai", "openai-v1", "openai_native"),
|
|
"anthropic": ("anthropic", "anthropic-v1", "anthropic_native"),
|
|
"custom-openai": (
|
|
"generic-openai-compatible",
|
|
"generic-openai-compatible-v1",
|
|
"openai_compatible",
|
|
),
|
|
}
|
|
providers: list[Mapping[str, Any]] = []
|
|
aliases: list[Mapping[str, Any]] = []
|
|
converted_models: list[str] = []
|
|
blocking: list[Mapping[str, str]] = []
|
|
warnings: list[Mapping[str, str]] = []
|
|
secret_mappings: list[Mapping[str, str]] = []
|
|
model_keys: dict[tuple[str, str], str] = {}
|
|
for provider_id, provider in source.providers.items():
|
|
referenced = {
|
|
route.endpoint
|
|
for selector_id in source.route_selectors
|
|
for route in source.concrete_routes(selector_id)
|
|
if route.provider == provider_id
|
|
}
|
|
if len(referenced) != 1:
|
|
blocking.append(
|
|
{
|
|
"code": "MULTIPLE_ENDPOINTS_REQUIRE_SPLIT",
|
|
"provider_id": provider_id,
|
|
"detail": ",".join(sorted(referenced)),
|
|
}
|
|
)
|
|
continue
|
|
mapped = adapter_map.get(provider.protocol)
|
|
if mapped is None:
|
|
blocking.append(
|
|
{
|
|
"code": "ADAPTER_MAPPING_REQUIRED",
|
|
"provider_id": provider_id,
|
|
"detail": provider.protocol,
|
|
}
|
|
)
|
|
continue
|
|
adapter_id, revision, wire = mapped
|
|
if provider.protocol == "custom-openai":
|
|
warnings.append(
|
|
{
|
|
"code": "GENERIC_ADAPTER_REQUIRES_CONFIRMATION",
|
|
"provider_id": provider_id,
|
|
"detail": "Base URL was not used to infer a vendor adapter",
|
|
}
|
|
)
|
|
endpoint = provider.endpoints[next(iter(referenced))]
|
|
credential_ref = (
|
|
f"secret://model-providers/{provider_id}#{endpoint.auth.revision}"
|
|
)
|
|
secret_mappings.append(
|
|
{
|
|
"provider_id": provider_id,
|
|
"old_ref": endpoint.auth.ref,
|
|
"new_ref": credential_ref,
|
|
}
|
|
)
|
|
models: list[Mapping[str, Any]] = []
|
|
for index, (legacy_key, model) in enumerate(provider.models.items()):
|
|
model_key = re.sub(r"[^A-Za-z0-9._-]+", "-", legacy_key).strip("-")
|
|
model_key = model_key or f"model-{index + 1}"
|
|
if any(item["model_key"] == model_key for item in models):
|
|
model_key = f"{model_key}-{index + 1}"
|
|
model_keys[(provider_id, legacy_key)] = model_key
|
|
converted_models.append(f"{provider_id}/{model_key}")
|
|
selector = next(
|
|
(
|
|
item
|
|
for item in source.route_selectors.values()
|
|
if item.provider == provider_id and item.model == legacy_key
|
|
),
|
|
None,
|
|
)
|
|
models.append(
|
|
{
|
|
"model_key": model_key,
|
|
"provider_model_id": model.model_id,
|
|
"version_policy": "rolling",
|
|
"resolved_model_revision": None,
|
|
"display_name": model.model_id,
|
|
"description": "Migrated from model routes V2",
|
|
"enabled": True,
|
|
"tags": ["migrated-v2"],
|
|
"invocation": {
|
|
"api_mode": selector.api_mode
|
|
if selector
|
|
else "chat_completions",
|
|
"tool_call_transport": "native",
|
|
},
|
|
"capabilities": {
|
|
"text": True,
|
|
"vision": model.supports_vision,
|
|
"video": False,
|
|
"documents": False,
|
|
"tools": True,
|
|
"structured_output": False,
|
|
"thinking": model.supports_reasoning,
|
|
},
|
|
"limits": {
|
|
"context_tokens": model.context_window,
|
|
"max_output_tokens": model.max_output_tokens,
|
|
},
|
|
"parameters": {
|
|
"defaults": dict(model.params),
|
|
"purpose_overrides": {},
|
|
"user_options": {},
|
|
"constraints": [],
|
|
},
|
|
"access": {
|
|
"visibility": "role_based"
|
|
if model.allowed_roles
|
|
else "authenticated",
|
|
"roles": list(model.allowed_roles),
|
|
"groups": [],
|
|
"users": [],
|
|
},
|
|
"billing": {
|
|
"sku": model.quote.billing_sku,
|
|
"pricing_revision": model.quote.pricing_revision,
|
|
"currency": model.quote.currency,
|
|
"unit_scale": model.quote.unit_scale,
|
|
"input_microunits_per_million": model.quote.input_microunits_per_million,
|
|
"output_microunits_per_million": model.quote.output_microunits_per_million,
|
|
"cached_microunits_per_million": model.quote.cached_input_microunits_per_million,
|
|
"multiplier": model.quote.multiplier,
|
|
},
|
|
}
|
|
)
|
|
providers.append(
|
|
{
|
|
"provider_id": provider_id,
|
|
"display_name": provider_id,
|
|
"adapter_id": adapter_id,
|
|
"adapter_revision": revision,
|
|
"wire_protocol": wire,
|
|
"enabled": True,
|
|
"connection": {
|
|
"base_url": endpoint.base_url,
|
|
"credential_ref": credential_ref,
|
|
},
|
|
"defaults": {},
|
|
"models": models,
|
|
}
|
|
)
|
|
for alias, selector_id in source.main_routes.selectable.items():
|
|
selector = source.route_selectors[selector_id]
|
|
key = model_keys.get((selector.provider, selector.model))
|
|
if key is None:
|
|
continue
|
|
aliases.append(
|
|
{
|
|
"alias": alias,
|
|
"display_name": alias,
|
|
"provider_ref": selector.provider,
|
|
"model_ref": key,
|
|
"enabled": True,
|
|
"access": {
|
|
"visibility": "authenticated",
|
|
"roles": [],
|
|
"groups": [],
|
|
"users": [],
|
|
},
|
|
"defaults": {},
|
|
}
|
|
)
|
|
default_alias = (
|
|
source.main_routes.default_alias
|
|
if any(item["alias"] == source.main_routes.default_alias for item in aliases)
|
|
else (str(aliases[0]["alias"]) if aliases else "")
|
|
)
|
|
title_route = source.concrete_routes(source.title_selector_id)[0]
|
|
title_alias = next(
|
|
(
|
|
str(item["alias"])
|
|
for item in aliases
|
|
if item["provider_ref"] == title_route.provider
|
|
and item["model_ref"]
|
|
== model_keys.get((title_route.provider, title_route.model))
|
|
),
|
|
default_alias,
|
|
)
|
|
draft = {
|
|
"schema_version": 3,
|
|
"config_revision": target_revision,
|
|
"config_identity_key_id": config_identity_key_id,
|
|
"runtime_defaults": {},
|
|
"providers": providers,
|
|
"aliases": aliases,
|
|
"purpose_defaults": {purpose: {} for purpose in _PURPOSES},
|
|
"purpose_routes": {
|
|
"main_agent": {"default_alias": default_alias},
|
|
"tool_selector": "inherit_main",
|
|
"deepagents_summarizer": "inherit_main",
|
|
"title": {"default_alias": title_alias},
|
|
},
|
|
"purpose_call_limits": {
|
|
purpose: asdict(limit)
|
|
for purpose, limit in source.purpose_call_limits.items()
|
|
},
|
|
"health_policy": {
|
|
"provider_connection": asdict(source.route_health),
|
|
"model_route": asdict(source.route_health),
|
|
},
|
|
"web_runtime": asdict(source.web_runtime),
|
|
"capability_evidence": [],
|
|
}
|
|
return draft, V2MigrationReport(
|
|
converted_providers=tuple(str(item["provider_id"]) for item in providers),
|
|
converted_models=tuple(converted_models),
|
|
converted_aliases=tuple(str(item["alias"]) for item in aliases),
|
|
blocking_issues=tuple(blocking),
|
|
warnings=tuple(warnings),
|
|
required_secret_mappings=tuple(secret_mappings),
|
|
)
|