fix(gateway): scope the OpenAI-compat Idempotency-Key cache by profile
_run_idempotent() (chat completions / responses) keyed the shared _idem_cache purely on the client-supplied Idempotency-Key header plus a body fingerprint, with no profile identity anywhere in the key. Under gateway.multiplex_profiles, both the native routes and their /p/<profile>/ mirrors share the same process-global cache, so two different profiles' clients sending the same Idempotency-Key with a matching fingerprint (a realistic case for automation/SDKs that derive the key deterministically from request content) silently share a cached response instead of each profile running its own turn. The sibling /v1/runs admission API already scopes its own idempotency namespace this way (_run_idempotency_scope() folds in _api_request_profile.get() or "default"); this mirrors that exact expression into _run_idempotent()'s cache key.
This commit is contained in:
@@ -600,15 +600,23 @@ class OpenAICompatRoutesMixin:
|
||||
async def _run_idempotent(
|
||||
self, request: "web.Request", body: Dict[str, Any], compute, *,
|
||||
log_label: str, fingerprint_keys: List[str]) -> tuple:
|
||||
"""Run ``compute()`` once per Idempotency-Key + body fingerprint ->
|
||||
``((result, usage), None)`` or ``(None, 500 response)``."""
|
||||
"""Run ``compute()`` once per profile + Idempotency-Key + body fingerprint ->
|
||||
``((result, usage), None)`` or ``(None, 500 response)``.
|
||||
|
||||
The profile is folded into the cache key (mirroring ``_run_idempotency_scope``'s
|
||||
``_api_request_profile.get() or "default"``) so two ``/p/<profile>/...`` mirrors
|
||||
under ``gateway.multiplex_profiles`` never share a cached response for a client-
|
||||
supplied Idempotency-Key that happens to collide across profiles.
|
||||
"""
|
||||
from gateway.platforms.api_server import (
|
||||
_error_response, _idem_cache, _make_request_fingerprint)
|
||||
_api_request_profile, _error_response, _idem_cache, _make_request_fingerprint)
|
||||
idempotency_key = request.headers.get("Idempotency-Key")
|
||||
try:
|
||||
if idempotency_key:
|
||||
profile = _api_request_profile.get() or "default"
|
||||
scoped_key = f"{profile}\0{idempotency_key}"
|
||||
fp = _make_request_fingerprint(body, keys=fingerprint_keys)
|
||||
result, usage = await _idem_cache.get_or_set(idempotency_key, fp, compute)
|
||||
result, usage = await _idem_cache.get_or_set(scoped_key, fp, compute)
|
||||
else:
|
||||
result, usage = await compute()
|
||||
return (result, usage), None
|
||||
|
||||
@@ -30,6 +30,7 @@ from gateway.config import GatewayConfig, Platform, PlatformConfig
|
||||
from gateway.platforms.api_server import (
|
||||
APIServerAdapter,
|
||||
ResponseStore,
|
||||
_api_request_profile,
|
||||
_IdempotencyCache,
|
||||
_derive_chat_session_id,
|
||||
_hermes_version,
|
||||
@@ -148,6 +149,94 @@ class TestIdempotencyCache:
|
||||
assert first_result == second_result == ("response", {"total_tokens": 1})
|
||||
|
||||
|
||||
class TestRunIdempotentProfileScope:
|
||||
"""``_run_idempotent`` (the chat completions / responses idempotency wrapper) must scope
|
||||
its cache key by the request's ``/p/<profile>/`` identity, mirroring
|
||||
``_run_idempotency_scope``'s ``_api_request_profile.get() or "default"`` — otherwise two
|
||||
multiplexed profiles sharing a client-supplied Idempotency-Key silently share a cached
|
||||
response instead of each running its own turn."""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_same_key_different_profiles_do_not_share_a_cached_response(self, adapter, monkeypatch):
|
||||
monkeypatch.setattr("gateway.platforms.api_server._idem_cache", _IdempotencyCache())
|
||||
request = MagicMock()
|
||||
request.headers = {"Idempotency-Key": "client-supplied-key"}
|
||||
body = {"model": "gpt-5.5", "messages": [{"role": "user", "content": "hi"}]}
|
||||
calls = []
|
||||
|
||||
async def compute():
|
||||
calls.append(1)
|
||||
return (f"response-{len(calls)}", {"total_tokens": len(calls)})
|
||||
|
||||
token_a = _api_request_profile.set("profile-a")
|
||||
try:
|
||||
outcome_a, err_a = await adapter._run_idempotent(
|
||||
request, body, compute, log_label="test", fingerprint_keys=["model", "messages"])
|
||||
finally:
|
||||
_api_request_profile.reset(token_a)
|
||||
|
||||
token_b = _api_request_profile.set("profile-b")
|
||||
try:
|
||||
outcome_b, err_b = await adapter._run_idempotent(
|
||||
request, body, compute, log_label="test", fingerprint_keys=["model", "messages"])
|
||||
finally:
|
||||
_api_request_profile.reset(token_b)
|
||||
|
||||
assert err_a is None and err_b is None
|
||||
assert len(calls) == 2, "each profile must run its own turn, not reuse the other's cached response"
|
||||
assert outcome_a != outcome_b
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_same_key_same_profile_still_dedupes(self, adapter, monkeypatch):
|
||||
"""Regression guard: profile-scoping the cache key must not break same-profile dedup,
|
||||
which is the whole point of the Idempotency-Key contract."""
|
||||
monkeypatch.setattr("gateway.platforms.api_server._idem_cache", _IdempotencyCache())
|
||||
request = MagicMock()
|
||||
request.headers = {"Idempotency-Key": "client-supplied-key"}
|
||||
body = {"model": "gpt-5.5", "messages": [{"role": "user", "content": "hi"}]}
|
||||
calls = []
|
||||
|
||||
async def compute():
|
||||
calls.append(1)
|
||||
return (f"response-{len(calls)}", {"total_tokens": len(calls)})
|
||||
|
||||
token = _api_request_profile.set("profile-a")
|
||||
try:
|
||||
outcome_1, err_1 = await adapter._run_idempotent(
|
||||
request, body, compute, log_label="test", fingerprint_keys=["model", "messages"])
|
||||
outcome_2, err_2 = await adapter._run_idempotent(
|
||||
request, body, compute, log_label="test", fingerprint_keys=["model", "messages"])
|
||||
finally:
|
||||
_api_request_profile.reset(token)
|
||||
|
||||
assert err_1 is None and err_2 is None
|
||||
assert len(calls) == 1, "second call with the same key+profile+fingerprint must reuse the cached response"
|
||||
assert outcome_1 == outcome_2
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_unset_profile_defaults_consistently(self, adapter, monkeypatch):
|
||||
"""No /p/<profile>/ prefix (single-profile / multiplexing off) must resolve to the
|
||||
same 'default' scope on every call, so dedup still works when multiplexing is off."""
|
||||
monkeypatch.setattr("gateway.platforms.api_server._idem_cache", _IdempotencyCache())
|
||||
request = MagicMock()
|
||||
request.headers = {"Idempotency-Key": "client-supplied-key"}
|
||||
body = {"model": "gpt-5.5", "messages": [{"role": "user", "content": "hi"}]}
|
||||
calls = []
|
||||
|
||||
async def compute():
|
||||
calls.append(1)
|
||||
return (f"response-{len(calls)}", {"total_tokens": len(calls)})
|
||||
|
||||
outcome_1, err_1 = await adapter._run_idempotent(
|
||||
request, body, compute, log_label="test", fingerprint_keys=["model", "messages"])
|
||||
outcome_2, err_2 = await adapter._run_idempotent(
|
||||
request, body, compute, log_label="test", fingerprint_keys=["model", "messages"])
|
||||
|
||||
assert err_1 is None and err_2 is None
|
||||
assert len(calls) == 1
|
||||
assert outcome_1 == outcome_2
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Adapter initialization
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
Reference in New Issue
Block a user