fix(gateway): scope the OpenAI-compat Idempotency-Key cache by profile

_run_idempotent() (chat completions / responses) keyed the shared
_idem_cache purely on the client-supplied Idempotency-Key header plus a
body fingerprint, with no profile identity anywhere in the key. Under
gateway.multiplex_profiles, both the native routes and their /p/<profile>/
mirrors share the same process-global cache, so two different profiles'
clients sending the same Idempotency-Key with a matching fingerprint (a
realistic case for automation/SDKs that derive the key deterministically
from request content) silently share a cached response instead of each
profile running its own turn.

The sibling /v1/runs admission API already scopes its own idempotency
namespace this way (_run_idempotency_scope() folds in
_api_request_profile.get() or "default"); this mirrors that exact
expression into _run_idempotent()'s cache key.
This commit is contained in:
nftpoetrist
2026-09-05 23:13:31 +03:00
committed by Teknium
parent 199c66f70f
commit 2b7eb97fbf
2 changed files with 101 additions and 4 deletions
+12 -4
View File
@@ -600,15 +600,23 @@ class OpenAICompatRoutesMixin:
async def _run_idempotent(
self, request: "web.Request", body: Dict[str, Any], compute, *,
log_label: str, fingerprint_keys: List[str]) -> tuple:
"""Run ``compute()`` once per Idempotency-Key + body fingerprint ->
``((result, usage), None)`` or ``(None, 500 response)``."""
"""Run ``compute()`` once per profile + Idempotency-Key + body fingerprint ->
``((result, usage), None)`` or ``(None, 500 response)``.
The profile is folded into the cache key (mirroring ``_run_idempotency_scope``'s
``_api_request_profile.get() or "default"``) so two ``/p/<profile>/...`` mirrors
under ``gateway.multiplex_profiles`` never share a cached response for a client-
supplied Idempotency-Key that happens to collide across profiles.
"""
from gateway.platforms.api_server import (
_error_response, _idem_cache, _make_request_fingerprint)
_api_request_profile, _error_response, _idem_cache, _make_request_fingerprint)
idempotency_key = request.headers.get("Idempotency-Key")
try:
if idempotency_key:
profile = _api_request_profile.get() or "default"
scoped_key = f"{profile}\0{idempotency_key}"
fp = _make_request_fingerprint(body, keys=fingerprint_keys)
result, usage = await _idem_cache.get_or_set(idempotency_key, fp, compute)
result, usage = await _idem_cache.get_or_set(scoped_key, fp, compute)
else:
result, usage = await compute()
return (result, usage), None
+89
View File
@@ -30,6 +30,7 @@ from gateway.config import GatewayConfig, Platform, PlatformConfig
from gateway.platforms.api_server import (
APIServerAdapter,
ResponseStore,
_api_request_profile,
_IdempotencyCache,
_derive_chat_session_id,
_hermes_version,
@@ -148,6 +149,94 @@ class TestIdempotencyCache:
assert first_result == second_result == ("response", {"total_tokens": 1})
class TestRunIdempotentProfileScope:
"""``_run_idempotent`` (the chat completions / responses idempotency wrapper) must scope
its cache key by the request's ``/p/<profile>/`` identity, mirroring
``_run_idempotency_scope``'s ``_api_request_profile.get() or "default"`` — otherwise two
multiplexed profiles sharing a client-supplied Idempotency-Key silently share a cached
response instead of each running its own turn."""
@pytest.mark.asyncio
async def test_same_key_different_profiles_do_not_share_a_cached_response(self, adapter, monkeypatch):
monkeypatch.setattr("gateway.platforms.api_server._idem_cache", _IdempotencyCache())
request = MagicMock()
request.headers = {"Idempotency-Key": "client-supplied-key"}
body = {"model": "gpt-5.5", "messages": [{"role": "user", "content": "hi"}]}
calls = []
async def compute():
calls.append(1)
return (f"response-{len(calls)}", {"total_tokens": len(calls)})
token_a = _api_request_profile.set("profile-a")
try:
outcome_a, err_a = await adapter._run_idempotent(
request, body, compute, log_label="test", fingerprint_keys=["model", "messages"])
finally:
_api_request_profile.reset(token_a)
token_b = _api_request_profile.set("profile-b")
try:
outcome_b, err_b = await adapter._run_idempotent(
request, body, compute, log_label="test", fingerprint_keys=["model", "messages"])
finally:
_api_request_profile.reset(token_b)
assert err_a is None and err_b is None
assert len(calls) == 2, "each profile must run its own turn, not reuse the other's cached response"
assert outcome_a != outcome_b
@pytest.mark.asyncio
async def test_same_key_same_profile_still_dedupes(self, adapter, monkeypatch):
"""Regression guard: profile-scoping the cache key must not break same-profile dedup,
which is the whole point of the Idempotency-Key contract."""
monkeypatch.setattr("gateway.platforms.api_server._idem_cache", _IdempotencyCache())
request = MagicMock()
request.headers = {"Idempotency-Key": "client-supplied-key"}
body = {"model": "gpt-5.5", "messages": [{"role": "user", "content": "hi"}]}
calls = []
async def compute():
calls.append(1)
return (f"response-{len(calls)}", {"total_tokens": len(calls)})
token = _api_request_profile.set("profile-a")
try:
outcome_1, err_1 = await adapter._run_idempotent(
request, body, compute, log_label="test", fingerprint_keys=["model", "messages"])
outcome_2, err_2 = await adapter._run_idempotent(
request, body, compute, log_label="test", fingerprint_keys=["model", "messages"])
finally:
_api_request_profile.reset(token)
assert err_1 is None and err_2 is None
assert len(calls) == 1, "second call with the same key+profile+fingerprint must reuse the cached response"
assert outcome_1 == outcome_2
@pytest.mark.asyncio
async def test_unset_profile_defaults_consistently(self, adapter, monkeypatch):
"""No /p/<profile>/ prefix (single-profile / multiplexing off) must resolve to the
same 'default' scope on every call, so dedup still works when multiplexing is off."""
monkeypatch.setattr("gateway.platforms.api_server._idem_cache", _IdempotencyCache())
request = MagicMock()
request.headers = {"Idempotency-Key": "client-supplied-key"}
body = {"model": "gpt-5.5", "messages": [{"role": "user", "content": "hi"}]}
calls = []
async def compute():
calls.append(1)
return (f"response-{len(calls)}", {"total_tokens": len(calls)})
outcome_1, err_1 = await adapter._run_idempotent(
request, body, compute, log_label="test", fingerprint_keys=["model", "messages"])
outcome_2, err_2 = await adapter._run_idempotent(
request, body, compute, log_label="test", fingerprint_keys=["model", "messages"])
assert err_1 is None and err_2 is None
assert len(calls) == 1
assert outcome_1 == outcome_2
# ---------------------------------------------------------------------------
# Adapter initialization
# ---------------------------------------------------------------------------