From 2b7eb97fbf69d7264a277bc80e008da9c70dd95d Mon Sep 17 00:00:00 2001 From: nftpoetrist <264138787+nftpoetrist@users.noreply.github.com> Date: Sat, 5 Sep 2026 23:13:31 +0300 Subject: [PATCH] fix(gateway): scope the OpenAI-compat Idempotency-Key cache by profile _run_idempotent() (chat completions / responses) keyed the shared _idem_cache purely on the client-supplied Idempotency-Key header plus a body fingerprint, with no profile identity anywhere in the key. Under gateway.multiplex_profiles, both the native routes and their /p// mirrors share the same process-global cache, so two different profiles' clients sending the same Idempotency-Key with a matching fingerprint (a realistic case for automation/SDKs that derive the key deterministically from request content) silently share a cached response instead of each profile running its own turn. The sibling /v1/runs admission API already scopes its own idempotency namespace this way (_run_idempotency_scope() folds in _api_request_profile.get() or "default"); this mirrors that exact expression into _run_idempotent()'s cache key. --- gateway/platforms/api_server_openai_routes.py | 16 +++- tests/gateway/test_api_server.py | 89 +++++++++++++++++++ 2 files changed, 101 insertions(+), 4 deletions(-) diff --git a/gateway/platforms/api_server_openai_routes.py b/gateway/platforms/api_server_openai_routes.py index 3d4c92dce9..3745690cd7 100644 --- a/gateway/platforms/api_server_openai_routes.py +++ b/gateway/platforms/api_server_openai_routes.py @@ -600,15 +600,23 @@ class OpenAICompatRoutesMixin: async def _run_idempotent( self, request: "web.Request", body: Dict[str, Any], compute, *, log_label: str, fingerprint_keys: List[str]) -> tuple: - """Run ``compute()`` once per Idempotency-Key + body fingerprint -> - ``((result, usage), None)`` or ``(None, 500 response)``.""" + """Run ``compute()`` once per profile + Idempotency-Key + body fingerprint -> + ``((result, usage), None)`` or ``(None, 500 response)``. + + The profile is folded into the cache key (mirroring ``_run_idempotency_scope``'s + ``_api_request_profile.get() or "default"``) so two ``/p//...`` mirrors + under ``gateway.multiplex_profiles`` never share a cached response for a client- + supplied Idempotency-Key that happens to collide across profiles. + """ from gateway.platforms.api_server import ( - _error_response, _idem_cache, _make_request_fingerprint) + _api_request_profile, _error_response, _idem_cache, _make_request_fingerprint) idempotency_key = request.headers.get("Idempotency-Key") try: if idempotency_key: + profile = _api_request_profile.get() or "default" + scoped_key = f"{profile}\0{idempotency_key}" fp = _make_request_fingerprint(body, keys=fingerprint_keys) - result, usage = await _idem_cache.get_or_set(idempotency_key, fp, compute) + result, usage = await _idem_cache.get_or_set(scoped_key, fp, compute) else: result, usage = await compute() return (result, usage), None diff --git a/tests/gateway/test_api_server.py b/tests/gateway/test_api_server.py index 8abb8daf90..4c256196a7 100644 --- a/tests/gateway/test_api_server.py +++ b/tests/gateway/test_api_server.py @@ -30,6 +30,7 @@ from gateway.config import GatewayConfig, Platform, PlatformConfig from gateway.platforms.api_server import ( APIServerAdapter, ResponseStore, + _api_request_profile, _IdempotencyCache, _derive_chat_session_id, _hermes_version, @@ -148,6 +149,94 @@ class TestIdempotencyCache: assert first_result == second_result == ("response", {"total_tokens": 1}) +class TestRunIdempotentProfileScope: + """``_run_idempotent`` (the chat completions / responses idempotency wrapper) must scope + its cache key by the request's ``/p//`` identity, mirroring + ``_run_idempotency_scope``'s ``_api_request_profile.get() or "default"`` — otherwise two + multiplexed profiles sharing a client-supplied Idempotency-Key silently share a cached + response instead of each running its own turn.""" + + @pytest.mark.asyncio + async def test_same_key_different_profiles_do_not_share_a_cached_response(self, adapter, monkeypatch): + monkeypatch.setattr("gateway.platforms.api_server._idem_cache", _IdempotencyCache()) + request = MagicMock() + request.headers = {"Idempotency-Key": "client-supplied-key"} + body = {"model": "gpt-5.5", "messages": [{"role": "user", "content": "hi"}]} + calls = [] + + async def compute(): + calls.append(1) + return (f"response-{len(calls)}", {"total_tokens": len(calls)}) + + token_a = _api_request_profile.set("profile-a") + try: + outcome_a, err_a = await adapter._run_idempotent( + request, body, compute, log_label="test", fingerprint_keys=["model", "messages"]) + finally: + _api_request_profile.reset(token_a) + + token_b = _api_request_profile.set("profile-b") + try: + outcome_b, err_b = await adapter._run_idempotent( + request, body, compute, log_label="test", fingerprint_keys=["model", "messages"]) + finally: + _api_request_profile.reset(token_b) + + assert err_a is None and err_b is None + assert len(calls) == 2, "each profile must run its own turn, not reuse the other's cached response" + assert outcome_a != outcome_b + + @pytest.mark.asyncio + async def test_same_key_same_profile_still_dedupes(self, adapter, monkeypatch): + """Regression guard: profile-scoping the cache key must not break same-profile dedup, + which is the whole point of the Idempotency-Key contract.""" + monkeypatch.setattr("gateway.platforms.api_server._idem_cache", _IdempotencyCache()) + request = MagicMock() + request.headers = {"Idempotency-Key": "client-supplied-key"} + body = {"model": "gpt-5.5", "messages": [{"role": "user", "content": "hi"}]} + calls = [] + + async def compute(): + calls.append(1) + return (f"response-{len(calls)}", {"total_tokens": len(calls)}) + + token = _api_request_profile.set("profile-a") + try: + outcome_1, err_1 = await adapter._run_idempotent( + request, body, compute, log_label="test", fingerprint_keys=["model", "messages"]) + outcome_2, err_2 = await adapter._run_idempotent( + request, body, compute, log_label="test", fingerprint_keys=["model", "messages"]) + finally: + _api_request_profile.reset(token) + + assert err_1 is None and err_2 is None + assert len(calls) == 1, "second call with the same key+profile+fingerprint must reuse the cached response" + assert outcome_1 == outcome_2 + + @pytest.mark.asyncio + async def test_unset_profile_defaults_consistently(self, adapter, monkeypatch): + """No /p// prefix (single-profile / multiplexing off) must resolve to the + same 'default' scope on every call, so dedup still works when multiplexing is off.""" + monkeypatch.setattr("gateway.platforms.api_server._idem_cache", _IdempotencyCache()) + request = MagicMock() + request.headers = {"Idempotency-Key": "client-supplied-key"} + body = {"model": "gpt-5.5", "messages": [{"role": "user", "content": "hi"}]} + calls = [] + + async def compute(): + calls.append(1) + return (f"response-{len(calls)}", {"total_tokens": len(calls)}) + + outcome_1, err_1 = await adapter._run_idempotent( + request, body, compute, log_label="test", fingerprint_keys=["model", "messages"]) + outcome_2, err_2 = await adapter._run_idempotent( + request, body, compute, log_label="test", fingerprint_keys=["model", "messages"]) + + assert err_1 is None and err_2 is None + assert len(calls) == 1 + assert outcome_1 == outcome_2 + + # --------------------------------------------------------------------------- # Adapter initialization # ---------------------------------------------------------------------------