fix(ccproxy): update Responses API handling and patch system role con… (#149)

* fix(ccproxy): update Responses API handling and patch system role conversion

* fix(ccproxy): streamline _agenerate method in system to developer patch

* fix(ccproxy): improve handling of None output in Codex compatibility patch
This commit is contained in:
Xi Zhang
2026-04-09 12:51:05 +02:00
committed by GitHub
parent 3e493233cf
commit d5b982c980
3 changed files with 210 additions and 12 deletions
+191
View File
@@ -7,6 +7,8 @@ Patches:
- _patch_anthropic_proxy_compat: ccproxy dict→Pydantic model mismatch
- _patch_openrouter_reasoning_details: reasoning_details schema errors
- _patch_openai_compat_content: list content→string for strict APIs
- _patch_ccproxy_codex_compat: ccproxy model fixes + langchain None guard
- _patch_ccproxy_system_to_developer: system→developer role for ccproxy
Utilities:
- _is_ccproxy_codex: detect ccproxy Codex OAuth adapter
@@ -59,6 +61,112 @@ def _patch_anthropic_proxy_compat() -> None:
_patch_anthropic_proxy_compat()
# ---------------------------------------------------------------------------
# Patch: ccproxy-api 0.2.7 Codex compatibility.
#
# 1) ResponseObject.output is required but upstream may omit it → 502.
# Fix: make output default to [].
# 2) CodexMessage.role only allows "user"/"assistant" → 400 on system msgs.
# Fix: widen to also accept "system" and "developer".
# 3) langchain-openai iterates response.output which can be None after the
# proxy strips it. Fix: guard in _construct_lc_result_from_responses_api.
# ---------------------------------------------------------------------------
def _patch_ccproxy_codex_compat() -> None:
"""Patch ccproxy-api models for Responses API compatibility."""
# 1) Make ResponseObject.output optional (default=[])
try:
import ccproxy.llms.models.openai as _oai_mod
_OrigResponse = _oai_mod.ResponseObject
from pydantic import Field as _PydanticField
class _PatchedResponseObject(_OrigResponse): # type: ignore[misc]
output: list = _PydanticField(default_factory=list) # type: ignore[assignment]
model_config = _OrigResponse.model_config.copy()
_PatchedResponseObject.__name__ = "ResponseObject"
_PatchedResponseObject.__qualname__ = "ResponseObject"
_oai_mod.ResponseObject = _PatchedResponseObject # type: ignore[misc]
# Also patch modules that import ResponseObject directly
for _mod_path in (
"ccproxy.llms.formatters.openai_to_openai.responses",
"ccproxy.llms.formatters.anthropic_to_openai.responses",
):
try:
import importlib
_mod = importlib.import_module(_mod_path)
if hasattr(_mod, "ResponseObject"):
_mod.ResponseObject = _PatchedResponseObject # type: ignore[attr-defined]
except Exception:
pass
except Exception:
pass
# 2) Widen CodexMessage.role to accept system/developer
try:
from typing import Annotated, Literal
import ccproxy.plugins.codex.models as _codex_mod
_OrigMessage = _codex_mod.CodexMessage
from pydantic import Field as _Field
class _PatchedCodexMessage(_OrigMessage): # type: ignore[misc]
role: Annotated[ # type: ignore[assignment]
Literal["user", "assistant", "system", "developer"],
_Field(description="Message role"),
]
_PatchedCodexMessage.__name__ = "CodexMessage"
_PatchedCodexMessage.__qualname__ = "CodexMessage"
_codex_mod.CodexMessage = _PatchedCodexMessage # type: ignore[misc]
except Exception:
pass
# 3) Fix StreamingBufferService returning response.completed event
# whose output is None/empty, instead of using accumulated outputs.
try:
from ccproxy.llms.streaming.accumulators import ResponsesAccumulator
_orig_get = ResponsesAccumulator.get_completed_response
def _patched_get(self: Any) -> dict | None:
result = _orig_get(self)
if result is not None:
output = result.get("output")
if output is None:
# output field lost — force rebuild from accumulated items
return None
return result
ResponsesAccumulator.get_completed_response = _patched_get # type: ignore[assignment]
except Exception:
pass
# 4) Guard langchain-openai against None output (final safety net)
try:
import langchain_openai.chat_models.base as _base
_orig_construct = _base._construct_lc_result_from_responses_api
def _safe(response: Any, *args: Any, **kwargs: Any) -> Any:
if response.output is None:
response.output = []
return _orig_construct(response, *args, **kwargs)
_base._construct_lc_result_from_responses_api = _safe
except Exception:
pass
_patch_ccproxy_codex_compat()
# ---------------------------------------------------------------------------
# Patch: langchain-openrouter v0.2.1 — _convert_message_to_dict() serializes
# reasoning_details back to the API, but streaming chunks use wrong field
@@ -217,3 +325,86 @@ def _patch_openai_compat_content(model: Any) -> None:
yield chunk
model._astream = _patched_astream
# ---------------------------------------------------------------------------
# Patch: ccproxy Codex Responses API rejects "system" role messages.
# Convert SystemMessage to use "developer" role via langchain-openai's
# __openai_role__ mechanism.
# ---------------------------------------------------------------------------
def _patch_ccproxy_system_to_developer(model: Any) -> None:
"""Convert SystemMessage role from 'system' to 'developer' for ccproxy.
ccproxy's Responses API endpoint rejects system role messages with
400 "System messages are not allowed". LangChain's ``langchain_openai``
checks ``additional_kwargs["__openai_role__"]`` and uses that value as
the message role when serializing to the API.
Args:
model: A LangChain chat model instance to patch in-place.
"""
import copy
import functools
from langchain_core.messages import BaseMessage, SystemMessage
def _system_to_developer(messages: list[BaseMessage]) -> list[BaseMessage]:
out: list[BaseMessage] = []
for msg in messages:
if isinstance(msg, SystemMessage):
if msg.additional_kwargs.get("__openai_role__") != "developer":
msg = copy.copy(msg)
msg.additional_kwargs = {
**msg.additional_kwargs,
"__openai_role__": "developer",
}
out.append(msg)
return out
orig_generate = getattr(model, "_generate", None)
if orig_generate is None:
return
@functools.wraps(orig_generate)
def _patched_generate(
messages: list[BaseMessage], *args: Any, **kwargs: Any
) -> Any:
return orig_generate(_system_to_developer(messages), *args, **kwargs)
model._generate = _patched_generate
orig_agenerate = getattr(model, "_agenerate", None)
if orig_agenerate is not None:
@functools.wraps(orig_agenerate)
async def _patched_agenerate(
messages: list[BaseMessage], *args: Any, **kwargs: Any
) -> Any:
return await orig_agenerate(_system_to_developer(messages), *args, **kwargs)
model._agenerate = _patched_agenerate
orig_stream = getattr(model, "_stream", None)
if orig_stream is not None:
@functools.wraps(orig_stream)
def _patched_stream(
messages: list[BaseMessage], *args: Any, **kwargs: Any
) -> Any:
return orig_stream(_system_to_developer(messages), *args, **kwargs)
model._stream = _patched_stream
orig_astream = getattr(model, "_astream", None)
if orig_astream is not None:
@functools.wraps(orig_astream)
async def _patched_astream(
messages: list[BaseMessage], *args: Any, **kwargs: Any
) -> Any:
async for chunk in orig_astream(
_system_to_developer(messages), *args, **kwargs
):
yield chunk
model._astream = _patched_astream