fix(agent): propagate prompt-cache TTL to MoA/aux, clamp Qwen 1h, re-preflight on failover (#84733)

This commit is contained in:
webtecnica
2026-08-12 16:51:43 -03:00
committed by kshitij
parent 7060ac7bed
commit 9a5cf83541
8 changed files with 421 additions and 20 deletions
+36
View File
@@ -120,6 +120,42 @@ def _build_marker(ttl: str) -> Dict[str, str]:
return marker
# Routes whose context cache documents a five-minute window (renewed on
# hit) and rejects the Anthropic 1h tier. Kept in parity with the
# alibaba-family set in agent_runtime_helpers.anthropic_prompt_cache_policy.
_QWEN_1H_UNSUPPORTED_PROVIDERS = frozenset({
"opencode",
"opencode-zen",
"opencode-go",
"alibaba",
})
def effective_cache_ttl(
ttl: str | None,
*,
model: str = "",
provider: str = "",
) -> str:
"""Clamp a requested cache TTL to what the destination route supports.
Qwen/Alibaba context caching documents an explicit five-minute window
(renewed on hit); the Anthropic ``1h`` tier is ignored/rejected there,
so a configured ``1h`` regresses to ``5m`` instead of shipping a marker
the provider drops and creating a false 1h-cache expectation (#84733).
All other caching routes keep the requested TTL.
``None`` (caching active with no explicit tier) resolves to ``5m``.
"""
if ttl != "1h":
return ttl or "5m"
if "qwen" in (model or "").lower():
return "5m"
if (provider or "").lower() in _QWEN_1H_UNSUPPORTED_PROVIDERS:
return "5m"
return "1h"
def _apply_system_cache_markers(
message: dict,
cache_marker: dict,