diff --git a/agent/turn_api_call.py b/agent/turn_api_call.py index 56557b17df..b6a2e05f19 100644 --- a/agent/turn_api_call.py +++ b/agent/turn_api_call.py @@ -243,8 +243,7 @@ def nous_rate_limit_guard( return _verdict("return", { "final_response": ( f"⏳ {_nous_msg}\n\n" - "No fallback provider available. " - "Try again after the reset, or add a " + "No fallback provider available. Try again after the reset, or add a " "fallback provider in config.yaml." ), "messages": messages, diff --git a/agent/turn_empty_response.py b/agent/turn_empty_response.py index 7f4543f739..a840cc2f68 100644 --- a/agent/turn_empty_response.py +++ b/agent/turn_empty_response.py @@ -64,8 +64,7 @@ def _retry_empty( n = agent._empty_content_retries wait_time = jittered_backoff(n, base_delay=5.0, max_delay=60.0) logger.warning( - "Empty response (no content or reasoning) — " - "retry %d/%d in %.1fs (model=%s)", + "Empty response (no content or reasoning) — retry %d/%d in %.1fs (model=%s)", n, budget, wait_time, agent.model, ) _budget_note = ( @@ -115,8 +114,7 @@ def _terminal_empty(agent: Any, assistant_message: Any, finish_reason: str, mess if not reasoning_text: logger.warning( - "Empty response (no content or reasoning) " - "after %d retries. No fallback available. " + "Empty response (no content or reasoning) after %d retries. No fallback available. " "model=%s provider=%s", agent._empty_content_retries, agent.model, agent.provider, @@ -130,20 +128,15 @@ def _terminal_empty(agent: Any, assistant_message: Any, finish_reason: str, mess reasoning_preview = reasoning_text[:500] + "..." if len(reasoning_text) > 500 else reasoning_text logger.warning( - "Reasoning-only response (no visible content) " - "after exhausting retries and fallback. " - "Reasoning: %s", reasoning_preview, + "Reasoning-only response (no visible content) after exhausting retries and fallback. Reasoning: %s", reasoning_preview, ) agent._emit_status( - "⚠️ Model produced reasoning but no visible " - "response after all retries. Returning empty." + "⚠️ Model produced reasoning but no visible response after all retries. Returning empty." ) return ( - "⚠️ The model produced only internal reasoning and " - "no final answer, despite retries" + "⚠️ The model produced only internal reasoning and no final answer, despite retries" + (" and fallback" if agent._fallback_chain else "") - + ". Its last reasoning, which may contain the " - "answer:\n\n" + reasoning_preview + + ". Its last reasoning, which may contain the answer:\n\n" + reasoning_preview ) @@ -175,8 +168,7 @@ def recover_empty_response( _turn_exit_reason = "partial_stream_recovery" _recovered = agent._strip_think_blocks(_partial_streamed).strip() logger.info( - "Partial stream content delivered (%d chars) " - "— using as final response", + "Partial stream content delivered (%d chars) — using as final response", len(_recovered), ) agent._emit_status("↻ Stream interrupted — using delivered content " "as final response") @@ -239,8 +231,7 @@ def recover_empty_response( if _has_structured and agent._thinking_prefill_retries < 2: agent._thinking_prefill_retries += 1 logger.info( - "Thinking-only response (no visible content) — " - "prefilling to continue (%d/2)", + "Thinking-only response (no visible content) — prefilling to continue (%d/2)", agent._thinking_prefill_retries, ) agent._buffer_status( @@ -266,23 +257,19 @@ def recover_empty_response( if _truly_empty and _deterministic_empty: logger.warning( - "Deterministic empty response detected " - "(consecutive zero-output completions, " - "model=%s provider=%s finish_reason=%s) — " - "skipping remaining retries", + "Deterministic empty response detected (consecutive zero-output completions, " + "model=%s provider=%s finish_reason=%s) — skipping remaining retries", agent.model, agent.provider, finish_reason, ) agent._buffer_status( - "⚠️ Model is deterministically returning empty " - "(zero output tokens) — skipping further retries " + "⚠️ Model is deterministically returning empty (zero output tokens) — skipping further retries " "to avoid repeat charges" ) # Exhausted retries — try the next provider in the chain before "(empty)". if _truly_empty and agent._fallback_chain: logger.warning( - "Empty response after %d retries — " - "attempting fallback (model=%s, provider=%s)", + "Empty response after %d retries — attempting fallback (model=%s, provider=%s)", agent._empty_content_retries, agent.model, agent.provider, ) @@ -292,8 +279,7 @@ def recover_empty_response( agent._empty_content_retries = 0 agent._buffer_status(f"↻ Switched to fallback: {agent.model} " f"({agent.provider})") logger.info( - "Fallback activated after empty responses: " - "now using %s on %s", + "Fallback activated after empty responses: now using %s on %s", agent.model, agent.provider, ) # OUTER loop: `continue` re-runs preflight against the fallback's window; diff --git a/agent/turn_overflow.py b/agent/turn_overflow.py index 09b086adfb..3ee123647a 100644 --- a/agent/turn_overflow.py +++ b/agent/turn_overflow.py @@ -367,12 +367,9 @@ def _recover_context_length(st: _Recovery, _retry: TurnRetryState, error_msg: st "max_tokens exceeds the provider's output cap for this model. " "Lower model.max_tokens in config.yaml.", notices=( - "❌ The provider rejected the request because " - "max_tokens exceeds its output cap for this model.", - " 💡 Lower model.max_tokens in your config.yaml to " - "at or below the model's max-output limit. " - "(This is an output-cap error, not a context overflow — " - "compression cannot fix it.)", + "❌ The provider rejected the request because max_tokens exceeds its output cap for this model.", + " 💡 Lower model.max_tokens in your config.yaml to at or below the model's max-output limit. " + "(This is an output-cap error, not a context overflow — compression cannot fix it.)", ), log=( f"{agent.log_prefix}Output-cap error not routed into compression " diff --git a/agent/turn_truncation.py b/agent/turn_truncation.py index a2ad233a80..b9eadaaa80 100644 --- a/agent/turn_truncation.py +++ b/agent/turn_truncation.py @@ -31,40 +31,26 @@ _FIRST_TRUNCATED_FINAL = "First response truncated due to output length limit" _THINKING_EXHAUSTED = ( "💭 Reasoning exhausted the output token budget — no visible response was produced.", - "⚠️ **Thinking Budget Exhausted**\n\n" - "The model used all its output tokens on reasoning " - "and had none left for the actual response.\n\n" - "To fix this:\n" + "⚠️ **Thinking Budget Exhausted**\n\nThe model used all its output tokens on reasoning " + "and had none left for the actual response.\n\nTo fix this:\n" "→ Lower reasoning effort: `/reasoning low` or `/reasoning minimal`\n" "→ Or switch to a larger/non-reasoning model with `/model`", "Model used all output tokens on reasoning with none left " - "for the response. Try lowering reasoning effort or " - "increasing max_tokens.", + "for the response. Try lowering reasoning effort or increasing max_tokens.", ) _REPETITION_DOMINATED = ( - "🔁 Response dominated by repeated text — stopping instead of " - "continuing a degenerate response.", - "⚠️ **Response Stopped — Repetition Detected**\n\n" - "The model fell into a repetition loop while " - "writing this response, so continuing would only " - "produce more repeated text. The partial response " - "was discarded.\n\n" - "→ Switch to a different model with `/model`\n" - "→ Or resend your message (your conversation " - "history is preserved)", - "Model output entered a repetition loop and was " - "truncated mid-loop; refusing to continue a " + "🔁 Response dominated by repeated text — stopping instead of continuing a degenerate response.", + "⚠️ **Response Stopped — Repetition Detected**\n\nThe model fell into a repetition loop while " + "writing this response, so continuing would only produce more repeated text. The partial response " + "was discarded.\n\n→ Switch to a different model with `/model`\n" + "→ Or resend your message (your conversation history is preserved)", + "Model output entered a repetition loop and was truncated mid-loop; refusing to continue a " "degenerate response.", ) _CEILING_NO_TEXT = ( - "⚠️ **No visible answer was produced.** The " - "model hit its output-token limit on every " - "continuation attempt — its reasoning " - "consumed the entire budget each time.\n\n" - "To fix this:\n" - "→ Lower reasoning effort: `/reasoning low` " - "or `/reasoning none`\n" - "→ Or raise max_tokens for this model" + "⚠️ **No visible answer was produced.** The model hit its output-token limit on every " + "continuation attempt — its reasoning consumed the entire budget each time.\n\nTo fix this:\n" + "→ Lower reasoning effort: `/reasoning low` or `/reasoning none`\n→ Or raise max_tokens for this model" ) @@ -526,8 +512,7 @@ def handle_content_policy_refusal( agent._flush_status_buffer() _refusal_log = _refusal_text[:500] + "..." if len(_refusal_text) > 500 else _refusal_text logger.warning( - "%sModel declined to respond (finish_reason=content_filter). " - "model=%s provider=%s refusal=%s", + "%sModel declined to respond (finish_reason=content_filter). model=%s provider=%s refusal=%s", agent.log_prefix, agent.model, agent.provider, _refusal_log or "(no text)", ) @@ -536,8 +521,7 @@ def handle_content_policy_refusal( f"Model's explanation: {_refusal_text}" if _refusal_text else "The model returned no explanation." ) _refusal_response = ( - "⚠️ The model declined to respond to this request " - "(safety refusal — not a Hermes/gateway failure).\n\n" + "⚠️ The model declined to respond to this request (safety refusal — not a Hermes/gateway failure).\n\n" f"{_refusal_detail}\n\n" f"{_CONTENT_POLICY_RECOVERY_HINT}" )