diff --git a/agent/rate_limit_credits.py b/agent/rate_limit_credits.py index e98da3695f..cd438a6293 100644 --- a/agent/rate_limit_credits.py +++ b/agent/rate_limit_credits.py @@ -62,9 +62,7 @@ class RateLimitCreditsMixin: except Exception: fixture = None if fixture is not None: - self._credits_state = fixture - if self._credits_session_start_micros is None: - self._credits_session_start_micros = fixture.remaining_micros + self._adopt_credits_state(fixture) latch = getattr(self, "_credits_latch", None) if isinstance(latch, dict): # Only seen_below_90 — priming seen_grant_unspent would fire grant_spent on first observation. @@ -96,10 +94,7 @@ class RateLimitCreditsMixin: ) return - self._credits_state = state - # Latch session-start remaining the first time we ever see a header. - if self._credits_session_start_micros is None: - self._credits_session_start_micros = state.remaining_micros + self._adopt_credits_state(state) if dev: # HERMES_DEV_CREDITS streams each capture to agent.log (`hermes logs -f`, grep 'credits ▸'). spent = self.get_credits_spent_micros() @@ -114,6 +109,12 @@ class RateLimitCreditsMixin: ) self._emit_credits_notices() + def _adopt_credits_state(self, state) -> None: + """Retain-last-known: overwrite state and latch session-start remaining on the first header ever seen.""" + self._credits_state = state + if self._credits_session_start_micros is None: + self._credits_session_start_micros = state.remaining_micros + def _emit_credits_notices(self) -> None: """Run the threshold policy on the current credits state and emit notices. diff --git a/agent/rate_limit_tracker.py b/agent/rate_limit_tracker.py index 7eccdcfee9..0633c11424 100644 --- a/agent/rate_limit_tracker.py +++ b/agent/rate_limit_tracker.py @@ -156,12 +156,7 @@ def format_rate_limit_display(state: RateLimitState) -> str: return "No rate limit data yet — make an API request first." age = state.age_seconds - if age < 5: - freshness = "just now" - elif age < 60: - freshness = f"{int(age)}s ago" - else: - freshness = f"{_fmt_seconds(age)} ago" + freshness = "just now" if age < 5 else f"{int(age)}s ago" if age < 60 else f"{_fmt_seconds(age)} ago" provider_label = state.provider.title() if state.provider else "Provider" labeled = [ diff --git a/agent/reasoning_params.py b/agent/reasoning_params.py index fafdab7c04..b06e567f2b 100644 --- a/agent/reasoning_params.py +++ b/agent/reasoning_params.py @@ -8,6 +8,7 @@ import time from typing import Optional from agent.lazy_forward import forward as _forward, forward_static as _forward_static +from agent.message_sanitization import matches_reasoning_echo_family from utils import base_url_host_matches # Static OpenRouter fallback when the live /v1/models capability cache is cold. @@ -52,10 +53,7 @@ class ReasoningParamsMixin: # Live-catalog metadata first (OpenRouter /v1/models supported_parameters) — the static prefix # allowlist repeatedly went stale one vendor at a time. Unknown falls back to the static list. try: - from hermes_cli.models import ( - openrouter_model_reasoning_capabilities, - warm_openrouter_reasoning_caps_async, - ) + from hermes_cli.models import openrouter_model_reasoning_capabilities, warm_openrouter_reasoning_caps_async caps = openrouter_model_reasoning_capabilities(self.model) if caps is None: warm_openrouter_reasoning_caps_async() # cache cold — warm in the background, never block @@ -102,9 +100,7 @@ class ReasoningParamsMixin: from hermes_cli.models import ollama_model_supports_thinking except Exception: return False - return bool(self._cached_probe( - "_ollama_thinking_cache", ollama_model_supports_thinking, None, lambda v: v is not None, - )) + return bool(self._cached_probe("_ollama_thinking_cache", ollama_model_supports_thinking, None, lambda v: v is not None)) def _resolve_lmstudio_summary_reasoning_effort(self) -> Optional[str]: """Safe top-level ``reasoning_effort`` for LM Studio; shared with the iteration-limit summary call.""" @@ -182,17 +178,14 @@ class ReasoningParamsMixin: # provider and no model (its rule matches exact provider ids + hosts only). def _needs_kimi_tool_reasoning(self) -> bool: """True when the current provider is Kimi / Moonshot thinking mode.""" - from agent.message_sanitization import matches_reasoning_echo_family return matches_reasoning_echo_family("kimi", self.provider, None, self.base_url) def _needs_deepseek_tool_reasoning(self) -> bool: """True when the current provider is DeepSeek thinking mode (omitting the echo is an HTTP 400).""" - from agent.message_sanitization import matches_reasoning_echo_family return matches_reasoning_echo_family("deepseek", (self.provider or "").lower(), self.model, self.base_url) def _needs_mimo_tool_reasoning(self) -> bool: """True when the current provider is Xiaomi MiMo thinking mode.""" - from agent.message_sanitization import matches_reasoning_echo_family return matches_reasoning_echo_family("mimo", (self.provider or "").lower(), self.model, self.base_url) _copy_reasoning_content_for_api = _forward("agent.agent_runtime_helpers", "copy_reasoning_content_for_api") diff --git a/agent/review_idle_queue.py b/agent/review_idle_queue.py index 404992c534..6b06b70970 100644 --- a/agent/review_idle_queue.py +++ b/agent/review_idle_queue.py @@ -24,6 +24,7 @@ import logging import threading import time import urllib.request +from dataclasses import dataclass from typing import Any, Callable, Dict, Optional logger = logging.getLogger(__name__) @@ -73,14 +74,12 @@ def review_targets_managed_local(agent: Any, task_cfg: Optional[Dict[str, Any]]) return False +@dataclass(slots=True) class _PendingReview: - __slots__ = ("agent", "kwargs", "enqueued_at", "session_key") - - def __init__(self, agent: Any, session_key: str, kwargs: Dict[str, Any], enqueued_at: float): - self.agent = agent - self.session_key = session_key - self.kwargs = kwargs - self.enqueued_at = enqueued_at + agent: Any + session_key: str + kwargs: Dict[str, Any] + enqueued_at: float class ReviewIdleQueue: @@ -155,14 +154,13 @@ class ReviewIdleQueue: aged = [p for p in self._pending.values() if now - p.enqueued_at >= defer_max_age_s(p.kwargs.get("task_cfg"))] candidate = min(aged, key=lambda p: p.enqueued_at) if aged else None - if candidate is None: - if self._quiet_for() < _IDLE_SETTLE_S or not self._server_idle(): - return None - with self._lock: + if candidate is None and (self._quiet_for() < _IDLE_SETTLE_S or not self._server_idle()): + return None + with self._lock: + if candidate is None: if not self._pending: return None candidate = min(self._pending.values(), key=lambda p: p.enqueued_at) - with self._lock: return self._pending.pop(candidate.session_key, None) def _run(self) -> None: diff --git a/agent/side_question.py b/agent/side_question.py index 4f4ade5c7b..5a6c890fbe 100644 --- a/agent/side_question.py +++ b/agent/side_question.py @@ -125,10 +125,7 @@ def _answer_via_fork(parent_agent: Any, question: str, history: Optional[List[Di stays byte-identical for cache parity, but the side question can never mutate anything. """ from agent.background_review import ( - _digest_history, - _record_review_usage_to_parent, - _snapshot_review_usage, - build_cache_parity_fork, + _digest_history, _record_review_usage_to_parent, _snapshot_review_usage, build_cache_parity_fork, ) from hermes_cli.plugins import clear_thread_tool_whitelist, set_thread_tool_whitelist @@ -173,15 +170,10 @@ def _answer_via_oneshot(question: str, history: Optional[List[Dict[str, Any]]], from agent.oneshot import run_oneshot user_input = ( - "Conversation transcript (snapshot):\n" - "-----\n" - f"{render_history_for_side_question(history)}\n" - "-----\n\n" + f"Conversation transcript (snapshot):\n-----\n{render_history_for_side_question(history)}\n-----\n\n" f"Side question: {question}" ) - return run_oneshot( - instructions=_ONESHOT_INSTRUCTIONS, user_input=user_input, task=SIDE_QUESTION_TASK, **run_kwargs - ) + return run_oneshot(instructions=_ONESHOT_INSTRUCTIONS, user_input=user_input, task=SIDE_QUESTION_TASK, **run_kwargs) def answer_side_question(