refactor(agent/G_small): hoist echo-family import, dataclass pending item, adopt_credits_state helper
This commit is contained in:
@@ -62,9 +62,7 @@ class RateLimitCreditsMixin:
|
||||
except Exception:
|
||||
fixture = None
|
||||
if fixture is not None:
|
||||
self._credits_state = fixture
|
||||
if self._credits_session_start_micros is None:
|
||||
self._credits_session_start_micros = fixture.remaining_micros
|
||||
self._adopt_credits_state(fixture)
|
||||
latch = getattr(self, "_credits_latch", None)
|
||||
if isinstance(latch, dict):
|
||||
# Only seen_below_90 — priming seen_grant_unspent would fire grant_spent on first observation.
|
||||
@@ -96,10 +94,7 @@ class RateLimitCreditsMixin:
|
||||
)
|
||||
return
|
||||
|
||||
self._credits_state = state
|
||||
# Latch session-start remaining the first time we ever see a header.
|
||||
if self._credits_session_start_micros is None:
|
||||
self._credits_session_start_micros = state.remaining_micros
|
||||
self._adopt_credits_state(state)
|
||||
if dev:
|
||||
# HERMES_DEV_CREDITS streams each capture to agent.log (`hermes logs -f`, grep 'credits ▸').
|
||||
spent = self.get_credits_spent_micros()
|
||||
@@ -114,6 +109,12 @@ class RateLimitCreditsMixin:
|
||||
)
|
||||
self._emit_credits_notices()
|
||||
|
||||
def _adopt_credits_state(self, state) -> None:
|
||||
"""Retain-last-known: overwrite state and latch session-start remaining on the first header ever seen."""
|
||||
self._credits_state = state
|
||||
if self._credits_session_start_micros is None:
|
||||
self._credits_session_start_micros = state.remaining_micros
|
||||
|
||||
def _emit_credits_notices(self) -> None:
|
||||
"""Run the threshold policy on the current credits state and emit notices.
|
||||
|
||||
|
||||
@@ -156,12 +156,7 @@ def format_rate_limit_display(state: RateLimitState) -> str:
|
||||
return "No rate limit data yet — make an API request first."
|
||||
|
||||
age = state.age_seconds
|
||||
if age < 5:
|
||||
freshness = "just now"
|
||||
elif age < 60:
|
||||
freshness = f"{int(age)}s ago"
|
||||
else:
|
||||
freshness = f"{_fmt_seconds(age)} ago"
|
||||
freshness = "just now" if age < 5 else f"{int(age)}s ago" if age < 60 else f"{_fmt_seconds(age)} ago"
|
||||
|
||||
provider_label = state.provider.title() if state.provider else "Provider"
|
||||
labeled = [
|
||||
|
||||
@@ -8,6 +8,7 @@ import time
|
||||
from typing import Optional
|
||||
|
||||
from agent.lazy_forward import forward as _forward, forward_static as _forward_static
|
||||
from agent.message_sanitization import matches_reasoning_echo_family
|
||||
from utils import base_url_host_matches
|
||||
|
||||
# Static OpenRouter fallback when the live /v1/models capability cache is cold.
|
||||
@@ -52,10 +53,7 @@ class ReasoningParamsMixin:
|
||||
# Live-catalog metadata first (OpenRouter /v1/models supported_parameters) — the static prefix
|
||||
# allowlist repeatedly went stale one vendor at a time. Unknown falls back to the static list.
|
||||
try:
|
||||
from hermes_cli.models import (
|
||||
openrouter_model_reasoning_capabilities,
|
||||
warm_openrouter_reasoning_caps_async,
|
||||
)
|
||||
from hermes_cli.models import openrouter_model_reasoning_capabilities, warm_openrouter_reasoning_caps_async
|
||||
caps = openrouter_model_reasoning_capabilities(self.model)
|
||||
if caps is None:
|
||||
warm_openrouter_reasoning_caps_async() # cache cold — warm in the background, never block
|
||||
@@ -102,9 +100,7 @@ class ReasoningParamsMixin:
|
||||
from hermes_cli.models import ollama_model_supports_thinking
|
||||
except Exception:
|
||||
return False
|
||||
return bool(self._cached_probe(
|
||||
"_ollama_thinking_cache", ollama_model_supports_thinking, None, lambda v: v is not None,
|
||||
))
|
||||
return bool(self._cached_probe("_ollama_thinking_cache", ollama_model_supports_thinking, None, lambda v: v is not None))
|
||||
|
||||
def _resolve_lmstudio_summary_reasoning_effort(self) -> Optional[str]:
|
||||
"""Safe top-level ``reasoning_effort`` for LM Studio; shared with the iteration-limit summary call."""
|
||||
@@ -182,17 +178,14 @@ class ReasoningParamsMixin:
|
||||
# provider and no model (its rule matches exact provider ids + hosts only).
|
||||
def _needs_kimi_tool_reasoning(self) -> bool:
|
||||
"""True when the current provider is Kimi / Moonshot thinking mode."""
|
||||
from agent.message_sanitization import matches_reasoning_echo_family
|
||||
return matches_reasoning_echo_family("kimi", self.provider, None, self.base_url)
|
||||
|
||||
def _needs_deepseek_tool_reasoning(self) -> bool:
|
||||
"""True when the current provider is DeepSeek thinking mode (omitting the echo is an HTTP 400)."""
|
||||
from agent.message_sanitization import matches_reasoning_echo_family
|
||||
return matches_reasoning_echo_family("deepseek", (self.provider or "").lower(), self.model, self.base_url)
|
||||
|
||||
def _needs_mimo_tool_reasoning(self) -> bool:
|
||||
"""True when the current provider is Xiaomi MiMo thinking mode."""
|
||||
from agent.message_sanitization import matches_reasoning_echo_family
|
||||
return matches_reasoning_echo_family("mimo", (self.provider or "").lower(), self.model, self.base_url)
|
||||
|
||||
_copy_reasoning_content_for_api = _forward("agent.agent_runtime_helpers", "copy_reasoning_content_for_api")
|
||||
|
||||
+10
-12
@@ -24,6 +24,7 @@ import logging
|
||||
import threading
|
||||
import time
|
||||
import urllib.request
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Callable, Dict, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -73,14 +74,12 @@ def review_targets_managed_local(agent: Any, task_cfg: Optional[Dict[str, Any]])
|
||||
return False
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class _PendingReview:
|
||||
__slots__ = ("agent", "kwargs", "enqueued_at", "session_key")
|
||||
|
||||
def __init__(self, agent: Any, session_key: str, kwargs: Dict[str, Any], enqueued_at: float):
|
||||
self.agent = agent
|
||||
self.session_key = session_key
|
||||
self.kwargs = kwargs
|
||||
self.enqueued_at = enqueued_at
|
||||
agent: Any
|
||||
session_key: str
|
||||
kwargs: Dict[str, Any]
|
||||
enqueued_at: float
|
||||
|
||||
|
||||
class ReviewIdleQueue:
|
||||
@@ -155,14 +154,13 @@ class ReviewIdleQueue:
|
||||
aged = [p for p in self._pending.values()
|
||||
if now - p.enqueued_at >= defer_max_age_s(p.kwargs.get("task_cfg"))]
|
||||
candidate = min(aged, key=lambda p: p.enqueued_at) if aged else None
|
||||
if candidate is None:
|
||||
if self._quiet_for() < _IDLE_SETTLE_S or not self._server_idle():
|
||||
return None
|
||||
with self._lock:
|
||||
if candidate is None and (self._quiet_for() < _IDLE_SETTLE_S or not self._server_idle()):
|
||||
return None
|
||||
with self._lock:
|
||||
if candidate is None:
|
||||
if not self._pending:
|
||||
return None
|
||||
candidate = min(self._pending.values(), key=lambda p: p.enqueued_at)
|
||||
with self._lock:
|
||||
return self._pending.pop(candidate.session_key, None)
|
||||
|
||||
def _run(self) -> None:
|
||||
|
||||
+3
-11
@@ -125,10 +125,7 @@ def _answer_via_fork(parent_agent: Any, question: str, history: Optional[List[Di
|
||||
stays byte-identical for cache parity, but the side question can never mutate anything.
|
||||
"""
|
||||
from agent.background_review import (
|
||||
_digest_history,
|
||||
_record_review_usage_to_parent,
|
||||
_snapshot_review_usage,
|
||||
build_cache_parity_fork,
|
||||
_digest_history, _record_review_usage_to_parent, _snapshot_review_usage, build_cache_parity_fork,
|
||||
)
|
||||
from hermes_cli.plugins import clear_thread_tool_whitelist, set_thread_tool_whitelist
|
||||
|
||||
@@ -173,15 +170,10 @@ def _answer_via_oneshot(question: str, history: Optional[List[Dict[str, Any]]],
|
||||
from agent.oneshot import run_oneshot
|
||||
|
||||
user_input = (
|
||||
"Conversation transcript (snapshot):\n"
|
||||
"-----\n"
|
||||
f"{render_history_for_side_question(history)}\n"
|
||||
"-----\n\n"
|
||||
f"Conversation transcript (snapshot):\n-----\n{render_history_for_side_question(history)}\n-----\n\n"
|
||||
f"Side question: {question}"
|
||||
)
|
||||
return run_oneshot(
|
||||
instructions=_ONESHOT_INSTRUCTIONS, user_input=user_input, task=SIDE_QUESTION_TASK, **run_kwargs
|
||||
)
|
||||
return run_oneshot(instructions=_ONESHOT_INSTRUCTIONS, user_input=user_input, task=SIDE_QUESTION_TASK, **run_kwargs)
|
||||
|
||||
|
||||
def answer_side_question(
|
||||
|
||||
Reference in New Issue
Block a user