refactor(agent/G_small): hoist echo-family import, dataclass pending item, adopt_credits_state helper

This commit is contained in:
Teknium
2026-09-02 18:58:29 -07:00
parent d038249b12
commit ca5f9f342a
5 changed files with 25 additions and 46 deletions
+8 -7
View File
@@ -62,9 +62,7 @@ class RateLimitCreditsMixin:
except Exception:
fixture = None
if fixture is not None:
self._credits_state = fixture
if self._credits_session_start_micros is None:
self._credits_session_start_micros = fixture.remaining_micros
self._adopt_credits_state(fixture)
latch = getattr(self, "_credits_latch", None)
if isinstance(latch, dict):
# Only seen_below_90 — priming seen_grant_unspent would fire grant_spent on first observation.
@@ -96,10 +94,7 @@ class RateLimitCreditsMixin:
)
return
self._credits_state = state
# Latch session-start remaining the first time we ever see a header.
if self._credits_session_start_micros is None:
self._credits_session_start_micros = state.remaining_micros
self._adopt_credits_state(state)
if dev:
# HERMES_DEV_CREDITS streams each capture to agent.log (`hermes logs -f`, grep 'credits ▸').
spent = self.get_credits_spent_micros()
@@ -114,6 +109,12 @@ class RateLimitCreditsMixin:
)
self._emit_credits_notices()
def _adopt_credits_state(self, state) -> None:
"""Retain-last-known: overwrite state and latch session-start remaining on the first header ever seen."""
self._credits_state = state
if self._credits_session_start_micros is None:
self._credits_session_start_micros = state.remaining_micros
def _emit_credits_notices(self) -> None:
"""Run the threshold policy on the current credits state and emit notices.
+1 -6
View File
@@ -156,12 +156,7 @@ def format_rate_limit_display(state: RateLimitState) -> str:
return "No rate limit data yet — make an API request first."
age = state.age_seconds
if age < 5:
freshness = "just now"
elif age < 60:
freshness = f"{int(age)}s ago"
else:
freshness = f"{_fmt_seconds(age)} ago"
freshness = "just now" if age < 5 else f"{int(age)}s ago" if age < 60 else f"{_fmt_seconds(age)} ago"
provider_label = state.provider.title() if state.provider else "Provider"
labeled = [
+3 -10
View File
@@ -8,6 +8,7 @@ import time
from typing import Optional
from agent.lazy_forward import forward as _forward, forward_static as _forward_static
from agent.message_sanitization import matches_reasoning_echo_family
from utils import base_url_host_matches
# Static OpenRouter fallback when the live /v1/models capability cache is cold.
@@ -52,10 +53,7 @@ class ReasoningParamsMixin:
# Live-catalog metadata first (OpenRouter /v1/models supported_parameters) — the static prefix
# allowlist repeatedly went stale one vendor at a time. Unknown falls back to the static list.
try:
from hermes_cli.models import (
openrouter_model_reasoning_capabilities,
warm_openrouter_reasoning_caps_async,
)
from hermes_cli.models import openrouter_model_reasoning_capabilities, warm_openrouter_reasoning_caps_async
caps = openrouter_model_reasoning_capabilities(self.model)
if caps is None:
warm_openrouter_reasoning_caps_async() # cache cold — warm in the background, never block
@@ -102,9 +100,7 @@ class ReasoningParamsMixin:
from hermes_cli.models import ollama_model_supports_thinking
except Exception:
return False
return bool(self._cached_probe(
"_ollama_thinking_cache", ollama_model_supports_thinking, None, lambda v: v is not None,
))
return bool(self._cached_probe("_ollama_thinking_cache", ollama_model_supports_thinking, None, lambda v: v is not None))
def _resolve_lmstudio_summary_reasoning_effort(self) -> Optional[str]:
"""Safe top-level ``reasoning_effort`` for LM Studio; shared with the iteration-limit summary call."""
@@ -182,17 +178,14 @@ class ReasoningParamsMixin:
# provider and no model (its rule matches exact provider ids + hosts only).
def _needs_kimi_tool_reasoning(self) -> bool:
"""True when the current provider is Kimi / Moonshot thinking mode."""
from agent.message_sanitization import matches_reasoning_echo_family
return matches_reasoning_echo_family("kimi", self.provider, None, self.base_url)
def _needs_deepseek_tool_reasoning(self) -> bool:
"""True when the current provider is DeepSeek thinking mode (omitting the echo is an HTTP 400)."""
from agent.message_sanitization import matches_reasoning_echo_family
return matches_reasoning_echo_family("deepseek", (self.provider or "").lower(), self.model, self.base_url)
def _needs_mimo_tool_reasoning(self) -> bool:
"""True when the current provider is Xiaomi MiMo thinking mode."""
from agent.message_sanitization import matches_reasoning_echo_family
return matches_reasoning_echo_family("mimo", (self.provider or "").lower(), self.model, self.base_url)
_copy_reasoning_content_for_api = _forward("agent.agent_runtime_helpers", "copy_reasoning_content_for_api")
+10 -12
View File
@@ -24,6 +24,7 @@ import logging
import threading
import time
import urllib.request
from dataclasses import dataclass
from typing import Any, Callable, Dict, Optional
logger = logging.getLogger(__name__)
@@ -73,14 +74,12 @@ def review_targets_managed_local(agent: Any, task_cfg: Optional[Dict[str, Any]])
return False
@dataclass(slots=True)
class _PendingReview:
__slots__ = ("agent", "kwargs", "enqueued_at", "session_key")
def __init__(self, agent: Any, session_key: str, kwargs: Dict[str, Any], enqueued_at: float):
self.agent = agent
self.session_key = session_key
self.kwargs = kwargs
self.enqueued_at = enqueued_at
agent: Any
session_key: str
kwargs: Dict[str, Any]
enqueued_at: float
class ReviewIdleQueue:
@@ -155,14 +154,13 @@ class ReviewIdleQueue:
aged = [p for p in self._pending.values()
if now - p.enqueued_at >= defer_max_age_s(p.kwargs.get("task_cfg"))]
candidate = min(aged, key=lambda p: p.enqueued_at) if aged else None
if candidate is None:
if self._quiet_for() < _IDLE_SETTLE_S or not self._server_idle():
return None
with self._lock:
if candidate is None and (self._quiet_for() < _IDLE_SETTLE_S or not self._server_idle()):
return None
with self._lock:
if candidate is None:
if not self._pending:
return None
candidate = min(self._pending.values(), key=lambda p: p.enqueued_at)
with self._lock:
return self._pending.pop(candidate.session_key, None)
def _run(self) -> None:
+3 -11
View File
@@ -125,10 +125,7 @@ def _answer_via_fork(parent_agent: Any, question: str, history: Optional[List[Di
stays byte-identical for cache parity, but the side question can never mutate anything.
"""
from agent.background_review import (
_digest_history,
_record_review_usage_to_parent,
_snapshot_review_usage,
build_cache_parity_fork,
_digest_history, _record_review_usage_to_parent, _snapshot_review_usage, build_cache_parity_fork,
)
from hermes_cli.plugins import clear_thread_tool_whitelist, set_thread_tool_whitelist
@@ -173,15 +170,10 @@ def _answer_via_oneshot(question: str, history: Optional[List[Dict[str, Any]]],
from agent.oneshot import run_oneshot
user_input = (
"Conversation transcript (snapshot):\n"
"-----\n"
f"{render_history_for_side_question(history)}\n"
"-----\n\n"
f"Conversation transcript (snapshot):\n-----\n{render_history_for_side_question(history)}\n-----\n\n"
f"Side question: {question}"
)
return run_oneshot(
instructions=_ONESHOT_INSTRUCTIONS, user_input=user_input, task=SIDE_QUESTION_TASK, **run_kwargs
)
return run_oneshot(instructions=_ONESHOT_INSTRUCTIONS, user_input=user_input, task=SIDE_QUESTION_TASK, **run_kwargs)
def answer_side_question(