refactor(agent/G_small): single-line signatures/calls, github effort fallback table, any() slot scan

This commit is contained in:
Teknium
2026-09-02 19:08:38 -07:00
parent 2ccf295128
commit 270128294e
12 changed files with 46 additions and 139 deletions
+9 -18
View File
@@ -70,9 +70,8 @@ class RateLimitCreditsMixin:
logger.info(
"credits ▸ [FIXTURE] remaining=%d (%s) · paid=%s · denom=%s · used=%s "
"(real headers bypassed — `echo clear` / unset HERMES_DEV_CREDITS_FIXTURE to restore)",
fixture.remaining_micros, fixture.remaining_usd or "?", fixture.paid_access,
fixture.denominator_kind, _pct(fixture.used_fraction),
)
fixture.remaining_micros, fixture.remaining_usd or "?", fixture.paid_access, fixture.denominator_kind,
_pct(fixture.used_fraction))
self._emit_credits_notices()
return
headers = _response_headers(http_response)
@@ -88,10 +87,8 @@ class RateLimitCreditsMixin:
return
if state is None:
if dev:
logger.info(
"credits ▸ response had no valid x-nous-credits-* headers "
"(miss — producer off / non-Nous path / >TTL stale)"
)
logger.info("credits ▸ response had no valid x-nous-credits-* headers "
"(miss — producer off / non-Nous path / >TTL stale)")
return
self._adopt_credits_state(state)
@@ -99,14 +96,11 @@ class RateLimitCreditsMixin:
# HERMES_DEV_CREDITS streams each capture to agent.log (`hermes logs -f`, grep 'credits ▸').
spent = self.get_credits_spent_micros()
logger.info(
"credits ▸ remaining=%d (%s) · paid=%s · denom=%s · used=%s "
"· Δspent=%s · age=%s%s",
state.remaining_micros, state.remaining_usd or "?", state.paid_access,
state.denominator_kind, _pct(state.used_fraction),
("%.1f¢" % (spent / 10000)) if spent is not None else "n/a",
"credits ▸ remaining=%d (%s) · paid=%s · denom=%s · used=%s · Δspent=%s · age=%s%s",
state.remaining_micros, state.remaining_usd or "?", state.paid_access, state.denominator_kind,
_pct(state.used_fraction), ("%.1f¢" % (spent / 10000)) if spent is not None else "n/a",
("%.0fs" % state.age_seconds) if state.age_seconds != float("inf") else "n/a",
(" · disabled=%s" % state.disabled_reason) if state.disabled_reason else "",
)
(" · disabled=%s" % state.disabled_reason) if state.disabled_reason else "")
self._emit_credits_notices()
def _adopt_credits_state(self, state) -> None:
@@ -133,10 +127,7 @@ class RateLimitCreditsMixin:
if latch is None:
latch = self._credits_latch = new_credits_latch()
# Free-model gate: a depleted account can still inference on a free model. Local data only.
model_is_free = is_free_tier_model(
getattr(self, "model", "") or "",
getattr(self, "base_url", "") or "",
)
model_is_free = is_free_tier_model(getattr(self, "model", "") or "", getattr(self, "base_url", "") or "")
to_show, to_clear = evaluate_credits_notices(state, latch, model_is_free=model_is_free)
for key in to_clear:
self._emit_notice_clear(key)
+3 -10
View File
@@ -86,10 +86,7 @@ def has_rate_limit_headers(lowered: Mapping[str, str]) -> bool:
return any(k.startswith("x-ratelimit-") for k in lowered)
def parse_rate_limit_headers(
headers: Mapping[str, str],
provider: str = "",
) -> Optional[RateLimitState]:
def parse_rate_limit_headers(headers: Mapping[str, str], provider: str = "") -> Optional[RateLimitState]:
"""Parse x-ratelimit-* headers into a RateLimitState (None if none present)."""
lowered = lower_headers(headers)
if not has_rate_limit_headers(lowered):
@@ -159,12 +156,8 @@ def format_rate_limit_display(state: RateLimitState) -> str:
freshness = "just now" if age < 5 else f"{int(age)}s ago" if age < 60 else f"{_fmt_seconds(age)} ago"
provider_label = state.provider.title() if state.provider else "Provider"
labeled = [
("Requests/min", state.requests_min),
("Requests/hr", state.requests_hour),
("Tokens/min", state.tokens_min),
("Tokens/hr", state.tokens_hour),
]
labeled = [("Requests/min", state.requests_min), ("Requests/hr", state.requests_hour),
("Tokens/min", state.tokens_min), ("Tokens/hr", state.tokens_hour)]
lines = [f"{provider_label} Rate Limits (captured {freshness}):", ""]
lines += [_bucket_line(label, bucket) for label, bucket in labeled[:2]]
lines += [""] + [_bucket_line(label, bucket) for label, bucket in labeled[2:]]
+1 -3
View File
@@ -86,9 +86,7 @@ def kimi_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
def clamp_effort(
effort: Optional[str],
supported: Optional[Sequence[str]],
overrides: Optional[dict[str, str]] = None,
effort: Optional[str], supported: Optional[Sequence[str]], overrides: Optional[dict[str, str]] = None,
) -> Optional[str]:
"""Clamp a requested reasoning effort onto a wire's supported levels.
+9 -22
View File
@@ -123,15 +123,10 @@ class ReasoningParamsMixin:
return None
effort = str(cfg.get("effort", "medium")).strip().lower()
if effort == "xhigh" and "xhigh" not in supported and "high" in supported:
effort = "high"
elif effort not in supported:
if effort == "minimal" and "low" in supported:
effort = "low"
elif "medium" in supported:
effort = "medium"
else:
effort = supported[0]
if effort not in supported:
# Nearest-neighbour fallbacks: xhigh→high, minimal→low, else medium, else the first published level.
nearest = {"xhigh": "high", "minimal": "low"}.get(effort)
effort = nearest if nearest in supported else "medium" if "medium" in supported else supported[0]
return {"effort": effort}
_build_assistant_message = _forward("agent.chat_completion_helpers", "build_assistant_message")
@@ -147,12 +142,8 @@ class ReasoningParamsMixin:
cached = getattr(self, "_thinking_pad_cache", None)
if cached is not None and cached[0] == key:
return cached[1]
result = (
self._needs_deepseek_tool_reasoning()
or self._needs_kimi_tool_reasoning()
or self._needs_mimo_tool_reasoning()
or self._reasoning_echo_opt_in()
)
result = (self._needs_deepseek_tool_reasoning() or self._needs_kimi_tool_reasoning()
or self._needs_mimo_tool_reasoning() or self._reasoning_echo_opt_in())
self._thinking_pad_cache = (key, result)
return result
@@ -204,13 +195,9 @@ class ReasoningParamsMixin:
if not isinstance(tool_calls, list):
return api_msg
from agent.transports.chat_completions import _model_consumes_thought_signature
strip = {"call_id", "response_item_id"}
if not _model_consumes_thought_signature(model):
strip.add("extra_content")
api_msg["tool_calls"] = [
{k: v for k, v in tc.items() if k not in strip} if isinstance(tc, dict) else tc
for tc in tool_calls
]
strip = {"call_id", "response_item_id"} | (set() if _model_consumes_thought_signature(model) else {"extra_content"})
api_msg["tool_calls"] = [{k: v for k, v in tc.items() if k not in strip} if isinstance(tc, dict) else tc
for tc in tool_calls]
return api_msg
_sanitize_tool_call_arguments = _forward_static("agent.agent_runtime_helpers", "sanitize_tool_call_arguments")
+2 -8
View File
@@ -66,8 +66,7 @@ def parse_retry_after_seconds(value_or_headers: Any) -> Optional[float]:
return max(0.0, (when - datetime.now(timezone.utc)).total_seconds())
def jittered_backoff(attempt: int, *, base_delay: float = 5.0, max_delay: float = 120.0,
jitter_ratio: float = 0.5) -> float:
def jittered_backoff(attempt: int, *, base_delay: float = 5.0, max_delay: float = 120.0, jitter_ratio: float = 0.5) -> float:
"""min(base * 2^(attempt-1), max_delay) + uniform jitter in
[0, jitter_ratio * delay]. ``attempt`` is 1-based."""
global _jitter_counter
@@ -102,12 +101,7 @@ def is_zai_coding_overload_error(*, base_url: str | None, model: str | None, err
def adaptive_rate_limit_backoff(
attempt: int,
*,
base_url: str | None,
model: str | None,
error: Any,
default_wait: float,
attempt: int, *, base_url: str | None, model: str | None, error: Any, default_wait: float,
short_attempts: int = _ZAI_CODING_OVERLOAD_SHORT_ATTEMPTS,
) -> tuple[float, str | None]:
"""Provider-aware rate-limit backoff → ``(wait_seconds, reason_label)``.
+7 -32
View File
@@ -41,20 +41,13 @@ def _message_text(message: Dict[str, Any]) -> str:
if isinstance(content, str):
return content
if isinstance(content, list):
parts = [
str(part.get("text") or "") if part.get("type") == "text"
else f"[{part.get('type', 'attachment')}]"
for part in content
if isinstance(part, dict)
]
parts = [str(part.get("text") or "") if part.get("type") == "text" else f"[{part.get('type', 'attachment')}]"
for part in content if isinstance(part, dict)]
return "\n".join(p for p in parts if p)
return ""
def snapshot_recent_messages(
messages: List[Dict[str, Any]],
limit: int = DEFAULT_CONTEXT_MESSAGES,
) -> List[Dict[str, str]]:
def snapshot_recent_messages(messages: List[Dict[str, Any]], limit: int = DEFAULT_CONTEXT_MESSAGES) -> List[Dict[str, str]]:
"""Last ``limit`` user/assistant messages as {role, text} dicts, oldest first.
System messages, tool results and empty-text messages (pure tool-call stubs) are excluded.
@@ -76,11 +69,7 @@ def snapshot_recent_messages(
return out
def collect_parent_loaded_skills(
parent_agent,
messages: List[Dict[str, Any]],
limit: int = 8,
) -> List[str]:
def collect_parent_loaded_skills(parent_agent, messages: List[Dict[str, Any]], limit: int = 8) -> List[str]:
"""Names of skills the parent agent was operating under.
Launch-preloaded skills come from the stable marker in the parent's
@@ -113,11 +102,7 @@ def collect_parent_loaded_skills(
return names[:limit]
def build_review_task(
snapshot: List[Dict[str, str]],
user_prompt: str = "",
loaded_skills: Optional[List[str]] = None,
) -> tuple:
def build_review_task(snapshot: List[Dict[str, str]], user_prompt: str = "", loaded_skills: Optional[List[str]] = None) -> tuple:
"""Compose the reviewer subagent's (goal, context) pair."""
lines = [
"You were spawned by the /review command. The following is an excerpt of the most recent conversation "
@@ -174,11 +159,7 @@ def _load_review_credentials_cfg() -> Optional[Dict[str, Any]]:
return cfg
def start_review(
parent_agent,
messages: List[Dict[str, Any]],
user_prompt: str = "",
) -> Dict[str, Any]:
def start_review(parent_agent, messages: List[Dict[str, Any]], user_prompt: str = "") -> Dict[str, Any]:
"""Dispatch the reviewer subagent in the background.
Returns the parsed ``delegate_task`` dispatch dict (``status: "dispatched"`` with a
@@ -199,13 +180,7 @@ def start_review(
from tools.delegate_tool import delegate_task
raw = delegate_task(
goal=goal,
context=context,
background=True,
parent_agent=parent_agent,
credentials_cfg=credentials_cfg,
)
raw = delegate_task(goal=goal, context=context, background=True, parent_agent=parent_agent, credentials_cfg=credentials_cfg)
try:
result = json.loads(raw)
except Exception:
+3 -5
View File
@@ -220,11 +220,9 @@ def _managed_server_idle() -> bool:
loaded = [m["id"] for m in _get("/models").get("data", [])
if (m.get("status") or {}).get("value") in ("loaded", "ready")]
for mid in loaded:
if any(s.get("is_processing") for s in _get(f"/slots?model={quote(mid)}")
if isinstance(s, dict)):
return False
return True
return not any(
s.get("is_processing") for mid in loaded for s in _get(f"/slots?model={quote(mid)}") if isinstance(s, dict)
)
except Exception: # noqa: BLE001
return True
+5 -11
View File
@@ -86,12 +86,10 @@ def render_history_for_side_question(
break
kept.append(line)
used += cost
kept.reverse()
if not kept:
return "(no prior conversation)"
prefix = "[...older conversation omitted...]\n" if len(kept) < len(lines) else ""
return prefix + "\n".join(kept)
return prefix + "\n".join(reversed(kept))
def _side_question_task_config() -> Dict[str, Any]:
@@ -118,10 +116,8 @@ def _answer_via_fork(parent_agent: Any, question: str, history: Optional[List[Di
)
from hermes_cli.plugins import clear_thread_tool_whitelist, set_thread_tool_whitelist
fork, _rt, routed = build_cache_parity_fork(
parent_agent, _side_question_task_config(),
max_iterations=_FORK_MAX_ITERATIONS, write_origin="side_question",
)
fork, _rt, routed = build_cache_parity_fork(parent_agent, _side_question_task_config(),
max_iterations=_FORK_MAX_ITERATIONS, write_origin="side_question")
try:
set_thread_tool_whitelist(
set(),
@@ -189,7 +185,5 @@ def answer_side_question(
except Exception:
logger.warning("/btw cache-parity fork failed; falling back to one-shot", exc_info=True)
return _answer_via_oneshot(
question, history,
main_runtime=main_runtime, max_tokens=max_tokens, temperature=temperature, timeout=timeout,
)
return _answer_via_oneshot(question, history, main_runtime=main_runtime, max_tokens=max_tokens,
temperature=temperature, timeout=timeout)
+2 -10
View File
@@ -14,12 +14,7 @@ _CA_BUNDLE_ENV_VARS = ("HERMES_CA_BUNDLE", "SSL_CERT_FILE", "REQUESTS_CA_BUNDLE"
_INSECURE_STRINGS = {"false", "0", "no", "off"}
def resolve_httpx_verify(
*,
ca_bundle: Optional[str] = None,
ssl_verify: Any = None,
base_url: str = "",
) -> bool | ssl.SSLContext:
def resolve_httpx_verify(*, ca_bundle: Optional[str] = None, ssl_verify: Any = None, base_url: str = "") -> bool | ssl.SSLContext:
"""Resolve httpx ``verify``: ``ssl_verify: false`` > explicit ``ca_bundle`` >
CA-bundle env vars > ``True`` (certifi default). ``base_url`` only feeds the warning."""
if ssl_verify is False or (isinstance(ssl_verify, str) and ssl_verify.strip().lower() in _INSECURE_STRINGS):
@@ -38,8 +33,5 @@ def resolve_httpx_verify(
ca_path = str(Path(effective_ca).expanduser())
if os.path.isfile(ca_path):
return ssl.create_default_context(cafile=ca_path)
logger.warning(
"CA bundle path does not exist: %s — falling back to default certificates",
effective_ca,
)
logger.warning("CA bundle path does not exist: %s — falling back to default certificates", effective_ca)
return True
+3 -11
View File
@@ -82,9 +82,7 @@ class TerminalEnvironmentProvider(ProviderBase):
def probe(self) -> Tuple[str, str]:
"""Dashboard picker health probe ``(status, detail)`` with status in
``ready`` / ``needs_setup`` / ``unavailable``. Must never raise; stay fast (<~2s)."""
if self.is_available():
return ("ready", "")
return ("needs_setup", f"{self.display_name} is not configured.")
return ("ready", "") if self.is_available() else ("needs_setup", f"{self.display_name} is not configured.")
def setup_instructions(self) -> List[str]:
"""Lines printed by ``hermes setup`` after this backend is selected. The wizard
@@ -107,14 +105,8 @@ class TerminalEnvironmentProvider(ProviderBase):
@abc.abstractmethod
def create_environment(
self,
*,
cwd: str,
timeout: int,
task_id: str = "default",
image: Optional[str] = None,
container_config: Optional[Dict[str, Any]] = None,
**kwargs: Any,
self, *, cwd: str, timeout: int, task_id: str = "default", image: Optional[str] = None,
container_config: Optional[Dict[str, Any]] = None, **kwargs: Any,
):
"""Create and return an execution environment (``BaseEnvironment`` duck type).
+1 -3
View File
@@ -35,9 +35,7 @@ def is_thinking_timeout(classified: object, model: str, error_msg: str) -> bool:
return any(p in error_msg_lower for p in _THINKING_TIMEOUT_SUBSTRINGS)
def build_thinking_timeout_guidance(
provider: str, model: str, model_label: Optional[str] = None,
) -> str:
def build_thinking_timeout_guidance(provider: str, model: str, model_label: Optional[str] = None) -> str:
"""User-facing guidance appended to the final response. ``model`` is used verbatim in
the config snippet so it is copy-pasteable; ``model_label`` is the optional prose name."""
label = model_label or model
+1 -6
View File
@@ -25,12 +25,7 @@ class TranscriptionProvider(CatalogProviderBase):
@abc.abstractmethod
def transcribe(
self,
file_path: str,
*,
model: Optional[str] = None,
language: Optional[str] = None,
**extra: Any,
self, file_path: str, *, model: Optional[str] = None, language: Optional[str] = None, **extra: Any,
) -> Dict[str, Any]:
"""Transcribe ``file_path`` (existence + size already validated) into the module envelope.