feat(plugins): route ctx.llm.complete(task=) through registered aux slots

Wire an optional `task=<key>` kwarg through the PluginLlm facade so a
plugin can route an LLM call through an auxiliary model slot it
registered via `ctx.register_auxiliary_task`. Registration already
existed; this adds the missing consumption half. Closes #44673.
Sub-issue 08/14 of the plugin-interface expansion tracking issue #64182.

- New optional `task:` kwarg on complete/acomplete/complete_structured/
  acomplete_structured. Unset or "auto" keeps today's main-model path
  byte-for-byte (task=None reaches call_llm exactly as before), so no
  prompt-cache or default-behavior change.
- A set task resolves provider/model through `auxiliary.<task>` via the
  existing auxiliary_client path, identical to built-in aux tasks.
- Trust gate (per the round-2 design correction): a plugin may only pass
  a key it registered itself; a built-in key additionally requires
  `plugins.entries.<id>.llm.allow_task_override: true`. A foreign or
  unknown key is rejected with a PluginLlmTrustError and a logged warning
  naming the offending plugin and key -- fail loud, NOT a silent fallback
  to auto (which would mask misconfiguration and could route to the main
  model the user steered elsewhere).
- The plugin_llm audit dict and audit-log line gain a `task` field.
- register_auxiliary_task now stores the plugin's canonical id
  (`key or name`, the same id ctx.llm is bound to) as the slot owner, so
  the trust gate matches ownership even when a manifest sets a distinct
  key. For the common no-key case this equals the name (unchanged).

Tests (tests/agent/test_plugin_llm_task_routing.py, 24): _check_task
resolution incl. own/foreign/unknown/built-in-gated keys and loud
rejection; end-to-end routing sync+async+structured; production-path
forwarding into call_llm/async_call_llm (covers the task=None->task line);
and ownership resolution against the real plugin registry incl. the
name/key reconciliation.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_013b1XyXitAxV7phGmKWigJX
This commit is contained in:
hans
2026-07-16 22:54:17 +08:00
committed by Teknium
parent e0bb71cb73
commit 1176222b7c
3 changed files with 561 additions and 12 deletions
+154 -10
View File
@@ -46,12 +46,21 @@ override knobs (``provider=``, ``model=``, ``agent_id=``,
allowed_models: [openai/gpt-4o-mini] # optional
allow_agent_id_override: false
allow_profile_override: false
allow_task_override: false # borrow the host's built-in aux tasks
Untrusted plugins still get the default surface — they just can't
steer provider, model, agent, or auth-profile selection. The trust
gate is fail-closed: a missing config block means "no overrides,"
not "anything goes."
The ``task=`` kwarg on the ``complete``/``acomplete`` family routes a
call through a plugin-registered auxiliary model slot
(``ctx.register_auxiliary_task``). A plugin may always name a slot it
registered itself; ``allow_task_override`` additionally lets it route
through the host's *built-in* auxiliary tasks. A foreign or unknown key
is rejected loudly (error + logged warning), never silently downgraded
to the main model.
Backed by :func:`agent.auxiliary_client.call_llm`, which already
handles every provider, fallback chain, and per-task override Hermes
supports.
@@ -173,6 +182,11 @@ class _TrustPolicy:
allow_any_model: bool = False # True when allowed_models == ["*"]
allow_agent_id_override: bool = False
allow_profile_override: bool = False
# Gates routing a call through a *built-in* auxiliary task slot via
# ``ctx.llm.complete(task=...)``. A plugin may always route through a
# slot it registered itself; this flag additionally lets it borrow the
# host's built-in aux tasks. Off by default (fail-closed).
allow_task_override: bool = False
def _normalize_ref(raw: str) -> str:
@@ -243,6 +257,7 @@ def _resolve_trust_policy(plugin_id: str) -> _TrustPolicy:
allow_any_model=allow_any_model,
allow_agent_id_override=bool(llm_cfg.get("allow_agent_id_override", False)),
allow_profile_override=bool(llm_cfg.get("allow_profile_override", False)),
allow_task_override=bool(llm_cfg.get("allow_task_override", False)),
)
@@ -329,6 +344,102 @@ def _check_overrides(
return final_provider, final_model, requested_agent_id, final_profile
def _resolve_task_ownership(plugin_id: str) -> tuple[frozenset, frozenset]:
"""Return ``(owned_keys, builtin_keys)`` for the task trust gate.
``owned_keys`` are auxiliary-task keys ``plugin_id`` registered itself
via ``ctx.register_auxiliary_task``; ``builtin_keys`` are the host's
reserved auxiliary tasks. Both imports are lazy so plugin discovery
doesn't hit a circular import at module load. A registry that can't be
read yields empty sets, which fails the gate closed (unknown → rejected).
Ownership matches on the same canonical id ``ctx.llm`` is bound to
(``manifest.key or manifest.name``); ``register_auxiliary_task`` stores
that same id as the entry's ``plugin`` owner.
"""
owned: set = set()
builtin: set = set()
try:
from hermes_cli.plugins import get_plugin_auxiliary_tasks
for entry in get_plugin_auxiliary_tasks():
if entry.get("plugin") == plugin_id:
key = entry.get("key")
if isinstance(key, str) and key:
owned.add(key)
except Exception: # pragma: no cover — registry unavailable
pass
try:
from hermes_cli.main import _AUX_TASKS
builtin = {k for k, _name, _desc in _AUX_TASKS}
except Exception: # pragma: no cover — main import failure
pass
return frozenset(owned), frozenset(builtin)
def _check_task(
policy: _TrustPolicy,
*,
plugin_id: str,
requested_task: Optional[str],
) -> Optional[str]:
"""Validate a plugin's requested auxiliary ``task`` routing key.
Returns the normalized key to route through, or ``None`` for the
default main-model path. Resolution:
* unset / ``""`` / ``"auto"`` → ``None`` (today's behavior, byte-for-byte).
* a key the plugin registered itself → allowed.
* a built-in auxiliary key → allowed only when
``plugins.entries.<id>.llm.allow_task_override`` is true.
* anything else (foreign or unknown) → **rejected loudly**.
A foreign/unknown key raises :class:`PluginLlmTrustError` and logs a
warning naming the offending plugin and key. It is deliberately *not*
silently downgraded to ``auto``: silent fallback masks the
misconfiguration and could route the call to the main model the user
may have steered elsewhere on purpose (round-2 design correction,
tracked in #64182 / #64174).
"""
if not requested_task:
return None
task = requested_task.strip()
if not task or task.lower() == "auto":
return None
owned, builtin = _resolve_task_ownership(plugin_id)
if task in owned:
return task
if task in builtin:
if policy.allow_task_override:
return task
logger.warning(
"plugin_llm task routing denied: plugin %r requested built-in "
"auxiliary task %r without plugins.entries.%s.llm.allow_task_override",
plugin_id,
task,
plugin_id,
)
raise PluginLlmTrustError(
f"Plugin {plugin_id!r} cannot route through the built-in auxiliary "
f"task {task!r} (set plugins.entries.{plugin_id}.llm."
f"allow_task_override to true to allow)."
)
logger.warning(
"plugin_llm task routing denied: plugin %r requested auxiliary task %r "
"it did not register",
plugin_id,
task,
)
raise PluginLlmTrustError(
f"Plugin {plugin_id!r} cannot route through auxiliary task {task!r} — a "
f"plugin may only pass a task key it registered itself via "
f"ctx.register_auxiliary_task() (or a built-in key when plugins.entries."
f"{plugin_id}.llm.allow_task_override is true)."
)
# ---------------------------------------------------------------------------
# Input normalization
# ---------------------------------------------------------------------------
@@ -631,6 +742,7 @@ class PluginLlm:
agent_id: Optional[str] = None,
profile: Optional[str] = None,
purpose: Optional[str] = None,
task: Optional[str] = None,
) -> PluginLlmCompleteResult:
"""Run a host-owned chat completion against the user's active model.
@@ -640,8 +752,15 @@ class PluginLlm:
+ ``model.model``). Each is independently gated by
``plugins.entries.<id>.llm.allow_*_override`` (see module
docstring).
``task`` optionally routes the call through a plugin-registered
auxiliary model slot (``ctx.register_auxiliary_task``): unset or
``"auto"`` keeps today's main-model behavior, a slot the plugin
registered itself resolves through ``auxiliary.<task>`` config,
and a foreign/unknown key is rejected (see :func:`_check_task`).
"""
policy = self._policy_loader(self._plugin_id)
eff_task = _check_task(policy, plugin_id=self._plugin_id, requested_task=task)
eff_provider, eff_model, eff_agent, eff_profile = _check_overrides(
policy,
requested_provider=provider,
@@ -657,6 +776,7 @@ class PluginLlm:
temperature=temperature,
max_tokens=max_tokens,
timeout=timeout,
task=eff_task,
)
text = _extract_text(response)
usage = _extract_usage(response)
@@ -670,13 +790,14 @@ class PluginLlm:
"plugin_id": self._plugin_id,
"purpose": purpose or "",
"profile": eff_profile or "",
"task": eff_task or "",
},
)
logger.info(
"plugin_llm.complete plugin=%s provider=%s model=%s purpose=%s "
"tokens=%d",
self._plugin_id, real_provider, real_model, purpose or "",
usage.total_tokens,
"plugin_llm.complete plugin=%s provider=%s model=%s task=%s "
"purpose=%s tokens=%d",
self._plugin_id, real_provider, real_model, eff_task or "",
purpose or "", usage.total_tokens,
)
return result
@@ -697,6 +818,7 @@ class PluginLlm:
agent_id: Optional[str] = None,
profile: Optional[str] = None,
purpose: Optional[str] = None,
task: Optional[str] = None,
) -> PluginLlmStructuredResult:
"""Run a bounded host-owned structured completion.
@@ -709,6 +831,9 @@ class PluginLlm:
Validation requires the optional ``jsonschema`` package. When it
isn't installed, JSON mode still works but schema enforcement is
skipped with a debug log.
``task`` routes through a plugin-registered auxiliary slot (see
:meth:`complete`).
"""
if not instructions or not instructions.strip():
raise ValueError("complete_structured requires non-empty instructions")
@@ -716,6 +841,7 @@ class PluginLlm:
raise ValueError("complete_structured requires at least one input block")
policy = self._policy_loader(self._plugin_id)
eff_task = _check_task(policy, plugin_id=self._plugin_id, requested_task=task)
eff_provider, eff_model, eff_agent, eff_profile = _check_overrides(
policy,
requested_provider=provider,
@@ -743,6 +869,7 @@ class PluginLlm:
max_tokens=max_tokens,
timeout=timeout,
extra_body=extra_body,
task=eff_task,
)
text = _extract_text(response)
usage = _extract_usage(response)
@@ -762,13 +889,14 @@ class PluginLlm:
"purpose": purpose or "",
"profile": eff_profile or "",
"schema_name": schema_name or "",
"task": eff_task or "",
},
)
logger.info(
"plugin_llm.complete_structured plugin=%s provider=%s model=%s "
"purpose=%s content_type=%s tokens=%d",
self._plugin_id, real_provider, real_model, purpose or "",
content_type, usage.total_tokens,
"task=%s purpose=%s content_type=%s tokens=%d",
self._plugin_id, real_provider, real_model, eff_task or "",
purpose or "", content_type, usage.total_tokens,
)
return result
@@ -786,9 +914,11 @@ class PluginLlm:
agent_id: Optional[str] = None,
profile: Optional[str] = None,
purpose: Optional[str] = None,
task: Optional[str] = None,
) -> PluginLlmCompleteResult:
"""Async sibling of :meth:`complete`."""
policy = self._policy_loader(self._plugin_id)
eff_task = _check_task(policy, plugin_id=self._plugin_id, requested_task=task)
eff_provider, eff_model, eff_agent, eff_profile = _check_overrides(
policy,
requested_provider=provider,
@@ -804,6 +934,7 @@ class PluginLlm:
temperature=temperature,
max_tokens=max_tokens,
timeout=timeout,
task=eff_task,
)
text = _extract_text(response)
usage = _extract_usage(response)
@@ -817,6 +948,7 @@ class PluginLlm:
"plugin_id": self._plugin_id,
"purpose": purpose or "",
"profile": eff_profile or "",
"task": eff_task or "",
},
)
@@ -837,6 +969,7 @@ class PluginLlm:
agent_id: Optional[str] = None,
profile: Optional[str] = None,
purpose: Optional[str] = None,
task: Optional[str] = None,
) -> PluginLlmStructuredResult:
"""Async sibling of :meth:`complete_structured`."""
if not instructions or not instructions.strip():
@@ -845,6 +978,7 @@ class PluginLlm:
raise ValueError("acomplete_structured requires at least one input block")
policy = self._policy_loader(self._plugin_id)
eff_task = _check_task(policy, plugin_id=self._plugin_id, requested_task=task)
eff_provider, eff_model, eff_agent, eff_profile = _check_overrides(
policy,
requested_provider=provider,
@@ -870,6 +1004,7 @@ class PluginLlm:
max_tokens=max_tokens,
timeout=timeout,
extra_body=extra_body,
task=eff_task,
)
text = _extract_text(response)
usage = _extract_usage(response)
@@ -889,6 +1024,7 @@ class PluginLlm:
"purpose": purpose or "",
"profile": eff_profile or "",
"schema_name": schema_name or "",
"task": eff_task or "",
},
)
@@ -927,10 +1063,15 @@ class PluginLlm:
max_tokens: Optional[int],
timeout: Optional[float],
extra_body: Optional[Dict[str, Any]] = None,
task: Optional[str] = None,
) -> tuple[str, str, Any]:
"""Invoke the host's ``call_llm``. Lazy-imports
``agent.auxiliary_client`` to avoid circular deps at plugin
discovery time."""
discovery time.
``task`` (already trust-checked by the caller) routes through the
matching ``auxiliary.<task>`` slot; ``None`` keeps the main model.
"""
if self._sync_caller is not None:
return self._sync_caller(
messages=messages,
@@ -941,13 +1082,14 @@ class PluginLlm:
max_tokens=max_tokens,
timeout=timeout,
extra_body=extra_body,
task=task,
)
from agent.auxiliary_client import call_llm
merged_extra = dict(extra_body or {})
if profile_override:
merged_extra.setdefault("metadata", {})["auth_profile"] = profile_override
response = call_llm(
task=None,
task=task,
provider=provider_override,
model=model_override,
messages=messages,
@@ -974,6 +1116,7 @@ class PluginLlm:
max_tokens: Optional[int],
timeout: Optional[float],
extra_body: Optional[Dict[str, Any]] = None,
task: Optional[str] = None,
) -> tuple[str, str, Any]:
if self._async_caller is not None:
return await self._async_caller(
@@ -985,13 +1128,14 @@ class PluginLlm:
max_tokens=max_tokens,
timeout=timeout,
extra_body=extra_body,
task=task,
)
from agent.auxiliary_client import async_call_llm
merged_extra = dict(extra_body or {})
if profile_override:
merged_extra.setdefault("metadata", {})["auth_profile"] = profile_override
response = await async_call_llm(
task=None,
task=task,
provider=provider_override,
model=model_override,
messages=messages,