358 lines
16 KiB
Python
358 lines
16 KiB
Python
"""Raw-YAML config and token/cost analytics dashboard routes.
|
|
|
|
Extracted from ``hermes_cli.web_server``; helpers/state that tests monkeypatch on
|
|
``web_server`` stay there and are imported lazily at call time (cycle-safe).
|
|
"""
|
|
|
|
import yaml
|
|
import asyncio
|
|
import time
|
|
from fastapi import APIRouter
|
|
from hermes_cli.web_deps import late
|
|
from fastapi import HTTPException, Query
|
|
from hermes_cli.config import get_config_path, read_raw_config
|
|
from hermes_cli.web_models import RawConfigUpdate
|
|
from typing import Any, Dict, List, Optional
|
|
|
|
router = APIRouter()
|
|
|
|
# web_server helpers, late-bound so monkeypatch.setattr(web_server, ...) stays authoritative.
|
|
_approval_mode_of = late("_approval_mode_of")
|
|
_aux_task_summary = late("_aux_task_summary")
|
|
_aux_usage_rows = late("_aux_usage_rows")
|
|
_broadcast_gateway_session_info = late("_broadcast_gateway_session_info")
|
|
_is_other_profile = late("_is_other_profile")
|
|
_merge_aux_into_by_model = late("_merge_aux_into_by_model")
|
|
_open_session_db_for_profile = late("_open_session_db_for_profile")
|
|
_profile_scope = late("_profile_scope")
|
|
save_config = late("save_config")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Computer Use (cua-driver) — cross-platform readiness + macOS permission grant
|
|
#
|
|
# cua-driver runs on macOS, Windows, and Linux. The desktop card reflects
|
|
# per-OS readiness: on macOS the Accessibility + Screen Recording TCC grants
|
|
# (which attach to cua-driver's OWN identity, com.trycua.driver — not Hermes,
|
|
# so no app entitlement is involved); elsewhere, driver health from
|
|
# `cua-driver doctor`. The grant flow is macOS-only (no TCC toggles to request
|
|
# on Windows/Linux).
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Raw YAML config endpoint
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@router.get("/api/config/raw")
|
|
async def get_config_raw(profile: Optional[str] = None):
|
|
"""Raw config.yaml text plus its resolved path.
|
|
|
|
``path`` is resolved inside ``_profile_scope`` so the Config page header
|
|
shows the file the switched profile actually reads/writes — /api/status's
|
|
``config_path`` is machine-global and always reports the dashboard
|
|
process's own profile, which is wrong under the global profile switcher.
|
|
"""
|
|
def _run():
|
|
with _profile_scope(profile):
|
|
path = get_config_path()
|
|
if not path.exists():
|
|
return {"yaml": "", "path": str(path)}
|
|
return {"yaml": path.read_text(encoding="utf-8"), "path": str(path)}
|
|
|
|
return await asyncio.to_thread(_run)
|
|
|
|
|
|
@router.put("/api/config/raw")
|
|
async def update_config_raw(body: RawConfigUpdate, profile: Optional[str] = None):
|
|
def _run():
|
|
parsed = yaml.safe_load(body.yaml_text)
|
|
if not isinstance(parsed, dict):
|
|
raise HTTPException(status_code=400, detail="YAML must be a mapping")
|
|
approvals_mode_changed = False
|
|
with _profile_scope(body.profile or profile):
|
|
# Full-document replacement: the editor owns the whole file; do not
|
|
# merge omitted sections back from disk (#62723).
|
|
approvals_mode_changed = _approval_mode_of(parsed) != _approval_mode_of(read_raw_config())
|
|
save_config(parsed, merge_existing=False)
|
|
# Same indicator refresh as the schema-driven save above.
|
|
if approvals_mode_changed and not _is_other_profile(body.profile or profile):
|
|
_broadcast_gateway_session_info()
|
|
return {"ok": True}
|
|
|
|
try:
|
|
return await asyncio.to_thread(_run)
|
|
except yaml.YAMLError as e:
|
|
raise HTTPException(status_code=400, detail=f"Invalid YAML: {e}")
|
|
|
|
|
|
def _get_usage_analytics(days: int = 30, profile: Optional[str] = None):
|
|
from agent.insights import InsightsEngine
|
|
|
|
db = _open_session_db_for_profile(profile, read_only=True)
|
|
try:
|
|
cutoff = time.time() - (days * 86400)
|
|
cur = db._conn.execute("""
|
|
SELECT date(started_at, 'unixepoch') as day,
|
|
SUM(input_tokens) as input_tokens,
|
|
SUM(output_tokens) as output_tokens,
|
|
SUM(cache_read_tokens) as cache_read_tokens,
|
|
SUM(reasoning_tokens) as reasoning_tokens,
|
|
COALESCE(SUM(estimated_cost_usd), 0) as estimated_cost,
|
|
COALESCE(SUM(actual_cost_usd), 0) as actual_cost,
|
|
COUNT(*) as sessions,
|
|
SUM(COALESCE(api_call_count, 0)) as api_calls
|
|
FROM sessions WHERE started_at > ?
|
|
GROUP BY day ORDER BY day
|
|
""", (cutoff,))
|
|
daily = [dict(r) for r in cur.fetchall()]
|
|
|
|
cur2 = db._conn.execute("""
|
|
SELECT model,
|
|
SUM(input_tokens) as input_tokens,
|
|
SUM(output_tokens) as output_tokens,
|
|
COALESCE(SUM(estimated_cost_usd), 0) as estimated_cost,
|
|
COUNT(*) as sessions,
|
|
SUM(COALESCE(api_call_count, 0)) as api_calls
|
|
FROM sessions WHERE started_at > ? AND model IS NOT NULL
|
|
GROUP BY model ORDER BY SUM(input_tokens) + SUM(output_tokens) DESC
|
|
""", (cutoff,))
|
|
by_model = [dict(r) for r in cur2.fetchall()]
|
|
|
|
# Fold in auxiliary usage (vision, compression, title_generation, ...)
|
|
# recorded per (model, task) in session_model_usage. Aux calls never
|
|
# touch the sessions counters, so this is add-only — no double count.
|
|
# Without it the models list shows only the main agent model even when
|
|
# aux models are actively burning tokens (issue #23270).
|
|
aux_rows = _aux_usage_rows(db, cutoff)
|
|
by_model = _merge_aux_into_by_model(by_model, aux_rows)
|
|
|
|
cur3 = db._conn.execute("""
|
|
SELECT SUM(input_tokens) as total_input,
|
|
SUM(output_tokens) as total_output,
|
|
SUM(cache_read_tokens) as total_cache_read,
|
|
SUM(reasoning_tokens) as total_reasoning,
|
|
COALESCE(SUM(estimated_cost_usd), 0) as total_estimated_cost,
|
|
COALESCE(SUM(actual_cost_usd), 0) as total_actual_cost,
|
|
COUNT(*) as total_sessions,
|
|
SUM(COALESCE(api_call_count, 0)) as total_api_calls
|
|
FROM sessions WHERE started_at > ?
|
|
""", (cutoff,))
|
|
totals = dict(cur3.fetchone())
|
|
usage = InsightsEngine(db).get_usage_breakdown(days=days)
|
|
|
|
return {
|
|
"daily": daily,
|
|
"by_model": by_model,
|
|
# Aux-task summary across models (vision, compression, ...). Lets
|
|
# the dashboard answer "what is compression costing me" directly.
|
|
"by_task": _aux_task_summary(aux_rows),
|
|
"totals": totals,
|
|
"period_days": days,
|
|
"skills": usage["skills"],
|
|
# Per-tool-name call counts (already computed by InsightsEngine);
|
|
# the desktop Capabilities page aggregates these per toolset.
|
|
"tools": usage["tools"],
|
|
}
|
|
finally:
|
|
db.close()
|
|
|
|
|
|
@router.get("/api/analytics/usage")
|
|
async def get_usage_analytics(
|
|
days: int = Query(30, ge=1, le=365),
|
|
profile: Optional[str] = None,
|
|
):
|
|
"""``days`` is clamped to 1-365 (idea from #74778): huge or non-positive
|
|
values would force expensive full-history SQL and InsightsEngine work, or
|
|
produce empty/inverted time windows. The UI only offers 7/30/90-day
|
|
presets."""
|
|
return await asyncio.to_thread(_get_usage_analytics, days, profile)
|
|
|
|
|
|
def _get_models_analytics(days: int = 30, profile: Optional[str] = None):
|
|
"""Rich per-model analytics for the Models dashboard page.
|
|
|
|
Returns token/cost/session breakdown per model plus capability metadata
|
|
from models.dev (context window, vision, tools, reasoning, etc.).
|
|
"""
|
|
db = _open_session_db_for_profile(profile, read_only=True)
|
|
try:
|
|
cutoff = time.time() - (days * 86400)
|
|
|
|
cur = db._conn.execute("""
|
|
SELECT model,
|
|
billing_provider,
|
|
SUM(input_tokens) as input_tokens,
|
|
SUM(output_tokens) as output_tokens,
|
|
SUM(cache_read_tokens) as cache_read_tokens,
|
|
SUM(reasoning_tokens) as reasoning_tokens,
|
|
COALESCE(SUM(estimated_cost_usd), 0) as estimated_cost,
|
|
COALESCE(SUM(actual_cost_usd), 0) as actual_cost,
|
|
COUNT(*) as sessions,
|
|
SUM(COALESCE(api_call_count, 0)) as api_calls,
|
|
SUM(tool_call_count) as tool_calls,
|
|
MAX(started_at) as last_used_at,
|
|
AVG(input_tokens + output_tokens) as avg_tokens_per_session
|
|
FROM sessions WHERE started_at > ? AND model IS NOT NULL AND model != ''
|
|
GROUP BY model, billing_provider
|
|
ORDER BY SUM(input_tokens) + SUM(output_tokens) DESC
|
|
""", (cutoff,))
|
|
raw_rows = [dict(r) for r in cur.fetchall()]
|
|
|
|
# Add auxiliary usage as (model, provider) rows so aux-only models
|
|
# (dedicated vision/compression models) appear on the Models page
|
|
# instead of being invisible (issue #23270). Keyed by
|
|
# model+billing_provider to match the GROUP BY above.
|
|
for aux in _aux_usage_rows(db, cutoff):
|
|
raw_rows.append({
|
|
"model": aux.get("model") or "unknown",
|
|
"billing_provider": aux.get("billing_provider") or "",
|
|
"input_tokens": aux.get("input_tokens") or 0,
|
|
"output_tokens": aux.get("output_tokens") or 0,
|
|
"cache_read_tokens": aux.get("cache_read_tokens") or 0,
|
|
"reasoning_tokens": aux.get("reasoning_tokens") or 0,
|
|
"estimated_cost": aux.get("estimated_cost") or 0,
|
|
"actual_cost": 0,
|
|
"sessions": aux.get("sessions") or 0,
|
|
"api_calls": aux.get("api_calls") or 0,
|
|
"tool_calls": 0,
|
|
"last_used_at": aux.get("last_used_at"),
|
|
"avg_tokens_per_session": 0,
|
|
"aux_task": aux.get("task") or "",
|
|
})
|
|
|
|
# Session rows can be created before the first billable provider call
|
|
# finishes. If that early row records only the model name, and a later
|
|
# row for the same model has real accounting + billing_provider, the
|
|
# Models page used to show a duplicate "0 tokens / — API calls" card
|
|
# next to the real provider card. Fold those session-only rows into
|
|
# the single accounted provider row when the ownership is unambiguous.
|
|
rows_by_model: Dict[str, List[Dict[str, Any]]] = {}
|
|
for row in raw_rows:
|
|
rows_by_model.setdefault(row.get("model") or "", []).append(row)
|
|
|
|
rows: List[Dict[str, Any]] = []
|
|
for model_rows in rows_by_model.values():
|
|
provider_rows = [r for r in model_rows if r.get("billing_provider")]
|
|
if len(provider_rows) == 1:
|
|
target = provider_rows[0]
|
|
for row in model_rows:
|
|
if row is target or row.get("billing_provider"):
|
|
continue
|
|
has_usage = any(
|
|
(row.get(key) or 0) != 0
|
|
for key in (
|
|
"input_tokens",
|
|
"output_tokens",
|
|
"cache_read_tokens",
|
|
"reasoning_tokens",
|
|
"estimated_cost",
|
|
"actual_cost",
|
|
"api_calls",
|
|
"tool_calls",
|
|
)
|
|
)
|
|
if has_usage:
|
|
continue
|
|
target["sessions"] = (target.get("sessions") or 0) + (row.get("sessions") or 0)
|
|
target["last_used_at"] = max(target.get("last_used_at") or 0, row.get("last_used_at") or 0)
|
|
total_tokens = (target.get("input_tokens") or 0) + (target.get("output_tokens") or 0)
|
|
sessions = target.get("sessions") or 0
|
|
target["avg_tokens_per_session"] = total_tokens / sessions if sessions else 0
|
|
rows.append(target)
|
|
rows.extend(
|
|
r for r in model_rows
|
|
if r is not target
|
|
and (r.get("billing_provider") or any(
|
|
(r.get(key) or 0) != 0
|
|
for key in (
|
|
"input_tokens",
|
|
"output_tokens",
|
|
"cache_read_tokens",
|
|
"reasoning_tokens",
|
|
"estimated_cost",
|
|
"actual_cost",
|
|
"api_calls",
|
|
"tool_calls",
|
|
)
|
|
))
|
|
)
|
|
else:
|
|
rows.extend(model_rows)
|
|
|
|
rows.sort(
|
|
key=lambda r: (r.get("input_tokens") or 0) + (r.get("output_tokens") or 0),
|
|
reverse=True,
|
|
)
|
|
|
|
models = []
|
|
for row in rows:
|
|
provider = row.get("billing_provider") or ""
|
|
model_name = row["model"]
|
|
caps = {}
|
|
try:
|
|
from agent.models_dev import get_model_capabilities
|
|
mc = get_model_capabilities(provider=provider, model=model_name)
|
|
if mc is not None:
|
|
caps = {
|
|
"supports_tools": mc.supports_tools,
|
|
"supports_vision": mc.supports_vision,
|
|
"supports_reasoning": mc.supports_reasoning,
|
|
"context_window": mc.context_window,
|
|
"max_output_tokens": mc.max_output_tokens,
|
|
"model_family": mc.model_family,
|
|
}
|
|
except Exception:
|
|
pass
|
|
|
|
models.append({
|
|
"model": model_name,
|
|
"provider": provider,
|
|
"input_tokens": row["input_tokens"],
|
|
"output_tokens": row["output_tokens"],
|
|
"cache_read_tokens": row["cache_read_tokens"],
|
|
"reasoning_tokens": row["reasoning_tokens"],
|
|
"estimated_cost": row["estimated_cost"],
|
|
"actual_cost": row["actual_cost"],
|
|
"sessions": row["sessions"],
|
|
"api_calls": row["api_calls"],
|
|
"tool_calls": row["tool_calls"],
|
|
"last_used_at": row["last_used_at"],
|
|
"avg_tokens_per_session": row["avg_tokens_per_session"],
|
|
"capabilities": caps,
|
|
})
|
|
|
|
totals_cur = db._conn.execute("""
|
|
SELECT COUNT(DISTINCT model) as distinct_models,
|
|
SUM(input_tokens) as total_input,
|
|
SUM(output_tokens) as total_output,
|
|
SUM(cache_read_tokens) as total_cache_read,
|
|
SUM(reasoning_tokens) as total_reasoning,
|
|
COALESCE(SUM(estimated_cost_usd), 0) as total_estimated_cost,
|
|
COALESCE(SUM(actual_cost_usd), 0) as total_actual_cost,
|
|
COUNT(*) as total_sessions,
|
|
SUM(COALESCE(api_call_count, 0)) as total_api_calls
|
|
FROM sessions WHERE started_at > ? AND model IS NOT NULL AND model != ''
|
|
""", (cutoff,))
|
|
totals = dict(totals_cur.fetchone())
|
|
|
|
return {
|
|
"models": models,
|
|
"totals": totals,
|
|
"period_days": days,
|
|
}
|
|
finally:
|
|
db.close()
|
|
|
|
|
|
@router.get("/api/analytics/models")
|
|
async def get_models_analytics(
|
|
days: int = Query(30, ge=1, le=365),
|
|
profile: Optional[str] = None,
|
|
):
|
|
# ``days`` clamped to 1-365 (idea from #74778) — see get_usage_analytics.
|
|
"""Return model analytics without blocking the serving event loop."""
|
|
return await asyncio.to_thread(_get_models_analytics, days, profile)
|