diff --git a/cli.py b/cli.py index c308b5608e..f2502538e4 100644 --- a/cli.py +++ b/cli.py @@ -6546,16 +6546,17 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): base_read = cur_read delta_prompt = cur_prompt - base_prompt delta_read = cur_read - base_read - if delta_prompt > 0 and delta_read >= 0: - pct = int(round((delta_read / delta_prompt) * 100)) - pct = max(0, min(100, pct)) + # A zero-read regime hides the segment entirely (no cache data + # is not the same as a 0% hit worth alarming about), and the pct + # stays a float so renderers control their own precision. + if delta_prompt > 0 and delta_read > 0: + pct = max(0.0, min(100.0, (delta_read / delta_prompt) * 100)) snapshot["cache_hit_pct"] = pct - snapshot["cache_hit_label"] = f"{pct}%" - elif cur_prompt > 0 and cur_read >= 0 and base_prompt == 0 and base_read == 0: - pct = int(round((cur_read / cur_prompt) * 100)) if cur_prompt else 0 - pct = max(0, min(100, pct)) + snapshot["cache_hit_label"] = f"{pct:.0f}%" + elif cur_prompt > 0 and cur_read > 0 and base_prompt == 0 and base_read == 0: + pct = max(0.0, min(100.0, (cur_read / cur_prompt) * 100)) snapshot["cache_hit_pct"] = pct - snapshot["cache_hit_label"] = f"{pct}%" + snapshot["cache_hit_label"] = f"{pct:.0f}%" else: snapshot["cache_hit_pct"] = None snapshot["cache_hit_label"] = "" diff --git a/tests/cli/test_cli_status_bar.py b/tests/cli/test_cli_status_bar.py index 5b07bc15e2..76d25a8578 100644 --- a/tests/cli/test_cli_status_bar.py +++ b/tests/cli/test_cli_status_bar.py @@ -612,3 +612,106 @@ class TestCacheHitRate: # cache_read / prompt_tokens = 5000 / 10000 = 50% assert "◎ 50.0%" in text + + +class TestRollingLatencyVelocity: + def _with_history(self, cli_obj, latencies, outputs): + from collections import deque + cli_obj.agent._api_latency_history = deque(latencies, maxlen=10) + cli_obj.agent._api_output_history = deque(outputs, maxlen=10) + return cli_obj + + def test_latency_and_tps_shown_in_wide_terminal(self): + cli_obj = _attach_agent( + _make_cli(), + prompt_tokens=10_000, completion_tokens=2_000, total_tokens=12_000, + api_calls=5, context_tokens=12_000, context_length=200_000, + ) + self._with_history(cli_obj, [2.0, 4.0], [120, 180]) + + text = cli_obj._build_status_bar_text(width=140) + + assert "\u25f7 3.0s" in text # mean latency (2+4)/2 + assert "\u2191 50 t/s" in text # true throughput 300/6.0 + + def test_latency_hidden_without_history(self): + cli_obj = _attach_agent( + _make_cli(), + prompt_tokens=10_000, completion_tokens=2_000, total_tokens=12_000, + api_calls=5, context_tokens=12_000, context_length=200_000, + ) + text = cli_obj._build_status_bar_text(width=140) + assert "\u25f7" not in text + assert "t/s" not in text + + def test_latency_and_tps_respect_field_filter(self): + cli_obj = _attach_agent( + _make_cli(), + prompt_tokens=10_000, completion_tokens=2_000, total_tokens=12_000, + api_calls=5, context_tokens=12_000, context_length=200_000, + ) + self._with_history(cli_obj, [2.0], [100]) + with patch.object(cli_mod, "CLI_CONFIG", {"display": {"status_bar": {"fields": ["model", "duration"]}}}): + text = cli_obj._build_status_bar_text(width=140) + assert "\u25f7" not in text + assert "t/s" not in text + + def test_negative_latency_guard(self): + cli_obj = _attach_agent( + _make_cli(), + prompt_tokens=10_000, completion_tokens=2_000, total_tokens=12_000, + api_calls=5, context_tokens=12_000, context_length=200_000, + ) + self._with_history(cli_obj, [-0.8], [100]) + snapshot = cli_obj._get_status_bar_snapshot() + assert snapshot["avg_latency"] is None + assert snapshot["avg_velocity"] is None + + +class TestCacheHitBaselineReset: + def test_baseline_resets_on_model_switch(self): + cli_obj = _attach_agent( + _make_cli(), + prompt_tokens=10_000, completion_tokens=2_000, total_tokens=12_000, + api_calls=5, context_tokens=12_000, context_length=200_000, + cache_read_tokens=9_000, + ) + first = cli_obj._get_status_bar_snapshot() + assert first["cache_hit_pct"] == 90.0 + + # Switch model. The bar repaints every frame, so the switch is + # observed (and the baseline reset) before new tokens accrue. + cli_obj.model = "openai/gpt-5" + cli_obj.agent.model = "openai/gpt-5" + reset_snap = cli_obj._get_status_bar_snapshot() + assert reset_snap["cache_hit_pct"] is None # new regime, no data yet + + cli_obj.agent.session_prompt_tokens = 12_000 + cli_obj.agent.session_cache_read_tokens = 9_500 + second = cli_obj._get_status_bar_snapshot() + # Delta since switch: 500/2000 = 25%, not the lifetime 79%. + assert second["cache_hit_pct"] == 25.0 + + def test_baseline_resets_on_compression(self): + cli_obj = _attach_agent( + _make_cli(), + prompt_tokens=10_000, completion_tokens=2_000, total_tokens=12_000, + api_calls=5, context_tokens=12_000, context_length=200_000, + cache_read_tokens=8_000, + ) + cli_obj._get_status_bar_snapshot() + + cli_obj.agent.context_compressor.compression_count = 1 + cli_obj._get_status_bar_snapshot() # repaint observes the compression + + cli_obj.agent.session_prompt_tokens = 14_000 + cli_obj.agent.session_cache_read_tokens = 8_400 + snap = cli_obj._get_status_bar_snapshot() + assert snap["cache_hit_pct"] == 10.0 # 400/4000 post-compression + + def test_title_field_filter_hides_session_badge(self): + cli_obj = _make_cli() + cli_obj._pending_title = "weekly-digest" + with patch.object(cli_mod, "CLI_CONFIG", {"display": {"status_bar": {"fields": ["model", "duration"]}}}): + text = cli_obj._build_status_bar_text(width=80) + assert "weekly-digest" not in text diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 3263158ba5..27ff51bd54 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -1980,14 +1980,15 @@ display: fields: ["model", "duration", "total_tokens"] # visibility only; built-in order is preserved ``` -Supported fields: `model`, `context_detail` (used/total tokens), `context_pct` (percent + meter), `compressions`, `bg_tasks`, `bg_processes`, `bg_subagents`, `goal`, `duration`, `prompt_elapsed`, `idle_since`, `focus`, `yolo`, and `total_tokens` (session Σ — opt-in only, never shown by default). +Supported fields: `model`, `context_detail` (used/total tokens), `context_pct` (percent + meter), `cache_hit` (prompt cache hit ratio — resets on model switch and compression), `latency` (rolling mean API latency, last 10 calls), `tps` (rolling output tokens/sec, last 10 calls), `compressions`, `bg_tasks`, `bg_processes`, `bg_subagents`, `goal`, `duration`, `prompt_elapsed`, `idle_since`, `focus`, `yolo`, `stash`, `battery`, `title` (right-aligned session badge), and `total_tokens` (session Σ — opt-in only, never shown by default). Notes: - An empty list (the default) keeps the standard set — everything except `total_tokens`. - The config controls **visibility, not order**; fields render in their built-in positions. -- Narrow terminals still drop wide-mode-only fields (`context_detail`, `prompt_elapsed`, `idle_since`) regardless of config. -- The battery indicator, session title, and prompt-stash indicator have their own toggles (`/battery`, `/title`) and are not governed by this list. +- Narrow terminals still drop wide-mode-only fields (`context_detail`, `cache_hit`, `latency`, `tps`, `prompt_elapsed`, `idle_since`) regardless of config (`cache_hit` also shows in the medium ≥52-col tier). +- `latency`/`tps` stay hidden until API calls have been recorded (e.g. the Codex app-server backend reports no latency). +- `battery` and `title` visibility here compose with their own toggles (`/battery`, `/title`) — both must be on for the segment to show. - Display-only: no effect on prompt caching or request payloads. Changes take effect on the next session start. ### Runtime-metadata footer (gateway only)