feat(cli): add --format stream-json for structured JSONL output

Adds a --format flag to hermes chat single-query mode. stream-json
emits newline-delimited JSON events (init, text, tool_use, tool_result,
result envelope with token stats + exit code) to stdout for CI
pipelines and external tooling. Session ID stays on stderr.

Salvaged from PR #12278 by @ProDrifterDK onto current main, including
the follow-up commit enforcing the single-query contract (implies
quiet, rejects --tui, emits a final result record with exit code 130
on interrupt).
This commit is contained in:
Alan
2026-08-19 18:09:28 -07:00
committed by Teknium
parent 39fbfbe4e6
commit 1657a1ce2d
6 changed files with 289 additions and 8 deletions
+26 -7
View File
@@ -4085,8 +4085,9 @@ def _sync_cli_session_id_from_agent(cli) -> None:
cli.session_id = cli.agent.session_id
def _run_quiet_single_query(cli, effective_query):
def _run_quiet_single_query(cli, effective_query, emitter=None):
"""Quiet (-Q) one-shot turn: run, print the response (stderr for errors/session_id), then sys.exit with the automation exit code.
With a ``StreamJsonEmitter`` the final answer and the exit line become the terminal ``result`` JSONL record instead.
HERMES_TURN_AUTHOR (set only by a bot-to-bot dispatcher) is consumed here so tool subprocesses do not inherit it.
Nested Bot Mode notifies bind this session's key (not the dispatcher's) and resume in-process
before stdout is printed, so a teammate reply is the quiet run's final answer rather than a
@@ -4106,6 +4107,8 @@ def _run_quiet_single_query(cli, effective_query):
)
except KeyboardInterrupt:
_emit_interrupted_session_end(cli, reason="keyboard_interrupt")
if emitter is not None:
sys.exit(emitter.emit_result({"failed": True, "error": "Interrupted"}, session_id=cli.session_id or "", exit_code=130))
print(f"\nsession_id: {cli.session_id}", file=sys.stderr)
sys.exit(130)
# The exit line below reports session_id to stderr for automation wrappers;
@@ -4146,7 +4149,9 @@ def _run_quiet_single_query(cli, effective_query):
response = result.get("final_response", "") if isinstance(result, dict) else str(result)
# Surface backend errors that produced no visible output (e.g. invalid model slug
# -> provider 4xx) on stderr so piped stdout stays clean.
if (
if emitter is not None:
pass # the result record below carries text/error; nothing else may touch stdout
elif (
not response and isinstance(result, dict) and result.get("error")
and (result.get("failed") or result.get("partial"))
):
@@ -4162,7 +4167,8 @@ def _run_quiet_single_query(cli, effective_query):
except Exception as _goal_exc:
logger.debug("kanban goal loop failed: %s", _goal_exc)
print(f"\nsession_id: {cli.session_id}", file=sys.stderr)
if emitter is None:
print(f"\nsession_id: {cli.session_id}", file=sys.stderr)
# Exit code 0/1 for automation wrappers. Kanban workers that failed purely on
# rate-limit/billing exit with the EX_TEMPFAIL sentinel so the dispatcher releases
@@ -4176,6 +4182,8 @@ def _run_quiet_single_query(cli, effective_query):
_exit_code = _RL_CODE
except Exception:
_exit_code = 1
if emitter is not None:
_exit_code = emitter.emit_result(result, session_id=cli.session_id or "", exit_code=_exit_code)
sys.exit(_exit_code)
@@ -4454,8 +4462,9 @@ def _configure_quiet_agent(agent) -> None:
agent.tool_progress_mode = "off"
def _run_single_query_mode(cli, query, image, quiet, oneshot):
"""``-q``/``--image`` entry: seed an interactive session on a TTY, else run the one-shot turn and exit."""
def _run_single_query_mode(cli, query, image, quiet, oneshot, stream_json: bool = False):
"""``-q``/``--image`` entry: seed an interactive session on a TTY, else run the one-shot turn and exit.
``stream_json`` (implies quiet) swaps the plain-text final answer for the JSONL event protocol."""
if _should_seed_interactive(query, image, quiet, oneshot):
seeded_query, seeded_images = _collect_query_images(query, image)
logger.info(
@@ -4495,7 +4504,11 @@ def _run_single_query_mode(cli, query, image, quiet, oneshot):
request_overrides=turn_route.get("request_overrides"),
):
_configure_quiet_agent(cli.agent)
_run_quiet_single_query(cli, effective_query)
emitter = None
if stream_json:
from hermes_cli.stream_json import StreamJsonEmitter
emitter = StreamJsonEmitter.attach(cli.agent, session_id=cli.session_id or "")
_run_quiet_single_query(cli, effective_query, emitter=emitter)
sys.exit(1) # credentials or agent init failed
# No welcome banner (~420 ms cold); session id / resume hint come from _print_exit_summary().
@@ -4534,6 +4547,7 @@ def main(
w: bool = False,
checkpoints: bool = False,
pass_session_id: bool = False,
output_format: str = "text",
ignore_user_config: bool = False,
ignore_rules: bool = False,
):
@@ -4592,6 +4606,11 @@ def main(
_join_worktree = _start_worktree_setup(list_tools, list_toolsets, worktree, w)
query = query or q
# ``hermes chat`` already validated this; the direct Fire entry point gets the same contract.
if output_format == "stream-json":
if not query:
raise ValueError("--format stream-json requires -q/--query")
quiet = True
cli = _build_cli_from_args(model, toolsets, provider, reasoning, api_key, base_url, max_turns, run_budget,
verbose, compact, resume, checkpoints, pass_session_id, ignore_rules, skills)
@@ -4621,7 +4640,7 @@ def main(
_install_single_query_signal_handlers(cli)
if query or image:
_run_single_query_mode(cli, query, image, quiet, oneshot)
_run_single_query_mode(cli, query, image, quiet, oneshot, stream_json=output_format == "stream-json")
return
cli.run()
+3
View File
@@ -224,6 +224,9 @@ def _build_chat_parser(subparsers) -> argparse.ArgumentParser:
add("-v", "--verbose", action="store_true", default=SUPPRESS, help="Verbose output")
add("-Q", "--quiet", action="store_true",
help="Quiet mode for programmatic use: suppress banner, spinner, and tool previews. Only output the final response and session info.")
add("--format", choices=["text", "stream-json"], default="text", dest="output_format", help=(
"Output format for single-query mode (-q). 'text' prints the final response as plain text (default). "
"'stream-json' emits newline-delimited JSON events (JSONL), implies --quiet, and cannot be combined with --tui."))
add("--resume", "-r", metavar="SESSION_ID", default=SUPPRESS, help=(
"Resume a previous session by ID (shown on exit), or 'latest' "
"for the most recent session"))
+4 -1
View File
@@ -1685,7 +1685,9 @@ def cmd_chat(args):
_apply_safe_mode(args)
_apply_user_config_bypass(args)
_guard_noninteractive_user_config(args)
use_tui = _resolve_use_tui(args)
from hermes_cli.stream_json import stream_json_requested
# Structured stdout is a non-interactive protocol: it overrides HERMES_TUI/display.interface too.
use_tui = False if stream_json_requested(args) else _resolve_use_tui(args)
_resolve_chat_session_args(args, use_tui)
@@ -1739,6 +1741,7 @@ def cmd_chat(args):
"query": args.query,
"oneshot": bool(getattr(args, "oneshot_exit", False)),
"run_budget": getattr(args, "run_budget", None),
"output_format": getattr(args, "output_format", "text"),
"ignore_rules": getattr(args, "ignore_rules", False) or safe_mode,
"ignore_user_config": getattr(args, "ignore_user_config", False) or safe_mode,
"compact": getattr(args, "compact", False),
+97
View File
@@ -0,0 +1,97 @@
"""``hermes chat -q … --format stream-json``: one JSON object per stdout line.
CI runners and orchestrators consume a one-shot run without scraping human-formatted text:
``system/init`` → ``text`` deltas / ``tool_use`` / ``tool_result`` → one terminal ``result``
envelope (exit code, final text, token stats). Diagnostics and ``session_id`` stay on stderr.
"""
from __future__ import annotations
import json
import sys
import time
from typing import Any
_TOOL_OUTPUT_CAP = 5000
def stream_json_requested(args) -> bool:
"""True when ``--format stream-json`` was passed; exits 2 on the combinations the protocol forbids
(no query to answer, or the interactive TUI transport) and forces quiet mode on ``args``."""
if getattr(args, "output_format", "text") != "stream-json":
return False
if not (getattr(args, "query", None) or getattr(args, "query_file", None)):
print("Error: --format stream-json requires -q/--query.", file=sys.stderr)
raise SystemExit(2)
if getattr(args, "tui", False):
print("Error: --format stream-json cannot be used with --tui.", file=sys.stderr)
raise SystemExit(2)
args.quiet = True
return True
def _now_ms() -> int:
return int(time.time() * 1000)
class StreamJsonEmitter:
"""Agent-callback sink that writes JSONL events to stdout and flushes each line."""
def __init__(self, model: str = "", session_id: str = ""):
self._session_id = session_id
self._start = time.time()
self._tool_started: dict[str, float] = {}
self._emit({"type": "system", "subtype": "init", "model": model, "session_id": session_id})
@classmethod
def attach(cls, agent, *, session_id: str = "") -> "StreamJsonEmitter":
"""Emit ``init`` and route the agent's streaming/tool callbacks into this emitter."""
emitter = cls(model=getattr(agent, "model", "") or "", session_id=session_id)
agent.stream_delta_callback = emitter.on_text_delta
agent.tool_progress_callback = emitter.on_tool_progress
return emitter
def on_text_delta(self, text: str | None) -> None:
if text and str(text).strip():
self._emit({"type": "text", "text": text})
def on_tool_progress(self, event_type: str, tool_name: str | None = None, preview: Any = None, args: Any = None,
**kwargs: Any) -> None:
"""``tool.started`` → ``tool_use`` (with ``input`` when the args are a dict); ``tool.completed`` →
``tool_result``. Other progress events (reasoning, output risk) are not part of the protocol."""
name = tool_name or "unknown"
if event_type == "tool.started":
self._tool_started[name] = time.time()
payload: dict[str, Any] = {"type": "tool_use", "name": name}
if isinstance(args, dict):
payload["input"] = args
self._emit(payload)
elif event_type == "tool.completed":
duration = kwargs.get("duration") or (time.time() - self._tool_started.pop(name, time.time()))
output = str(kwargs.get("result") or "")
self._emit({"type": "tool_result", "name": name,
"output": output if len(output) <= _TOOL_OUTPUT_CAP else output[:_TOOL_OUTPUT_CAP] + "...",
"duration_ms": int(float(duration) * 1000), "is_error": bool(kwargs.get("is_error", False))})
def emit_result(self, result: Any, session_id: str = "", exit_code: int = 0) -> int:
"""Write the terminal ``result`` record (once) and return the process exit code it reports."""
data = result if isinstance(result, dict) else {"final_response": "" if result is None else str(result)}
exit_code = exit_code or (1 if data.get("failed") else 0)
payload = {"type": "result", "session_id": session_id or self._session_id, "exit_code": exit_code,
"text": data.get("final_response") or "",
"tokens": {"input": data.get("input_tokens") or 0, "output": data.get("output_tokens") or 0,
"total": data.get("total_tokens") or 0, "cache_read": data.get("cache_read_tokens") or 0,
"cache_write": data.get("cache_write_tokens") or 0},
"duration_ms": int((time.time() - self._start) * 1000)}
if data.get("error"):
payload["error"] = str(data["error"])
self._emit(payload)
print(f"\nsession_id: {session_id or self._session_id}", file=sys.stderr) # same stderr contract as -Q
return exit_code
def _emit(self, obj: dict) -> None:
try:
sys.stdout.write(json.dumps({**obj, "timestamp": _now_ms()}, ensure_ascii=False) + "\n")
sys.stdout.flush()
except (BrokenPipeError, OSError):
pass # consumer closed the pipe — nothing left to report to
+132
View File
@@ -0,0 +1,132 @@
"""``hermes chat -q … --format stream-json`` emits a parseable JSONL event stream and nothing else on stdout."""
import json
import signal
import pytest
from hermes_cli.stream_json import StreamJsonEmitter
def _events(capsys):
out = capsys.readouterr().out
return [json.loads(line) for line in out.splitlines() if line]
def test_emitter_event_stream_is_valid_jsonl(capsys):
emitter = StreamJsonEmitter(model="test-model", session_id="s-1")
emitter.on_text_delta("hel")
emitter.on_text_delta(" ") # whitespace-only deltas carry no information
emitter.on_text_delta(None) # the turn-end sentinel the agent sends
emitter.on_tool_progress("tool.started", "read_file", "preview", {"path": "x"})
emitter.on_tool_progress("reasoning.available", "_thinking", "hmm", None) # not part of the protocol
emitter.on_tool_progress("tool.completed", "read_file", None, None, duration=0.5, is_error=False, result="x" * 6000)
code = emitter.emit_result({"final_response": "", "failed": True, "error": "boom", "input_tokens": 3}, exit_code=0)
events = _events(capsys)
assert [e["type"] for e in events] == ["system", "text", "tool_use", "tool_result", "result"]
assert events[0]["subtype"] == "init" and events[0]["model"] == "test-model"
assert events[2]["input"] == {"path": "x"}
assert events[3]["duration_ms"] == 500 and events[3]["output"].endswith("...") and len(events[3]["output"]) == 5003
assert code == 1 and events[-1] == {**events[-1], "exit_code": 1, "error": "boom", "session_id": "s-1"}
assert events[-1]["tokens"]["input"] == 3
assert all("timestamp" in e for e in events)
def _run_stream_json_chat(monkeypatch, capsys, run_conversation):
"""parser → cmd_chat → cli.main → quiet single-query path with a deterministic fake agent."""
import cli
import hermes_cli.main as cli_entry
from hermes_cli._parser import build_top_level_parser
class FakeAgent:
model = "test-model"
session_id = "session-123"
def run_conversation(self, **_kwargs):
return run_conversation(self)
class FakeCLI:
def __init__(self, **_kwargs):
self.session_id = "session-123"
self.conversation_history = []
self.agent = None
self._active_agent_route_signature = None
self.tool_progress_mode = None
def _claim_active_session(self, *_a, **_k):
return True
def _ensure_runtime_credentials(self):
return True
def _resolve_turn_agent_config(self, _query):
return {"signature": "r", "model": None, "runtime": None, "request_overrides": None}
def _init_agent(self, **_kwargs):
self.agent = FakeAgent()
return True
def chat(self, _query, images=None):
print("human output") # must never reach stdout under stream-json
monkeypatch.setattr(cli, "HermesCLI", FakeCLI)
monkeypatch.setattr(cli, "_finalize_single_query", lambda _cli: None)
monkeypatch.setattr(cli, "_emit_interrupted_session_end", lambda *_a, **_k: None)
monkeypatch.setattr(cli, "_start_worktree_setup", lambda *_a, **_k: None)
monkeypatch.setattr(cli.atexit, "register", lambda *_a, **_k: None)
monkeypatch.setattr(signal, "signal", lambda *_a, **_k: None)
monkeypatch.setattr(cli_entry, "_resolve_use_tui", lambda _args: pytest.fail("TUI resolution consulted"))
monkeypatch.setattr(cli_entry, "_has_any_provider_configured", lambda: True)
monkeypatch.setattr(cli_entry, "_start_chat_background_prefetch", lambda: None)
monkeypatch.setattr(cli_entry, "_pin_kanban_board_env", lambda: None)
monkeypatch.setattr(cli_entry, "_confirm_startup_expensive_model_override", lambda _a: None)
monkeypatch.setattr(cli_entry, "_warn_retired_xai_models", lambda: None)
monkeypatch.setattr("hermes_cli.free_tier_bootstrap.run_bootstrap", lambda **_k: None)
monkeypatch.setattr("hermes_cli.quiet_single_query.continue_quiet_notify_completions", lambda *_a, **_k: None)
parser, _, _ = build_top_level_parser()
args = parser.parse_args(["chat", "-q", "hello", "--format", "stream-json"])
with pytest.raises(SystemExit) as exc_info:
cli_entry.cmd_chat(args)
return exc_info.value.code, _events(capsys)
def _ok_turn(agent):
agent.stream_delta_callback("hello")
agent.tool_progress_callback("tool.started", "read_file", "p", {"path": "f"})
agent.tool_progress_callback("tool.completed", "read_file", None, None, duration=0.01, result="contents")
return {"final_response": "hello", "failed": False}
def _interrupted_turn(_agent):
raise KeyboardInterrupt
@pytest.mark.parametrize("turn, exit_code, types", [
(_ok_turn, 0, ["system", "text", "tool_use", "tool_result", "result"]),
(_interrupted_turn, 130, ["system", "result"]),
])
def test_chat_stream_json_implies_quiet_and_closes_with_result(monkeypatch, capsys, turn, exit_code, types):
"""No ``-Q`` needed; stdout is only JSONL; the stream always ends in a ``result`` carrying the exit code."""
code, events = _run_stream_json_chat(monkeypatch, capsys, turn)
assert code == exit_code
assert [e["type"] for e in events] == types
assert events[-1]["exit_code"] == exit_code and events[-1]["session_id"] == "session-123"
@pytest.mark.parametrize("argv, message", [
(["chat", "--format", "stream-json"], "requires -q/--query"),
(["chat", "-q", "hi", "--format", "stream-json", "--tui"], "cannot be used with --tui"),
(["--tui", "chat", "-q", "hi", "--format", "stream-json"], "cannot be used with --tui"),
])
def test_chat_stream_json_rejects_interactive_combinations(monkeypatch, capsys, argv, message):
import hermes_cli.main as cli_entry
from hermes_cli._parser import build_top_level_parser
monkeypatch.setattr(cli_entry, "_launch_tui", lambda *_a, **_k: pytest.fail("TUI launched"))
parser, _, _ = build_top_level_parser()
with pytest.raises(SystemExit) as exc_info:
cli_entry.cmd_chat(parser.parse_args(argv))
assert exc_info.value.code == 2
assert message in capsys.readouterr().err
+27
View File
@@ -121,6 +121,7 @@ Common options:
| `-s`, `--skills <name>` | Preload one or more skills for the session (can be repeated or comma-separated). |
| `-v`, `--verbose` | Verbose output. |
| `-Q`, `--quiet` | Programmatic mode: suppress banner/spinner/tool previews. |
| `--format stream-json` | Emit structured JSONL for a `-q` / `--query` invocation. Implies `--quiet`; cannot be combined with `--tui`. |
| `--image <path>` | Attach a local image to a single query. |
| `--resume <session>` / `--continue [name]` | Resume a session directly from `chat`. |
| `--worktree` | Create an isolated git worktree for this run. |
@@ -142,11 +143,37 @@ hermes chat --oneshot -q "Summarize the latest PRs" # answer and exit
hermes chat --provider openrouter --model anthropic/claude-sonnet-4.6
hermes chat --toolsets web,terminal,skills
hermes chat --quiet -q "Return only JSON"
hermes chat -q "Inspect this repository" --format stream-json
hermes chat --worktree -q "Review this repo and open a PR"
hermes chat --ignore-user-config --ignore-rules -q "Repro without my personal setup"
hermes chat --safe-mode -q "Is this bug mine or Hermes'?"
```
### `--format stream-json` — structured JSONL output
Use `--format stream-json` when a program needs to consume progress without
scraping terminal output. It requires `-q` / `--query` (or `--query-file`), implies
quiet non-interactive CLI mode, and rejects an explicit `--tui` request. Every
stdout line is one JSON object; diagnostics and the `session_id:` line stay on stderr.
```bash
hermes chat -q "Summarize this repository" --format stream-json
```
Every event carries `timestamp` (Unix epoch milliseconds).
| Event `type` | Fields |
|---|---|
| `system` | `subtype: "init"`, `model`, `session_id` |
| `text` | `text` — a streamed assistant text delta |
| `tool_use` | `name`; `input` when the tool arguments are available |
| `tool_result` | `name`, `output` (capped at 5000 chars), `duration_ms`, `is_error` |
| `result` | `session_id`, `exit_code`, `text`, `tokens` (`input`, `output`, `total`, `cache_read`, `cache_write`), `duration_ms`; `error` when the turn failed |
Once a conversation starts, its terminal record is always `result` — including
`exit_code: 130` when it is interrupted with Ctrl-C. Treat that record as the
completion signal; the process exit code matches its `exit_code`.
#### Delegation in finite chat runs
When chat answers and exits (`-Q`, `chat --oneshot`, or a query with non-TTY