From 1657a1ce2ddccdc2d3e6884bc4d0415484a2e16c Mon Sep 17 00:00:00 2001 From: Alan Date: Wed, 19 Aug 2026 18:09:28 -0700 Subject: [PATCH] feat(cli): add --format stream-json for structured JSONL output Adds a --format flag to hermes chat single-query mode. stream-json emits newline-delimited JSON events (init, text, tool_use, tool_result, result envelope with token stats + exit code) to stdout for CI pipelines and external tooling. Session ID stays on stderr. Salvaged from PR #12278 by @ProDrifterDK onto current main, including the follow-up commit enforcing the single-query contract (implies quiet, rejects --tui, emits a final result record with exit code 130 on interrupt). --- cli.py | 33 +++++-- hermes_cli/_parser.py | 3 + hermes_cli/main.py | 5 +- hermes_cli/stream_json.py | 97 ++++++++++++++++++ tests/hermes_cli/test_stream_json.py | 132 +++++++++++++++++++++++++ website/docs/reference/cli-commands.md | 27 +++++ 6 files changed, 289 insertions(+), 8 deletions(-) create mode 100644 hermes_cli/stream_json.py create mode 100644 tests/hermes_cli/test_stream_json.py diff --git a/cli.py b/cli.py index a3542610c8..05a383ad9d 100644 --- a/cli.py +++ b/cli.py @@ -4085,8 +4085,9 @@ def _sync_cli_session_id_from_agent(cli) -> None: cli.session_id = cli.agent.session_id -def _run_quiet_single_query(cli, effective_query): +def _run_quiet_single_query(cli, effective_query, emitter=None): """Quiet (-Q) one-shot turn: run, print the response (stderr for errors/session_id), then sys.exit with the automation exit code. + With a ``StreamJsonEmitter`` the final answer and the exit line become the terminal ``result`` JSONL record instead. HERMES_TURN_AUTHOR (set only by a bot-to-bot dispatcher) is consumed here so tool subprocesses do not inherit it. Nested Bot Mode notifies bind this session's key (not the dispatcher's) and resume in-process before stdout is printed, so a teammate reply is the quiet run's final answer rather than a @@ -4106,6 +4107,8 @@ def _run_quiet_single_query(cli, effective_query): ) except KeyboardInterrupt: _emit_interrupted_session_end(cli, reason="keyboard_interrupt") + if emitter is not None: + sys.exit(emitter.emit_result({"failed": True, "error": "Interrupted"}, session_id=cli.session_id or "", exit_code=130)) print(f"\nsession_id: {cli.session_id}", file=sys.stderr) sys.exit(130) # The exit line below reports session_id to stderr for automation wrappers; @@ -4146,7 +4149,9 @@ def _run_quiet_single_query(cli, effective_query): response = result.get("final_response", "") if isinstance(result, dict) else str(result) # Surface backend errors that produced no visible output (e.g. invalid model slug # -> provider 4xx) on stderr so piped stdout stays clean. - if ( + if emitter is not None: + pass # the result record below carries text/error; nothing else may touch stdout + elif ( not response and isinstance(result, dict) and result.get("error") and (result.get("failed") or result.get("partial")) ): @@ -4162,7 +4167,8 @@ def _run_quiet_single_query(cli, effective_query): except Exception as _goal_exc: logger.debug("kanban goal loop failed: %s", _goal_exc) - print(f"\nsession_id: {cli.session_id}", file=sys.stderr) + if emitter is None: + print(f"\nsession_id: {cli.session_id}", file=sys.stderr) # Exit code 0/1 for automation wrappers. Kanban workers that failed purely on # rate-limit/billing exit with the EX_TEMPFAIL sentinel so the dispatcher releases @@ -4176,6 +4182,8 @@ def _run_quiet_single_query(cli, effective_query): _exit_code = _RL_CODE except Exception: _exit_code = 1 + if emitter is not None: + _exit_code = emitter.emit_result(result, session_id=cli.session_id or "", exit_code=_exit_code) sys.exit(_exit_code) @@ -4454,8 +4462,9 @@ def _configure_quiet_agent(agent) -> None: agent.tool_progress_mode = "off" -def _run_single_query_mode(cli, query, image, quiet, oneshot): - """``-q``/``--image`` entry: seed an interactive session on a TTY, else run the one-shot turn and exit.""" +def _run_single_query_mode(cli, query, image, quiet, oneshot, stream_json: bool = False): + """``-q``/``--image`` entry: seed an interactive session on a TTY, else run the one-shot turn and exit. + ``stream_json`` (implies quiet) swaps the plain-text final answer for the JSONL event protocol.""" if _should_seed_interactive(query, image, quiet, oneshot): seeded_query, seeded_images = _collect_query_images(query, image) logger.info( @@ -4495,7 +4504,11 @@ def _run_single_query_mode(cli, query, image, quiet, oneshot): request_overrides=turn_route.get("request_overrides"), ): _configure_quiet_agent(cli.agent) - _run_quiet_single_query(cli, effective_query) + emitter = None + if stream_json: + from hermes_cli.stream_json import StreamJsonEmitter + emitter = StreamJsonEmitter.attach(cli.agent, session_id=cli.session_id or "") + _run_quiet_single_query(cli, effective_query, emitter=emitter) sys.exit(1) # credentials or agent init failed # No welcome banner (~420 ms cold); session id / resume hint come from _print_exit_summary(). @@ -4534,6 +4547,7 @@ def main( w: bool = False, checkpoints: bool = False, pass_session_id: bool = False, + output_format: str = "text", ignore_user_config: bool = False, ignore_rules: bool = False, ): @@ -4592,6 +4606,11 @@ def main( _join_worktree = _start_worktree_setup(list_tools, list_toolsets, worktree, w) query = query or q + # ``hermes chat`` already validated this; the direct Fire entry point gets the same contract. + if output_format == "stream-json": + if not query: + raise ValueError("--format stream-json requires -q/--query") + quiet = True cli = _build_cli_from_args(model, toolsets, provider, reasoning, api_key, base_url, max_turns, run_budget, verbose, compact, resume, checkpoints, pass_session_id, ignore_rules, skills) @@ -4621,7 +4640,7 @@ def main( _install_single_query_signal_handlers(cli) if query or image: - _run_single_query_mode(cli, query, image, quiet, oneshot) + _run_single_query_mode(cli, query, image, quiet, oneshot, stream_json=output_format == "stream-json") return cli.run() diff --git a/hermes_cli/_parser.py b/hermes_cli/_parser.py index 35a48e2a3c..ff72682418 100644 --- a/hermes_cli/_parser.py +++ b/hermes_cli/_parser.py @@ -224,6 +224,9 @@ def _build_chat_parser(subparsers) -> argparse.ArgumentParser: add("-v", "--verbose", action="store_true", default=SUPPRESS, help="Verbose output") add("-Q", "--quiet", action="store_true", help="Quiet mode for programmatic use: suppress banner, spinner, and tool previews. Only output the final response and session info.") + add("--format", choices=["text", "stream-json"], default="text", dest="output_format", help=( + "Output format for single-query mode (-q). 'text' prints the final response as plain text (default). " + "'stream-json' emits newline-delimited JSON events (JSONL), implies --quiet, and cannot be combined with --tui.")) add("--resume", "-r", metavar="SESSION_ID", default=SUPPRESS, help=( "Resume a previous session by ID (shown on exit), or 'latest' " "for the most recent session")) diff --git a/hermes_cli/main.py b/hermes_cli/main.py index 43ed1b686d..535580b596 100644 --- a/hermes_cli/main.py +++ b/hermes_cli/main.py @@ -1685,7 +1685,9 @@ def cmd_chat(args): _apply_safe_mode(args) _apply_user_config_bypass(args) _guard_noninteractive_user_config(args) - use_tui = _resolve_use_tui(args) + from hermes_cli.stream_json import stream_json_requested + # Structured stdout is a non-interactive protocol: it overrides HERMES_TUI/display.interface too. + use_tui = False if stream_json_requested(args) else _resolve_use_tui(args) _resolve_chat_session_args(args, use_tui) @@ -1739,6 +1741,7 @@ def cmd_chat(args): "query": args.query, "oneshot": bool(getattr(args, "oneshot_exit", False)), "run_budget": getattr(args, "run_budget", None), + "output_format": getattr(args, "output_format", "text"), "ignore_rules": getattr(args, "ignore_rules", False) or safe_mode, "ignore_user_config": getattr(args, "ignore_user_config", False) or safe_mode, "compact": getattr(args, "compact", False), diff --git a/hermes_cli/stream_json.py b/hermes_cli/stream_json.py new file mode 100644 index 0000000000..6fc85227c6 --- /dev/null +++ b/hermes_cli/stream_json.py @@ -0,0 +1,97 @@ +"""``hermes chat -q … --format stream-json``: one JSON object per stdout line. + +CI runners and orchestrators consume a one-shot run without scraping human-formatted text: +``system/init`` → ``text`` deltas / ``tool_use`` / ``tool_result`` → one terminal ``result`` +envelope (exit code, final text, token stats). Diagnostics and ``session_id`` stay on stderr. +""" + +from __future__ import annotations + +import json +import sys +import time +from typing import Any + +_TOOL_OUTPUT_CAP = 5000 + + +def stream_json_requested(args) -> bool: + """True when ``--format stream-json`` was passed; exits 2 on the combinations the protocol forbids + (no query to answer, or the interactive TUI transport) and forces quiet mode on ``args``.""" + if getattr(args, "output_format", "text") != "stream-json": + return False + if not (getattr(args, "query", None) or getattr(args, "query_file", None)): + print("Error: --format stream-json requires -q/--query.", file=sys.stderr) + raise SystemExit(2) + if getattr(args, "tui", False): + print("Error: --format stream-json cannot be used with --tui.", file=sys.stderr) + raise SystemExit(2) + args.quiet = True + return True + + +def _now_ms() -> int: + return int(time.time() * 1000) + + +class StreamJsonEmitter: + """Agent-callback sink that writes JSONL events to stdout and flushes each line.""" + + def __init__(self, model: str = "", session_id: str = ""): + self._session_id = session_id + self._start = time.time() + self._tool_started: dict[str, float] = {} + self._emit({"type": "system", "subtype": "init", "model": model, "session_id": session_id}) + + @classmethod + def attach(cls, agent, *, session_id: str = "") -> "StreamJsonEmitter": + """Emit ``init`` and route the agent's streaming/tool callbacks into this emitter.""" + emitter = cls(model=getattr(agent, "model", "") or "", session_id=session_id) + agent.stream_delta_callback = emitter.on_text_delta + agent.tool_progress_callback = emitter.on_tool_progress + return emitter + + def on_text_delta(self, text: str | None) -> None: + if text and str(text).strip(): + self._emit({"type": "text", "text": text}) + + def on_tool_progress(self, event_type: str, tool_name: str | None = None, preview: Any = None, args: Any = None, + **kwargs: Any) -> None: + """``tool.started`` → ``tool_use`` (with ``input`` when the args are a dict); ``tool.completed`` → + ``tool_result``. Other progress events (reasoning, output risk) are not part of the protocol.""" + name = tool_name or "unknown" + if event_type == "tool.started": + self._tool_started[name] = time.time() + payload: dict[str, Any] = {"type": "tool_use", "name": name} + if isinstance(args, dict): + payload["input"] = args + self._emit(payload) + elif event_type == "tool.completed": + duration = kwargs.get("duration") or (time.time() - self._tool_started.pop(name, time.time())) + output = str(kwargs.get("result") or "") + self._emit({"type": "tool_result", "name": name, + "output": output if len(output) <= _TOOL_OUTPUT_CAP else output[:_TOOL_OUTPUT_CAP] + "...", + "duration_ms": int(float(duration) * 1000), "is_error": bool(kwargs.get("is_error", False))}) + + def emit_result(self, result: Any, session_id: str = "", exit_code: int = 0) -> int: + """Write the terminal ``result`` record (once) and return the process exit code it reports.""" + data = result if isinstance(result, dict) else {"final_response": "" if result is None else str(result)} + exit_code = exit_code or (1 if data.get("failed") else 0) + payload = {"type": "result", "session_id": session_id or self._session_id, "exit_code": exit_code, + "text": data.get("final_response") or "", + "tokens": {"input": data.get("input_tokens") or 0, "output": data.get("output_tokens") or 0, + "total": data.get("total_tokens") or 0, "cache_read": data.get("cache_read_tokens") or 0, + "cache_write": data.get("cache_write_tokens") or 0}, + "duration_ms": int((time.time() - self._start) * 1000)} + if data.get("error"): + payload["error"] = str(data["error"]) + self._emit(payload) + print(f"\nsession_id: {session_id or self._session_id}", file=sys.stderr) # same stderr contract as -Q + return exit_code + + def _emit(self, obj: dict) -> None: + try: + sys.stdout.write(json.dumps({**obj, "timestamp": _now_ms()}, ensure_ascii=False) + "\n") + sys.stdout.flush() + except (BrokenPipeError, OSError): + pass # consumer closed the pipe — nothing left to report to diff --git a/tests/hermes_cli/test_stream_json.py b/tests/hermes_cli/test_stream_json.py new file mode 100644 index 0000000000..9327f26a15 --- /dev/null +++ b/tests/hermes_cli/test_stream_json.py @@ -0,0 +1,132 @@ +"""``hermes chat -q … --format stream-json`` emits a parseable JSONL event stream and nothing else on stdout.""" + +import json +import signal + +import pytest + +from hermes_cli.stream_json import StreamJsonEmitter + + +def _events(capsys): + out = capsys.readouterr().out + return [json.loads(line) for line in out.splitlines() if line] + + +def test_emitter_event_stream_is_valid_jsonl(capsys): + emitter = StreamJsonEmitter(model="test-model", session_id="s-1") + emitter.on_text_delta("hel") + emitter.on_text_delta(" ") # whitespace-only deltas carry no information + emitter.on_text_delta(None) # the turn-end sentinel the agent sends + emitter.on_tool_progress("tool.started", "read_file", "preview", {"path": "x"}) + emitter.on_tool_progress("reasoning.available", "_thinking", "hmm", None) # not part of the protocol + emitter.on_tool_progress("tool.completed", "read_file", None, None, duration=0.5, is_error=False, result="x" * 6000) + code = emitter.emit_result({"final_response": "", "failed": True, "error": "boom", "input_tokens": 3}, exit_code=0) + + events = _events(capsys) + assert [e["type"] for e in events] == ["system", "text", "tool_use", "tool_result", "result"] + assert events[0]["subtype"] == "init" and events[0]["model"] == "test-model" + assert events[2]["input"] == {"path": "x"} + assert events[3]["duration_ms"] == 500 and events[3]["output"].endswith("...") and len(events[3]["output"]) == 5003 + assert code == 1 and events[-1] == {**events[-1], "exit_code": 1, "error": "boom", "session_id": "s-1"} + assert events[-1]["tokens"]["input"] == 3 + assert all("timestamp" in e for e in events) + + +def _run_stream_json_chat(monkeypatch, capsys, run_conversation): + """parser → cmd_chat → cli.main → quiet single-query path with a deterministic fake agent.""" + import cli + import hermes_cli.main as cli_entry + from hermes_cli._parser import build_top_level_parser + + class FakeAgent: + model = "test-model" + session_id = "session-123" + + def run_conversation(self, **_kwargs): + return run_conversation(self) + + class FakeCLI: + def __init__(self, **_kwargs): + self.session_id = "session-123" + self.conversation_history = [] + self.agent = None + self._active_agent_route_signature = None + self.tool_progress_mode = None + + def _claim_active_session(self, *_a, **_k): + return True + + def _ensure_runtime_credentials(self): + return True + + def _resolve_turn_agent_config(self, _query): + return {"signature": "r", "model": None, "runtime": None, "request_overrides": None} + + def _init_agent(self, **_kwargs): + self.agent = FakeAgent() + return True + + def chat(self, _query, images=None): + print("human output") # must never reach stdout under stream-json + + monkeypatch.setattr(cli, "HermesCLI", FakeCLI) + monkeypatch.setattr(cli, "_finalize_single_query", lambda _cli: None) + monkeypatch.setattr(cli, "_emit_interrupted_session_end", lambda *_a, **_k: None) + monkeypatch.setattr(cli, "_start_worktree_setup", lambda *_a, **_k: None) + monkeypatch.setattr(cli.atexit, "register", lambda *_a, **_k: None) + monkeypatch.setattr(signal, "signal", lambda *_a, **_k: None) + monkeypatch.setattr(cli_entry, "_resolve_use_tui", lambda _args: pytest.fail("TUI resolution consulted")) + monkeypatch.setattr(cli_entry, "_has_any_provider_configured", lambda: True) + monkeypatch.setattr(cli_entry, "_start_chat_background_prefetch", lambda: None) + monkeypatch.setattr(cli_entry, "_pin_kanban_board_env", lambda: None) + monkeypatch.setattr(cli_entry, "_confirm_startup_expensive_model_override", lambda _a: None) + monkeypatch.setattr(cli_entry, "_warn_retired_xai_models", lambda: None) + monkeypatch.setattr("hermes_cli.free_tier_bootstrap.run_bootstrap", lambda **_k: None) + monkeypatch.setattr("hermes_cli.quiet_single_query.continue_quiet_notify_completions", lambda *_a, **_k: None) + + parser, _, _ = build_top_level_parser() + args = parser.parse_args(["chat", "-q", "hello", "--format", "stream-json"]) + with pytest.raises(SystemExit) as exc_info: + cli_entry.cmd_chat(args) + return exc_info.value.code, _events(capsys) + + +def _ok_turn(agent): + agent.stream_delta_callback("hello") + agent.tool_progress_callback("tool.started", "read_file", "p", {"path": "f"}) + agent.tool_progress_callback("tool.completed", "read_file", None, None, duration=0.01, result="contents") + return {"final_response": "hello", "failed": False} + + +def _interrupted_turn(_agent): + raise KeyboardInterrupt + + +@pytest.mark.parametrize("turn, exit_code, types", [ + (_ok_turn, 0, ["system", "text", "tool_use", "tool_result", "result"]), + (_interrupted_turn, 130, ["system", "result"]), +]) +def test_chat_stream_json_implies_quiet_and_closes_with_result(monkeypatch, capsys, turn, exit_code, types): + """No ``-Q`` needed; stdout is only JSONL; the stream always ends in a ``result`` carrying the exit code.""" + code, events = _run_stream_json_chat(monkeypatch, capsys, turn) + assert code == exit_code + assert [e["type"] for e in events] == types + assert events[-1]["exit_code"] == exit_code and events[-1]["session_id"] == "session-123" + + +@pytest.mark.parametrize("argv, message", [ + (["chat", "--format", "stream-json"], "requires -q/--query"), + (["chat", "-q", "hi", "--format", "stream-json", "--tui"], "cannot be used with --tui"), + (["--tui", "chat", "-q", "hi", "--format", "stream-json"], "cannot be used with --tui"), +]) +def test_chat_stream_json_rejects_interactive_combinations(monkeypatch, capsys, argv, message): + import hermes_cli.main as cli_entry + from hermes_cli._parser import build_top_level_parser + + monkeypatch.setattr(cli_entry, "_launch_tui", lambda *_a, **_k: pytest.fail("TUI launched")) + parser, _, _ = build_top_level_parser() + with pytest.raises(SystemExit) as exc_info: + cli_entry.cmd_chat(parser.parse_args(argv)) + assert exc_info.value.code == 2 + assert message in capsys.readouterr().err diff --git a/website/docs/reference/cli-commands.md b/website/docs/reference/cli-commands.md index 709218d382..14cce22769 100644 --- a/website/docs/reference/cli-commands.md +++ b/website/docs/reference/cli-commands.md @@ -121,6 +121,7 @@ Common options: | `-s`, `--skills ` | Preload one or more skills for the session (can be repeated or comma-separated). | | `-v`, `--verbose` | Verbose output. | | `-Q`, `--quiet` | Programmatic mode: suppress banner/spinner/tool previews. | +| `--format stream-json` | Emit structured JSONL for a `-q` / `--query` invocation. Implies `--quiet`; cannot be combined with `--tui`. | | `--image ` | Attach a local image to a single query. | | `--resume ` / `--continue [name]` | Resume a session directly from `chat`. | | `--worktree` | Create an isolated git worktree for this run. | @@ -142,11 +143,37 @@ hermes chat --oneshot -q "Summarize the latest PRs" # answer and exit hermes chat --provider openrouter --model anthropic/claude-sonnet-4.6 hermes chat --toolsets web,terminal,skills hermes chat --quiet -q "Return only JSON" +hermes chat -q "Inspect this repository" --format stream-json hermes chat --worktree -q "Review this repo and open a PR" hermes chat --ignore-user-config --ignore-rules -q "Repro without my personal setup" hermes chat --safe-mode -q "Is this bug mine or Hermes'?" ``` +### `--format stream-json` — structured JSONL output + +Use `--format stream-json` when a program needs to consume progress without +scraping terminal output. It requires `-q` / `--query` (or `--query-file`), implies +quiet non-interactive CLI mode, and rejects an explicit `--tui` request. Every +stdout line is one JSON object; diagnostics and the `session_id:` line stay on stderr. + +```bash +hermes chat -q "Summarize this repository" --format stream-json +``` + +Every event carries `timestamp` (Unix epoch milliseconds). + +| Event `type` | Fields | +|---|---| +| `system` | `subtype: "init"`, `model`, `session_id` | +| `text` | `text` — a streamed assistant text delta | +| `tool_use` | `name`; `input` when the tool arguments are available | +| `tool_result` | `name`, `output` (capped at 5000 chars), `duration_ms`, `is_error` | +| `result` | `session_id`, `exit_code`, `text`, `tokens` (`input`, `output`, `total`, `cache_read`, `cache_write`), `duration_ms`; `error` when the turn failed | + +Once a conversation starts, its terminal record is always `result` — including +`exit_code: 130` when it is interrupted with Ctrl-C. Treat that record as the +completion signal; the process exit code matches its `exit_code`. + #### Delegation in finite chat runs When chat answers and exits (`-Q`, `chat --oneshot`, or a query with non-TTY