docs(skills): regenerate skills catalog; fix generator sentence truncation

- Fix generate-skill-docs.py short-description truncation: split on
  sentence boundary (dot + space/end) instead of the first dot, which
  mangled descriptions containing dotted paths ('.hermes/plans/') or
  abbreviations.
- Regenerate all 181 skill pages + both catalog indexes + sidebar:
  adds missing pages for inspecting-hermes-desktop-dom, tldraw-offline
  (fixes broken catalog link), and pinecone-research.
- Delete orphaned kanban-codex-lane page (skill removed in #39028).
This commit is contained in:
Teknium
2026-07-29 00:54:22 -07:00
parent c19fd5c505
commit 355b376225
53 changed files with 1604 additions and 904 deletions
@@ -159,7 +159,7 @@ hermes skills uninstall <skill-name>
| [**obliteratus**](/docs/user-guide/skills/optional/mlops/mlops-obliteratus) | OBLITERATUS: abliterate LLM refusals (diff-in-means). |
| [**outlines**](/docs/user-guide/skills/optional/mlops/mlops-inference-outlines) | Outlines: structured JSON/regex/Pydantic LLM generation. |
| [**peft-fine-tuning**](/docs/user-guide/skills/optional/mlops/mlops-peft) | Fine-tune large LLMs with LoRA on limited GPU memory. |
| [**pinecone**](/docs/user-guide/skills/optional/mlops/mlops-pinecone) | Managed vector database for production AI applications. Fully managed, auto-scaling, with hybrid search (dense + sparse), metadata filtering, and namespaces. Low latency (&lt;100ms p95). Use for production RAG, recommendation systems, or se... |
| [**pinecone**](/docs/user-guide/skills/optional/mlops/mlops-pinecone) | Managed vector DB for production RAG and search. |
| [**pytorch-fsdp**](/docs/user-guide/skills/optional/mlops/mlops-pytorch-fsdp) | Fully sharded data-parallel training for large models. |
| [**pytorch-lightning**](/docs/user-guide/skills/optional/mlops/mlops-pytorch-lightning) | Clean training loops with built-in distributed support. |
| [**qdrant-vector-search**](/docs/user-guide/skills/optional/mlops/mlops-qdrant) | Vector search engine for production RAG systems. |
@@ -170,7 +170,7 @@ hermes skills uninstall <skill-name>
| [**stable-diffusion-image-generation**](/docs/user-guide/skills/optional/mlops/mlops-stable-diffusion) | Text-to-image generation, inpainting, and img2img. |
| [**tensorrt-llm**](/docs/user-guide/skills/optional/mlops/mlops-tensorrt-llm) | High-throughput LLM inference on NVIDIA GPUs. |
| [**distributed-llm-pretraining-torchtitan**](/docs/user-guide/skills/optional/mlops/mlops-torchtitan) | Pretrain LLMs at scale with PyTorch 4D parallelism. |
| [**fine-tuning-with-trl**](/docs/user-guide/skills/optional/mlops/mlops-training-trl-fine-tuning) | TRL: SFT, DPO, PPO, GRPO, reward modeling for LLM RLHF. |
| [**fine-tuning-with-trl**](/docs/user-guide/skills/optional/mlops/mlops-training-trl-fine-tuning) | TRL: SFT, DPO, GRPO, RLOO reward modeling for LLM RLHF. |
| [**unsloth**](/docs/user-guide/skills/optional/mlops/mlops-training-unsloth) | Unsloth: 2-5x faster LoRA/QLoRA fine-tuning, less VRAM. |
| [**whisper**](/docs/user-guide/skills/optional/mlops/mlops-whisper) | Transcribe and translate speech in 99 languages. |
@@ -206,6 +206,7 @@ hermes skills uninstall <skill-name>
| [**gitnexus-explorer**](/docs/user-guide/skills/optional/research/research-gitnexus-explorer) | Serve an interactive codebase knowledge graph web UI. |
| [**osint-investigation**](/docs/user-guide/skills/optional/research/research-osint-investigation) | Follow the money via public records and sanctions data. |
| [**parallel-cli**](/docs/user-guide/skills/optional/research/research-parallel-cli) | Agent-native web search, deep research, and enrichment. |
| [**pinecone-research**](/docs/user-guide/skills/optional/research/research-pinecone-research) | Agent RAG and long-term memory with Pinecone. |
| [**qmd**](/docs/user-guide/skills/optional/research/research-qmd) | Hybrid local search over notes, docs, and transcripts. |
| [**scrapling**](/docs/user-guide/skills/optional/research/research-scrapling) | Scrape sites with stealth browsing and Cloudflare bypass. |
| [**searxng-search**](/docs/user-guide/skills/optional/research/research-searxng-search) | Free keyless meta-search aggregating 70+ engines. |
+2 -1
View File
@@ -81,9 +81,9 @@ If a skill is missing from this list but present in the repo, the catalog is reg
| Skill | Description | Path |
|-------|-------------|------|
| [`evaluating-llms-harness`](/docs/user-guide/skills/bundled/mlops/mlops-evaluation-evaluating-llms-harness) | lm-eval-harness: benchmark LLMs (MMLU, GSM8K, etc.). | `mlops/evaluation/evaluating-llms-harness` |
| [`huggingface-hub`](/docs/user-guide/skills/bundled/mlops/mlops-huggingface-hub) | HuggingFace hf CLI: search/download/upload models, datasets. | `mlops/huggingface-hub` |
| [`llama-cpp`](/docs/user-guide/skills/bundled/mlops/mlops-inference-llama-cpp) | llama.cpp local GGUF inference + HF Hub model discovery. | `mlops/inference/llama-cpp` |
| [`evaluating-llms-harness`](/docs/user-guide/skills/bundled/mlops/mlops-evaluation-evaluating-llms-harness) | lm-eval-harness: benchmark LLMs (MMLU, GSM8K, etc.). | `mlops/evaluation/evaluating-llms-harness` |
| [`serving-llms-vllm`](/docs/user-guide/skills/bundled/mlops/mlops-inference-serving-llms-vllm) | vLLM: high-throughput LLM serving, OpenAI API, quantization. | `mlops/inference/serving-llms-vllm` |
| [`weights-and-biases`](/docs/user-guide/skills/bundled/mlops/mlops-evaluation-weights-and-biases) | W&B: log ML experiments, sweeps, model registry, dashboards. | `mlops/evaluation/weights-and-biases` |
@@ -137,6 +137,7 @@ If a skill is missing from this list but present in the repo, the catalog is reg
|-------|-------------|------|
| [`dogfood`](/docs/user-guide/skills/bundled/software-development/software-development-dogfood) | Exploratory QA of web apps: find bugs, evidence, reports. | `software-development/dogfood` |
| [`hermes-agent-skill-authoring`](/docs/user-guide/skills/bundled/software-development/software-development-hermes-agent-skill-authoring) | Author in-repo SKILL.md files: frontmatter and structure. | `software-development/hermes-agent-skill-authoring` |
| [`inspecting-hermes-desktop-dom`](/docs/user-guide/skills/bundled/software-development/software-development-inspecting-hermes-desktop-dom) | Read the live Hermes desktop DOM/CSS over CDP. | `software-development/inspecting-hermes-desktop-dom` |
| [`node-inspect-debugger`](/docs/user-guide/skills/bundled/software-development/software-development-node-inspect-debugger) | Debug Node.js via --inspect + Chrome DevTools Protocol CLI. | `software-development/node-inspect-debugger` |
| [`plan`](/docs/user-guide/skills/bundled/software-development/software-development-plan) | Write a markdown plan to .hermes/plans/; no execution. | `software-development/plan` |
| [`python-debugpy`](/docs/user-guide/skills/bundled/software-development/software-development-python-debugpy) | Debug Python: pdb REPL + debugpy remote (DAP). | `software-development/python-debugpy` |
@@ -16,7 +16,7 @@ Manage Apple Notes via memo CLI: create, search, edit.
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/apple/apple-notes` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Hermes Agent |
| License | MIT |
| Platforms | macos |
@@ -65,10 +65,14 @@ memo notes -s "query" # Search notes (fuzzy)
### Create Notes
```bash
memo notes -a # Interactive editor
memo notes -a "Note Title" # Quick add with title
memo notes -a # Add a note (opens your $EDITOR)
memo notes -a -f "Folder Name" # Add a note into a specific folder
```
`-a`/`--add` is a bare flag — it opens your `$EDITOR` to compose the note; it does
not take a title argument. Use `-f/--folder` to target a folder. Set `$EDITOR`
first (e.g. `export EDITOR=vim`).
### Edit Notes
```bash
@@ -1,7 +1,7 @@
---
title: "Findmy — Track Apple devices/AirTags via FindMy"
title: "Findmy — Track Apple devices/AirTags via FindMy.app on macOS"
sidebar_label: "Findmy"
description: "Track Apple devices/AirTags via FindMy"
description: "Track Apple devices/AirTags via FindMy.app on macOS"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -16,7 +16,7 @@ Delegate coding to Claude Code CLI (features, PRs).
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/autonomous-ai-agents/claude-code` |
| Version | `2.2.0` |
| Version | `2.2.1` |
| Author | Hermes Agent + Teknium |
| License | MIT |
| Platforms | linux, macos, windows |
@@ -283,7 +283,7 @@ Automatically falls back to the specified model when the default is overloaded (
| Flag | Effect |
|------|--------|
| `--model <alias>` | Model selection: `sonnet`, `opus`, `haiku`, or full name like `claude-sonnet-4-6` |
| `--effort <level>` | Reasoning depth: `low`, `medium`, `high`, `max`, `auto` | Both |
| `--effort <level>` | Reasoning depth: `low`, `medium`, `high`, `xhigh`, `max` |
| `--max-turns <n>` | Limit agentic loops (print mode only; prevents runaway) |
| `--max-budget-usd <n>` | Cap API spend in dollars (print mode only) |
| `--fallback-model <model>` | Auto-fallback when default model is overloaded (print mode only) |
@@ -407,7 +407,7 @@ Use the `#` prefix in interactive mode to quickly add to memory: `# Always use 2
| Command | Purpose |
|---------|---------|
| `/model [model]` | Switch models mid-session (use arrow keys to adjust effort) |
| `/effort [level]` | Set reasoning effort: `low`, `medium`, `high`, `max`, or `auto` |
| `/effort [level]` | Set reasoning effort: `low`, `medium`, `high`, `xhigh`, or `max` |
| `/init` | Create a CLAUDE.md file for project memory |
| `/memory` | Open CLAUDE.md for editing |
| `/config` | Open interactive settings configuration |
@@ -16,7 +16,7 @@ Delegate coding to OpenAI Codex CLI (features, PRs).
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/autonomous-ai-agents/codex` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Hermes Agent |
| License | MIT |
| Platforms | linux, macos, windows |
@@ -71,7 +71,7 @@ terminal(command="cd $(mktemp -d) && git init && codex exec 'Build a snake game
```
# Start in background with PTY
terminal(command="codex exec --full-auto 'Refactor the auth module'", workdir="~/project", background=true, pty=true)
terminal(command="codex exec --sandbox workspace-write 'Refactor the auth module'", workdir="~/project", background=true, pty=true)
# Returns session_id
# Monitor progress
@@ -90,10 +90,12 @@ process(action="kill", session_id="<id>")
| Flag | Effect |
|------|--------|
| `exec "prompt"` | One-shot execution, exits when done |
| `--full-auto` | Sandboxed but auto-approves file changes in workspace |
| `--yolo` | No sandbox, no approvals (fastest, most dangerous) |
| `--sandbox workspace-write` (`-s`) | Sandboxed but auto-approves file changes in the workspace (the recommended auto-build mode) |
| `--dangerously-bypass-approvals-and-sandbox` | No sandbox, no approvals (fastest, most dangerous; `--yolo` still works as a hidden alias) |
| `--sandbox danger-full-access` | No Codex sandbox; useful when the host service context breaks bubblewrap |
> **Deprecated:** `--full-auto` still works but the live CLI warns to use `--sandbox workspace-write` instead.
## Hermes Gateway Caveat
When invoking the Codex CLI from a Hermes gateway/service context (for example,
@@ -128,8 +130,8 @@ terminal(command="git worktree add -b fix/issue-78 /tmp/issue-78 main", workdir=
terminal(command="git worktree add -b fix/issue-99 /tmp/issue-99 main", workdir="~/project")
# Launch Codex in each
terminal(command="codex --yolo exec 'Fix issue #78: <description>. Commit when done.'", workdir="/tmp/issue-78", background=true, pty=true)
terminal(command="codex --yolo exec 'Fix issue #99: <description>. Commit when done.'", workdir="/tmp/issue-99", background=true, pty=true)
terminal(command="codex --sandbox workspace-write exec 'Fix issue #78: <description>. Commit when done.'", workdir="/tmp/issue-78", background=true, pty=true)
terminal(command="codex --sandbox workspace-write exec 'Fix issue #99: <description>. Commit when done.'", workdir="/tmp/issue-99", background=true, pty=true)
# Monitor
process(action="list")
@@ -161,7 +163,7 @@ terminal(command="gh pr comment 86 --body '<review>'", workdir="~/project")
1. **Always use `pty=true`** — Codex is an interactive terminal app and hangs without a PTY
2. **Git repo required** — Codex won't run outside a git directory. Use `mktemp -d && git init` for scratch
3. **Use `exec` for one-shots** — `codex exec "prompt"` runs and exits cleanly
4. **`--full-auto` for building** — auto-approves changes within the sandbox
4. **`--sandbox workspace-write` for building** — auto-approves changes within the sandbox (`--full-auto` is deprecated for this)
5. **Background for long tasks** — use `background=true` and monitor with `process` tool
6. **Don't interfere** — monitor with `poll`/`log`, be patient with long-running tasks
7. **Parallel is fine** — run multiple Codex processes at once for batch work
@@ -1,295 +0,0 @@
---
title: "Kanban Codex Lane"
sidebar_label: "Kanban Codex Lane"
description: "Use when a Hermes Kanban worker wants to run Codex CLI as an isolated implementation lane while Hermes keeps ownership of task lifecycle, reconciliation, tes..."
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
# Kanban Codex Lane
Use when a Hermes Kanban worker wants to run Codex CLI as an isolated implementation lane while Hermes keeps ownership of task lifecycle, reconciliation, testing, and handoff.
## Skill metadata
| | |
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/autonomous-ai-agents/kanban-codex-lane` |
| Version | `1.0.0` |
| Author | Hermes Agent |
| License | MIT |
| Tags | `kanban`, `codex`, `worktrees`, `autonomous-agents`, `prediction-market-bot` |
| Related skills | [`codex`](/docs/user-guide/skills/bundled/autonomous-ai-agents/autonomous-ai-agents-codex), [`hermes-agent`](/docs/user-guide/skills/bundled/autonomous-ai-agents/autonomous-ai-agents-hermes-agent) |
## Reference: full SKILL.md
:::info
The following is the complete skill definition that Hermes loads when this skill is triggered. This is what the agent sees as instructions when the skill is active.
:::
# Kanban Codex Lane
## Overview
This skill defines the lightweight Hermes+Codex dual-lane convention for Kanban workers. Hermes is always the task owner: it calls `kanban_show`, decides whether Codex is appropriate, creates or selects an isolated workspace, starts and monitors Codex, reconciles any diff, runs verification, and writes the final `kanban_complete` or `kanban_block` handoff. Codex is an input lane only. Codex output is not a task completion signal, not a trusted reviewer, and not allowed to write durable Kanban state directly.
The convention exists so a Hermes worker can use Codex for bounded implementation help without changing the dispatcher. The dispatcher must still spawn Hermes workers. A worker may optionally spawn Codex inside its own run, then accept, partially accept, or reject the lane after independent review and tests.
## When to Use
Use the Codex lane when all of these are true:
- The Kanban task is a coding, refactor, documentation, test, or mechanical migration task with clear acceptance criteria.
- A bounded diff can be evaluated by Hermes in one run.
- The repo can be copied or checked out in an isolated git worktree/branch.
- Hermes can run the relevant tests itself after Codex exits.
- The prompt can state all safety constraints and files that must not change.
Do not use the Codex lane when any of these are true:
- The task requires human judgment that is not already captured in the Kanban body.
- The worker lacks repo access, Codex auth, or time to reconcile the result.
- The change touches secrets, credential stores, private user data, or production order-entry systems.
- A small direct edit is faster and safer than spawning another agent.
- The task is research-only and should produce a written handoff rather than a diff.
- The worker would be tempted to mark Done based only on Codex self-report.
## Ownership Rules
1. Hermes owns the Kanban lifecycle. Codex must never call `kanban_complete`, `kanban_block`, `kanban_create`, gateway messaging, or any Hermes board CLI as a substitute for the worker.
2. Hermes owns final acceptance. Treat Codex commits/diffs as untrusted patches until reviewed and verified.
3. Hermes owns test execution. Codex may run tests, but those runs are advisory; repeat required verification from Hermes with the repo's canonical wrapper.
4. Hermes owns safety. If Codex changes safety boundaries, risk gates, live trading behavior, or secrets handling, reject the lane even if tests pass.
5. Hermes owns cleanup. Kill stuck Codex processes and remove temporary worktrees when they are no longer needed.
## Required Worktree and Branch Pattern
Never run Codex directly in a shared dirty checkout. Use a branch/worktree name that ties the lane to the Kanban task and keeps untrusted edits isolated.
Recommended variables:
```bash
TASK_ID="${HERMES_KANBAN_TASK:-t_manual}"
REPO="/path/to/repo"
BASE="$(git -C "$REPO" rev-parse --abbrev-ref HEAD)"
SAFE_TASK="$(printf '%s' "$TASK_ID" | tr -cd '[:alnum:]_-')"
BRANCH="codex/${SAFE_TASK}/$(date -u +%Y%m%d%H%M%S)"
WORKTREE="/tmp/${SAFE_TASK}-codex-lane"
```
Create the isolated lane:
```bash
git -C "$REPO" fetch --all --prune
git -C "$REPO" worktree add -b "$BRANCH" "$WORKTREE" "$BASE"
git -C "$WORKTREE" status --short --branch
```
If the current Kanban workspace is already an isolated git worktree created for this task, you may create a sibling Codex branch inside it only if `git status --short` is clean except for intentional Hermes edits. Otherwise create a separate temporary worktree and cherry-pick or copy accepted commits back after reconciliation.
Cleanup after reconciliation:
```bash
git -C "$REPO" worktree remove "$WORKTREE"
git -C "$REPO" branch -D "$BRANCH" # only after accepted commits were copied/cherry-picked or intentionally rejected
```
Keep the worktree if it is needed as an artifact for review; record it in `codex_lane.artifacts` and mention it in the handoff.
## Codex Capability Checks
Run these before spawning Codex. Missing Codex is a normal reason to skip the lane, not a task blocker if Hermes can do the task directly.
```bash
command -v codex
codex --version
codex features list | grep -i goals || true
```
If `/goal` support is required, enable or launch with the feature flag only after checking availability:
```bash
codex features enable goals || true
codex --enable goals --version
```
Authentication can be via `OPENAI_API_KEY` or the Codex CLI OAuth state (often `~/.codex/auth.json`). Do not print token files. A missing `OPENAI_API_KEY` is not proof that auth is unavailable.
## Mode Selection
Use `codex exec` for bounded one-shot edits where Codex should exit on its own:
```python
terminal(
command="codex exec --full-auto '$(cat /tmp/codex_prompt.md)'",
workdir=WORKTREE,
background=True,
pty=True,
notify_on_complete=True,
)
```
Use Codex `/goal` only for broader multi-step work that benefits from durable objective tracking. Launch interactively in a PTY/tmux session or with `codex --enable goals` if the feature is disabled by default. Keep the goal objective self-contained: repo path, task id, safety constraints, allowed scope, acceptance criteria, tests, and commit expectations.
Example `/goal` objective text to paste into Codex:
```text
/goal Work in this repository only: <WORKTREE>. Task: <TASK_ID> <TITLE>.
Hermes owns the Kanban lifecycle; do not call Hermes kanban tools or messaging.
Create small commits on branch <BRANCH>. Follow the PMB safety constraints in the prompt.
Run the requested verification commands and report exact outputs. Stop after producing a diff and summary.
```
Do not use `--yolo` for prediction-market-bot or safety-sensitive repos. Prefer `--full-auto` inside the isolated worktree, then rely on Hermes reconciliation.
## Prompt Construction
Use the linked template at `templates/pmb-codex-lane-prompt.md` for prediction-market-bot work. For other repos, keep the same structure and replace the PMB-specific safety block with repo-specific invariants.
Every Codex prompt must include:
- `task_id`, title, and full Kanban acceptance criteria.
- Repo path, worktree path, branch name, and allowed file scope.
- Explicit statement: Hermes owns Kanban lifecycle; Codex is an input lane only.
- Required output: concise summary, files changed, commits, tests run, and known risks.
- Prohibited actions: secrets access, external messaging, board mutation, unrelated refactors, dependency upgrades unless required.
- Verification commands Codex may run and commands Hermes will run afterward.
For PMB, include these mandatory safety constraints verbatim:
```text
PMB safety constraints:
- live-SIM is paper-only; do not add or enable live REST order entry.
- Never use market orders.
- Do not add execution crossing or bypass price/risk checks.
- Do not fake passive fills, fills, PnL, order states, or reconciliation evidence.
- Do not weaken risk gates, limits, kill switches, or fail-closed behavior.
- Keep research/selection outside the C++ hot path unless explicitly requested.
- Do not read, print, write, or require secrets/tokens/credentials.
```
## Monitoring, Timeout, and Kill Behavior
Start long Codex lanes in the background with PTY and completion notification:
```python
result = terminal(
command="codex exec --full-auto '$(cat /tmp/codex_prompt.md)'",
workdir=WORKTREE,
background=True,
pty=True,
notify_on_complete=True,
)
session_id = result["session_id"]
```
Monitor without interfering:
```python
process(action="poll", session_id=session_id)
process(action="log", session_id=session_id, limit=200)
process(action="wait", session_id=session_id, timeout=300)
```
Send a Kanban heartbeat every few minutes for lanes longer than two minutes, e.g. `kanban_heartbeat(note="Codex lane running in <WORKTREE>; waiting for tests/diff")`.
Kill conditions:
- No useful output for the task's remaining runtime budget.
- Codex requests secrets, production credentials, or external permissions.
- Codex attempts to modify files outside the worktree.
- Codex starts unrelated rewrites or dependency churn.
- Codex is still running near the worker timeout and no safe partial artifact exists.
Kill command:
```python
process(action="kill", session_id=session_id)
```
After kill, inspect `git status --short`, preserve useful patches only if safe, and record `codex_lane.result: timed_out` or `rejected` with a concrete `rejected_reason`.
## Reconciliation Checklist
Hermes must perform this checklist before accepting any Codex lane result:
- [ ] `git -C <WORKTREE> status --short --branch` shows only expected files.
- [ ] `git -C <WORKTREE> diff --stat` and `git diff` were reviewed by Hermes.
- [ ] No secrets, credentials, generated caches, unrelated data, or local artifacts are included.
- [ ] PMB safety constraints were preserved: no live REST order entry, no market orders, no execution crossing, no fake passive fills/PnL, no risk-gate weakening, no secrets.
- [ ] Codex commits are small enough to cherry-pick or squash cleanly.
- [ ] Hermes ran the canonical tests itself, using `scripts/run_tests.sh` for Hermes Agent or the repo's documented wrapper for other repos.
- [ ] Any Codex-run tests are listed separately from Hermes-run tests.
- [ ] Accepted commits/diffs were applied to the Hermes-owned workspace/branch.
- [ ] Rejected or partial work has a concrete reason and artifact path if useful.
Acceptance outcomes:
- `accepted`: Codex diff/commits were reviewed, applied, and verified.
- `partial`: Some Codex work was accepted after edits or cherry-picks; rejected parts are documented.
- `rejected`: No Codex changes were accepted; reason is documented.
- `timed_out`: Codex exceeded the lane budget; useful artifacts may or may not exist.
## kanban_complete Metadata Schema
Include this object under `metadata.codex_lane` for every task where the lane was considered. If Codex was not used, set `used: false` and explain why in `rejected_reason` or a sibling `notes` field.
```json
{
"codex_lane": {
"used": true,
"mode": "exec | goal | skipped",
"worktree": "/absolute/path/to/codex/worktree",
"branch": "codex/t_caa69668/20260508100000",
"command": "codex exec --full-auto ...",
"result": "accepted | rejected | partial | timed_out",
"accepted_commits": ["<sha1>", "<sha2>"],
"rejected_reason": "empty when fully accepted; otherwise concrete reason",
"tests_run": [
{"command": "scripts/run_tests.sh tests/tools/test_x.py", "exit_code": 0, "owner": "hermes"},
{"command": "codex-reported: npm test", "exit_code": 0, "owner": "codex"}
],
"artifacts": ["/absolute/path/to/log-or-patch"]
}
}
```
For tasks that intentionally skip Codex:
```json
{
"codex_lane": {
"used": false,
"mode": "skipped",
"worktree": null,
"branch": null,
"command": null,
"result": "rejected",
"accepted_commits": [],
"rejected_reason": "Direct Hermes edit was smaller and safer than spawning Codex.",
"tests_run": [],
"artifacts": []
}
}
```
## Common Pitfalls
1. Treating Codex self-report as verification. Always inspect the diff and rerun tests from Hermes.
2. Running Codex in the user's dirty main checkout. Always isolate in a worktree/branch.
3. Letting Codex own Kanban. Codex may summarize progress, but Hermes writes board state.
4. Forgetting PMB safety invariants in the prompt. Missing safety text is a lane setup failure.
5. Using `/goal` for quick edits. Prefer `codex exec` unless durable multi-step continuation is needed.
6. Killing a stuck lane without recording why. `rejected_reason` must explain the decision.
7. Accepting broad unrelated cleanup because tests pass. Reject or cherry-pick only the scoped changes.
## Verification Checklist
- [ ] Codex was skipped or started only after `command -v codex`, `codex --version`, and optional goals feature checks.
- [ ] Codex ran only in an isolated worktree/branch.
- [ ] Prompt included task scope, ownership rules, PMB safety constraints when applicable, and verification commands.
- [ ] Hermes reviewed `git diff` and safety-sensitive files.
- [ ] Hermes ran canonical tests independently.
- [ ] `kanban_complete.metadata.codex_lane` follows the schema above.
- [ ] Temporary processes and unnecessary worktrees were cleaned up.
@@ -1,7 +1,7 @@
---
title: "Design Md — Author/validate/export Google's DESIGN"
title: "Design Md — Author/validate/export Google's DESIGN.md token spec files"
sidebar_label: "Design Md"
description: "Author/validate/export Google's DESIGN"
description: "Author/validate/export Google's DESIGN.md token spec files"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -16,7 +16,7 @@ Hand-drawn Excalidraw JSON diagrams (arch, flow, seq).
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/creative/excalidraw` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Hermes Agent |
| License | MIT |
| Platforms | linux, macos, windows |
@@ -66,7 +66,7 @@ Save to any path, e.g. `~/diagrams/my_diagram.excalidraw`.
Run the upload script (located in this skill's `scripts/` directory) via terminal:
```bash
python skills/diagramming/excalidraw/scripts/upload.py ~/diagrams/my_diagram.excalidraw
python skills/creative/excalidraw/scripts/upload.py ~/diagrams/my_diagram.excalidraw
```
This uploads to excalidraw.com (no account needed) and prints a shareable URL. Requires the `cryptography` pip package (`pip install cryptography`).
@@ -44,27 +44,27 @@ Load this skill whenever the user asks to:
- match their voice in writing they're producing
- review text for AI tells before publishing
Also apply this skill to **your own** output when writing user-facing prose — release notes, PR descriptions, documentation, long-form explanations, summaries. Hermes's baseline voice already strips most of these, but a focused pass catches what slips through.
Also apply this skill to **your own** output when writing user-facing prose such as release notes, PR descriptions, docs, and summaries. Hermes's baseline voice already strips most of these, but a focused pass catches what slips through.
## How to use it in Hermes
The text usually arrives one of three ways:
1. **Inline** — user pastes the text directly into the message. Work on it in-place, reply with the rewrite.
2. **File** — user points at a file. Use `read_file` to load it, then `patch` or `write_file` to apply edits. For markdown docs in a repo, a targeted `patch` per section is cleaner than rewriting the whole file.
3. **Voice calibration sample** — user provides an additional sample of their own writing (inline or by file path) and asks you to match it. Read the sample first, then rewrite. See the Voice Calibration section below.
1. **Inline.** The user pastes the text into the message. Work on it in place and reply with the rewrite.
2. **File.** The user points at a file. Use `read_file` to load it, then `patch` or `write_file` to apply edits. For a markdown doc in a repo, a targeted `patch` per section is cleaner than rewriting the whole file.
3. **Voice calibration sample.** The user provides a sample of their own writing (inline or by file path) and asks you to match it. Read the sample first, then rewrite. See the Voice Calibration section below.
Always show the rewrite to the user. For file edits, show a diff or the changed section — don't silently overwrite.
Always show the rewrite to the user. For file edits, show a diff or the changed section instead of silently overwriting.
## Your task
When given text to humanize:
1. **Identify AI patterns** — scan for the 29 patterns listed below.
2. **Rewrite problematic sections** — replace AI-isms with natural alternatives.
3. **Preserve meaning** — keep the core message intact.
4. **Maintain voice** — match the intended tone (formal, casual, technical, etc.). If a voice sample was provided, match it specifically.
5. **Add soul** — don't just remove bad patterns, inject actual personality. See PERSONALITY AND SOUL below.
6. **Do a final anti-AI pass** — ask yourself: "What makes the below so obviously AI generated?" Answer briefly with any remaining tells, then revise one more time.
1. **Identify AI patterns.** Scan for the 34 patterns listed below.
2. **Rewrite problematic sections.** Replace AI-isms with natural alternatives.
3. **Preserve meaning.** Keep the core message intact.
4. **Maintain voice.** Match the intended tone (formal, casual, technical, and so on). If a voice sample was provided, match it specifically.
5. **Add soul.** Removing bad patterns is only half the job; the rewrite also needs real personality. See PERSONALITY AND SOUL below.
6. **Do a final anti-AI pass.** Ask yourself: "What makes the below so obviously AI generated?" Answer briefly with any remaining tells, then revise one more time.
## Voice Calibration (optional)
@@ -79,7 +79,7 @@ If the user provides a writing sample (their own previous writing), analyze it b
- Any recurring phrases or verbal tics
- How they handle transitions (explicit connectors? Just start the next point?)
2. **Match their voice in the rewrite.** Don't just remove AI patterns — replace them with patterns from the sample. If they write short sentences, don't produce long ones. If they use "stuff" and "things," don't upgrade to "elements" and "components."
2. **Match their voice in the rewrite.** Removing AI patterns is only half of it; swap in patterns from the sample as well. If they write short sentences, do not produce long ones. If they use "stuff" and "things," do not upgrade to "elements" and "components."
3. **When no sample is provided,** fall back to the default behavior (natural, varied, opinionated voice from the PERSONALITY AND SOUL section below).
@@ -102,23 +102,23 @@ Avoiding AI patterns is only half the job. Sterile, voiceless writing is just as
### How to add voice:
**Have opinions.** Don't just report facts — react to them. "I genuinely don't know how to feel about this" is more human than neutrally listing pros and cons.
**Have opinions.** Report the facts, then react to them. "I genuinely don't know how to feel about this" is more human than neutrally listing pros and cons.
**Vary your rhythm.** Short punchy sentences. Then longer ones that take their time getting where they're going. Mix it up.
**Acknowledge complexity.** Real humans have mixed feelings. "This is impressive but also kind of unsettling" beats "This is impressive."
**Use "I" when it fits.** First person isn't unprofessional — it's honest. "I keep coming back to..." or "Here's what gets me..." signals a real person thinking.
**Use "I" when it fits.** First person reads as honest and fits most prose. "I keep coming back to..." or "Here's what gets me..." signals a real person thinking.
**Let some mess in.** Perfect structure feels algorithmic. Tangents, asides, and half-formed thoughts are human.
**Be specific about feelings.** Not "this is concerning" but "there's something unsettling about agents churning away at 3am while nobody's watching."
**Be specific about feelings.** Instead of "this is concerning," write "there's something unsettling about agents churning away at 3am while nobody's watching."
### Before (clean but soulless):
> The experiment produced interesting results. The agents generated 3 million lines of code. Some developers were impressed while others were skeptical. The implications remain unclear.
### After (has a pulse):
> I genuinely don't know how to feel about this one. 3 million lines of code, generated while the humans presumably slept. Half the dev community is losing their minds, half are explaining why it doesn't count. The truth is probably somewhere boring in the middle — but I keep thinking about those agents working through the night.
> I genuinely don't know how to feel about this one. 3 million lines of code, generated while the humans presumably slept. Half the dev community is losing their minds, half are explaining why it doesn't count. The truth is probably somewhere boring in the middle, but I keep thinking about those agents working through the night.
## CONTENT PATTERNS
@@ -207,6 +207,8 @@ Avoiding AI patterns is only half the job. Sterile, voiceless writing is just as
**High-frequency AI words:** Actually, additionally, align with, crucial, delve, emphasizing, enduring, enhance, fostering, garner, highlight (verb), interplay, intricate/intricacies, key (adjective), landscape (abstract noun), pivotal, showcase, tapestry (abstract noun), testament, underscore (verb), valuable, vibrant
**Marketing and blog clichés (same tell, different register):** at the end of the day, when it comes to, in a world where, moving forward, circle back, deep dive, game-changer, double down, take a step back, on the same page, make no mistake, it turns out, let me be clear, navigate (for challenges), lean into, unpack (before analysis), straightforward (to describe anything)
**Problem:** These words appear far more frequently in post-2023 text. They often co-occur.
**Before:**
@@ -493,6 +495,73 @@ Avoiding AI patterns is only half the job. Sterile, voiceless writing is just as
>
> When users hit a slow page, they leave.
## STYLE, RHYTHM, AND RHETORIC PATTERNS
### 30. Forced Metaphors and Figurative Overwriting
**Signs to watch:** original but strained metaphors, mixed metaphors, figurative substitutions where a plain word is clearer, a metaphor that gets explained right after it is used
**Problem:** Beyond the stock figurative words flagged in patterns 4 and 7, LLMs invent decorative metaphors that add imagery without adding meaning, then often explain them. Plain description is usually clearer and more honest. If the metaphor does not earn its place, cut it and say the literal thing.
**Before:**
> The codebase is a garden we must tend, pruning dead branches and planting seeds of innovation so the whole ecosystem can flourish. In other words, delete unused code and add features.
**After:**
> Delete unused code and add the features users are asking for.
### 31. Dramatic Fragmentation and Punchy Kickers
**Signs to watch:** two- or three-word subjectless sentences used for drama, staccato "X. And Y. And Z." runs, a short quotable line ending every paragraph or section, cutesy appositive fragments ("the catalog, honestly priced")
**Problem:** LLMs chop sentences into fragments for false emphasis and end sections with a quotable "mic-drop" line. It reads like ad copy or a motivational poster. If a line sounds like it belongs on a poster, cut it or fold it back into a real sentence with a subject. This is distinct from pattern 13 (which is about grammatical passive voice); here the tell is rhythm and showmanship, not a hidden actor.
**Before:**
> The catalog, honestly priced. Pay for what it does. Not promises. It just works. Every time.
**After:**
> The catalog is priced by usage, so you pay for the calls you actually make rather than a flat monthly fee.
### 32. Rhetorical Questions Answered Immediately
**Signs to watch:** "What if...?", "The question is...", "Ever wondered...?", a question immediately followed by its own answer, "Think about it."
**Problem:** LLMs pose a question only to answer it a beat later. The question adds no information and stalls the sentence. State the point directly.
**Before:**
> What makes an API good? It comes down to predictability. Think about it: developers want to know exactly what they will get back.
**After:**
> A good API is predictable, so developers know exactly what they will get back.
### 33. Sentence-Opener Tics
**Words to watch:** So..., Look,, habitual sentence-initial And/But, "I think"/"I believe" when stating a fact, adverb openers (Interestingly, Importantly, Notably, Crucially, Essentially, Ultimately)
**Problem:** LLMs lean on a small set of openers. Adverb openers tell the reader how to feel instead of earning it, and "So" or "Look" fake conversational warmth. Drop the opener and start with the substance.
**Before:**
> So, the results were mixed. Interestingly, adoption went up. Importantly, churn went up too. I think that means the feature still needs work.
**After:**
> The results were mixed: adoption rose, but churn rose alongside it, so the feature still needs work.
### 34. Reassurance Kickers
**Signs to watch:** And that's okay., And that's fine., There's nothing wrong with that., no shame in..., you're not alone, it's completely normal
**Problem:** LLMs tack on reassurance the reader never asked for. It softens the writing and assumes the reader needs comforting. Trust the reader: make the point and stop.
**Before:**
> You might not have a testing setup yet. And that's okay. Plenty of teams start without one, and there's nothing wrong with that.
**After:**
> Many teams start without a testing setup and add one once regressions begin costing real time.
---
## Process
@@ -589,6 +658,6 @@ Provide:
This skill is ported from [blader/humanizer](https://github.com/blader/humanizer) (MIT licensed), which is itself based on [Wikipedia: Signs of AI writing](https://en.wikipedia.org/wiki/Wikipedia:Signs_of_AI_writing), maintained by WikiProject AI Cleanup. The patterns documented there come from observations of thousands of instances of AI-generated text on Wikipedia.
Original author: Siqi Chen ([@blader](https://github.com/blader)). Original repo: https://github.com/blader/humanizer (version 2.5.1). Ported to Hermes Agent with Hermes-native tool references (`read_file`, `patch`, `write_file`) and guidance for when to load the skill; the 29 patterns, personality/soul section, and full worked example are preserved verbatim from the source. Original MIT license preserved in the `LICENSE` file alongside this `SKILL.md`.
Original author: Siqi Chen ([@blader](https://github.com/blader)). Original repo: https://github.com/blader/humanizer (version 2.5.1). Ported to Hermes Agent with Hermes-native tool references (`read_file`, `patch`, `write_file`) and guidance for when to load the skill. The original 29 patterns come from the source, and the before/after examples (including the full worked example) are kept as demonstrations. Patterns 30-34 and the "marketing and blog clichés" list added to pattern 7 are Hermes additions and are not part of the upstream source. The skill's own instructional prose has also been lightly edited to follow its own guidance (for example, removing em dashes and negative parallelism from the narration) so the skill models the writing it asks for. Original MIT license preserved in the `LICENSE` file alongside this `SKILL.md`.
Key insight from Wikipedia: "LLMs use statistical algorithms to guess what should come next. The result tends toward the most statistically likely result that applies to the widest variety of cases."
@@ -1,7 +1,7 @@
---
title: "P5Js — p5"
title: "P5Js — p5.js sketches: gen art, shaders, interactive, 3D"
sidebar_label: "P5Js"
description: "p5"
description: "p5.js sketches: gen art, shaders, interactive, 3D"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -16,7 +16,7 @@ Throwaway HTML mockups: 2-3 design variants to compare.
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/creative/sketch` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Hermes Agent (adapted from gsd-build/get-shit-done) |
| License | MIT |
| Platforms | linux, macos, windows |
@@ -44,7 +44,9 @@ Load this when the user says things like "sketch this screen", "show me what X c
## If the user has the full GSD system installed
If `gsd-sketch` shows up as a sibling skill (installed via `npx get-shit-done-cc --hermes`), prefer **`gsd-sketch`** for the full workflow: persistent `.planning/sketches/` with MANIFEST, frontier mode analysis, consistency audits across past sketches, and integration with the rest of GSD. This skill is the lightweight standalone version — one-off sketching without the state machinery.
If `gsd-sketch` shows up as a sibling skill (installed via `npx get-shit-done-cc --hermes`), you can use **`gsd-sketch`** for the fuller workflow: persistent `.planning/sketches/` with MANIFEST, frontier mode analysis, consistency audits across past sketches, and integration with the rest of GSD. This skill is the lightweight standalone version — one-off sketching without the state machinery.
> **Note:** The upstream GSD project ([gsd-build/get-shit-done](https://github.com/gsd-build/get-shit-done)) is **archived / no longer maintained** on GitHub. The npm package (`get-shit-done-cc`) still installs, but treat it as an archived community project — this standalone `sketch` skill is the maintained path and needs nothing extra.
## Core method
@@ -235,4 +237,4 @@ Repeat for each variant, then present the comparison table.
## Attribution
Adapted from the GSD (Get Shit Done) project's `/gsd-sketch` workflow — MIT © 2025 Lex Christopherson ([gsd-build/get-shit-done](https://github.com/gsd-build/get-shit-done)). The full GSD system ships persistent sketch state, theme/variant pattern references, and consistency-audit workflows; install with `npx get-shit-done-cc --hermes --global`.
Adapted from the GSD (Get Shit Done) project's `/gsd-sketch` workflow — MIT © 2025 Lex Christopherson ([gsd-build/get-shit-done](https://github.com/gsd-build/get-shit-done)). The upstream GSD repo is now **archived/unmaintained** on GitHub; the `get-shit-done-cc` npm package still installs (`npx get-shit-done-cc --hermes --global`) and ships persistent sketch state, theme/variant pattern references, and consistency-audit workflows, but treat it as an archived community project.
@@ -228,13 +228,13 @@ Note: `himalaya message write` without piped input opens `$EDITOR`. This works w
### Move/Copy Emails
Move to folder:
Move to folder (target folder comes first, then the message ID):
```bash
himalaya message move "Archive" 42
```
Copy to folder:
Copy to folder (target folder comes first, then the message ID):
```bash
himalaya message copy "Important" 42
@@ -1,7 +1,7 @@
---
title: "Evaluating Llms Harness — lm-eval-harness: benchmark LLMs (MMLU, GSM8K, etc"
title: "Evaluating Llms Harness — lm-eval-harness: benchmark LLMs (MMLU, GSM8K, etc.)"
sidebar_label: "Evaluating Llms Harness"
description: "lm-eval-harness: benchmark LLMs (MMLU, GSM8K, etc"
description: "lm-eval-harness: benchmark LLMs (MMLU, GSM8K, etc.)"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -16,7 +16,7 @@ lm-eval-harness: benchmark LLMs (MMLU, GSM8K, etc.).
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/mlops/evaluation/evaluating-llms-harness` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `lm-eval`, `transformers`, `vllm` |
@@ -55,7 +55,7 @@ lm_eval --model hf \
**View available tasks**:
```bash
lm_eval --tasks list
lm-eval ls tasks
```
## Common workflows
@@ -468,19 +468,19 @@ Verify model and tokenizer match:
**Issue: HumanEval not executing code**
Install execution dependencies:
```bash
pip install human-eval
```
Code-executing tasks (HumanEval, MBPP, etc.) are gated behind an explicit
confirmation flag — you must pass `--confirm_run_unsafe_code` to run them:
Enable code execution:
```bash
lm_eval --model hf \
--model_args pretrained=model-name \
--tasks humaneval \
--allow_code_execution # Required for HumanEval
--confirm_run_unsafe_code # Required to run tasks that execute generated code
```
Without this flag lm-eval refuses to run the task rather than silently skipping
code execution.
## Advanced topics
**Benchmark descriptions**: See [references/benchmark-guide.md](https://github.com/NousResearch/hermes-agent/blob/main/skills/mlops/evaluation/evaluating-llms-harness/references/benchmark-guide.md) for detailed description of all 60+ tasks, what they measure, and interpretation.
@@ -16,7 +16,7 @@ W&B: log ML experiments, sweeps, model registry, dashboards.
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/mlops/evaluation/weights-and-biases` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `wandb` |
@@ -255,7 +255,7 @@ sweep_config = {
},
'parameters': {
'learning_rate': {
'distribution': 'log_uniform',
'distribution': 'log_uniform_values',
'min': 1e-5,
'max': 1e-1
},
@@ -334,7 +334,7 @@ sweep_config = {
'method': 'bayes',
'metric': {'name': 'val/loss', 'goal': 'minimize'},
'parameters': {
'lr': {'distribution': 'log_uniform', 'min': 1e-5, 'max': 1e-1}
'lr': {'distribution': 'log_uniform_values', 'min': 1e-5, 'max': 1e-1}
}
}
```
@@ -450,17 +450,21 @@ trainer.fit(model, datamodule=dm)
```python
import wandb
from wandb.keras import WandbCallback
from wandb.integration.keras import WandbMetricsLogger, WandbModelCheckpoint
# Initialize
wandb.init(project="keras-demo")
# Add callback
# Add callbacks (the monolithic WandbCallback was removed;
# use the dedicated callbacks from wandb.integration.keras instead)
model.fit(
x_train, y_train,
validation_data=(x_val, y_val),
epochs=10,
callbacks=[WandbCallback()] # Auto-logs metrics
callbacks=[
WandbMetricsLogger(), # Auto-logs metrics
WandbModelCheckpoint("models/model-{epoch}") # Saves checkpoints
]
)
```
@@ -16,7 +16,7 @@ HuggingFace hf CLI: search/download/upload models, datasets.
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/mlops/huggingface-hub` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Hugging Face |
| License | MIT |
| Platforms | linux, macos, windows |
@@ -44,8 +44,8 @@ The `hf` command is the modern command-line interface for interacting with the H
### General Operations
* `hf download REPO_ID`: Download files from the Hub.
* `hf upload REPO_ID`: Upload files/folders (recommended for single-commit).
* `hf upload-large-folder REPO_ID LOCAL_PATH`: Recommended for resumable uploads of large directories.
* `hf upload REPO_ID`: Upload files/folders (recommended for single-commit; also handles resumable uploads of large directories).
* `hf upload-large-folder REPO_ID LOCAL_PATH`: **[Deprecated]** — use `hf upload` instead.
* `hf sync`: Sync files between a local directory and a bucket.
* `hf env` / `hf version`: View environment and version details.
@@ -69,7 +69,7 @@ The `hf` command is the modern command-line interface for interacting with the H
* **Datasets:** `hf datasets list`, `info`, and `parquet` (list parquet URLs).
* **SQL Queries:** `hf datasets sql SQL` — Execute raw SQL via DuckDB against dataset parquet URLs.
* **Models:** `hf models list` and `info`.
* **Papers:** `hf papers list` — View daily papers.
* **Papers:** `hf papers ls` — View daily papers.
### Discussions & Pull Requests (`hf discussions`)
* Manage the lifecycle of Hub contributions: `list`, `create`, `info`, `comment`, `close`, `reopen`, and `rename`.
@@ -1,7 +1,7 @@
---
title: "Llama Cpp — llama"
title: "Llama Cpp — llama.cpp local GGUF inference + HF Hub model discovery"
sidebar_label: "Llama Cpp"
description: "llama"
description: "llama.cpp local GGUF inference + HF Hub model discovery"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -16,7 +16,7 @@ vLLM: high-throughput LLM serving, OpenAI API, quantization.
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/mlops/inference/serving-llms-vllm` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `vllm`, `torch`, `transformers` |
@@ -48,7 +48,7 @@ pip install vllm
```python
from vllm import LLM, SamplingParams
llm = LLM(model="meta-llama/Llama-3-8B-Instruct")
llm = LLM(model="meta-llama/Meta-Llama-3-8B-Instruct")
sampling = SamplingParams(temperature=0.7, max_tokens=256)
outputs = llm.generate(["Explain quantum computing"], sampling)
@@ -57,14 +57,14 @@ print(outputs[0].outputs[0].text)
**OpenAI-compatible server**:
```bash
vllm serve meta-llama/Llama-3-8B-Instruct
vllm serve meta-llama/Meta-Llama-3-8B-Instruct
# Query with OpenAI SDK
python -c "
from openai import OpenAI
client = OpenAI(base_url='http://localhost:8000/v1', api_key='EMPTY')
print(client.chat.completions.create(
model='meta-llama/Llama-3-8B-Instruct',
model='meta-llama/Meta-Llama-3-8B-Instruct',
messages=[{'role': 'user', 'content': 'Hello!'}]
).choices[0].message.content)
"
@@ -91,24 +91,23 @@ Choose configuration based on your model size:
```bash
# For 7B-13B models on single GPU
vllm serve meta-llama/Llama-3-8B-Instruct \
vllm serve meta-llama/Meta-Llama-3-8B-Instruct \
--gpu-memory-utilization 0.9 \
--max-model-len 8192 \
--port 8000
# For 30B-70B models with tensor parallelism
vllm serve meta-llama/Llama-2-70b-hf \
vllm serve meta-llama/Meta-Llama-3-70B-Instruct \
--tensor-parallel-size 4 \
--gpu-memory-utilization 0.9 \
--quantization awq \
--port 8000
# For production with caching and metrics
vllm serve meta-llama/Llama-3-8B-Instruct \
# For production with caching (Prometheus metrics are exposed
# automatically at /metrics on the API port)
vllm serve meta-llama/Meta-Llama-3-8B-Instruct \
--gpu-memory-utilization 0.9 \
--enable-prefix-caching \
--enable-metrics \
--metrics-port 9090 \
--port 8000 \
--host 0.0.0.0
```
@@ -129,10 +128,10 @@ Verify TTFT (time to first token) &lt; 500ms and throughput > 100 req/sec.
**Step 3: Enable monitoring**
vLLM exposes Prometheus metrics on port 9090:
vLLM exposes Prometheus metrics at `/metrics` on the API port (default 8000):
```bash
curl http://localhost:9090/metrics | grep vllm
curl http://localhost:8000/metrics | grep vllm
```
Key metrics to monitor:
@@ -148,7 +147,7 @@ Use Docker for consistent deployment:
# Run vLLM in Docker
docker run --gpus all -p 8000:8000 \
vllm/vllm-openai:latest \
--model meta-llama/Llama-3-8B-Instruct \
--model meta-llama/Meta-Llama-3-8B-Instruct \
--gpu-memory-utilization 0.9 \
--enable-prefix-caching
```
@@ -192,7 +191,7 @@ print(f"Loaded {len(prompts)} prompts")
from vllm import LLM, SamplingParams
llm = LLM(
model="meta-llama/Llama-3-8B-Instruct",
model="meta-llama/Meta-Llama-3-8B-Instruct",
tensor_parallel_size=2, # Use 2 GPUs
gpu_memory_utilization=0.9,
max_model_len=4096
@@ -355,9 +354,11 @@ Verify tensor parallelism uses power of 2 GPUs:
vllm serve MODEL --tensor-parallel-size 4 # Not 3
```
Enable speculative decoding for faster generation:
Enable speculative decoding for faster generation (pass config as JSON;
`--speculative-model` was removed in favor of `--speculative-config`):
```bash
vllm serve MODEL --speculative-model DRAFT_MODEL
vllm serve MODEL \
--speculative-config '{"model": "DRAFT_MODEL", "num_speculative_tokens": 5, "method": "draft_model"}'
```
## Advanced topics
@@ -1,7 +1,7 @@
---
title: "Docx — Create, read, edit Word"
title: "Docx — Create, read, edit Word .docx documents and templates"
sidebar_label: "Docx"
description: "Create, read, edit Word"
description: "Create, read, edit Word .docx documents and templates"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -1,7 +1,7 @@
---
title: "Powerpoint — Create, read, edit"
title: "Powerpoint — Create, read, edit .pptx decks, slides, notes, templates"
sidebar_label: "Powerpoint"
description: "Create, read, edit"
description: "Create, read, edit .pptx decks, slides, notes, templates"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -1,7 +1,7 @@
---
title: "Xlsx — Create, read, edit Excel"
title: "Xlsx — Create, read, edit Excel .xlsx spreadsheets and CSVs"
sidebar_label: "Xlsx"
description: "Create, read, edit Excel"
description: "Create, read, edit Excel .xlsx spreadsheets and CSVs"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -16,7 +16,7 @@ Control Philips Hue lights, scenes, rooms via OpenHue CLI.
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/smart-home/openhue` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | community |
| License | MIT |
| Platforms | linux, macos, windows |
@@ -35,8 +35,11 @@ Control Philips Hue lights and scenes via a Hue Bridge from the terminal.
## Prerequisites
```bash
# Linux (pre-built binary)
curl -sL https://github.com/openhue/openhue-cli/releases/latest/download/openhue-linux-amd64 -o ~/.local/bin/openhue && chmod +x ~/.local/bin/openhue
# Linux (pre-built binary — releases ship tarballs, not bare binaries)
curl -sL "https://github.com/openhue/openhue-cli/releases/latest/download/openhue_Linux_x86_64.tar.gz" \
| tar -xz -C /tmp openhue \
&& install -m 0755 /tmp/openhue ~/.local/bin/openhue
# (use openhue_Linux_arm64.tar.gz on ARM64)
# macOS
brew install openhue/cli/openhue-cli
@@ -1,14 +1,14 @@
---
title: "Xurl — X/Twitter via xurl CLI: post, search, DM, media, v2 API"
title: "Xurl — X/Twitter via xurl CLI: raw post search, posting, DM, media"
sidebar_label: "Xurl"
description: "X/Twitter via xurl CLI: post, search, DM, media, v2 API"
description: "X/Twitter via xurl CLI: raw post search, posting, DM, media"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
# Xurl
X/Twitter via xurl CLI: post, search, DM, media, v2 API.
X/Twitter via xurl CLI: raw post search, posting, DM, media.
## Skill metadata
@@ -16,7 +16,7 @@ X/Twitter via xurl CLI: post, search, DM, media, v2 API.
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/social-media/xurl` |
| Version | `1.1.1` |
| Version | `1.1.3` |
| Author | xdevplatform + openclaw + Hermes Agent |
| License | MIT |
| Platforms | linux, macos |
@@ -34,7 +34,7 @@ The following is the complete skill definition that Hermes loads when this skill
Use this skill for:
- posting, replying, quoting, deleting posts
- searching posts and reading timelines/mentions
- searching for raw posts (actual post JSON with IDs you can engage with) and reading timelines/mentions
- liking, reposting, bookmarking
- following, unfollowing, blocking, muting
- direct messages
@@ -197,6 +197,8 @@ xurl delete 1234567890
### Reading & Search
`xurl search` queries the X index as your authenticated account and returns raw post objects — IDs, authors, full text — so results can be immediately engaged with (reply, like, repost, quote). Use it when you need the actual posts rather than a summarized answer about a topic.
```bash
xurl read 1234567890
xurl read https://x.com/user/status/1234567890
@@ -403,12 +405,14 @@ xurl --app staging /2/users/me # one-off against staging
## Agent Workflow
1. Verify prerequisites: `xurl --help` and `xurl auth status`.
2. **Check default app has credentials.** Parse the `auth status` output. The default app is marked with `▸`. If the default app shows `oauth2: (none)` but another app has a valid oauth2 user, tell the user to run `xurl auth default <that-app>` to fix it. This is the most common setup mistake — the user added an app with a custom name but never set it as default, so xurl keeps trying the empty `default` profile.
3. If auth is missing entirely, stop and direct the user to the "One-Time User Setup" section — do NOT attempt to register apps or pass secrets yourself.
4. Start with a cheap read (`xurl whoami`, `xurl user @handle`, `xurl search ... -n 3`) to confirm reachability.
5. Confirm the target post/user and the user's intent before any write action (post, reply, like, repost, DM, follow, block, delete).
6. Use JSON output directly — every response is already structured.
7. Never paste `~/.xurl` contents back into the conversation.
2. Before using `xurl search`, check intent. Reach for it when the task needs actual post objects, authenticated account context, or leads into an X write action — it is the right surface when the user wants posts they can engage with, not just a summary of a topic.
3. **Check default app has credentials.** Parse the `auth status` output. The default app is marked with `▸`. If the default app shows `oauth2: (none)` but another app has a valid oauth2 user, tell the user to run `xurl auth default <that-app>` to fix it. This is the most common setup mistake — the user added an app with a custom name but never set it as default, so xurl keeps trying the empty `default` profile.
4. If auth is missing entirely, stop and direct the user to the "One-Time User Setup" section — do NOT attempt to register apps or pass secrets yourself.
5. Start with a cheap read (`xurl whoami`, `xurl user @handle`, `xurl search ... -n 3`) to confirm reachability.
6. Confirm the target post/user and the user's intent before any write action (post, reply, like, repost, DM, follow, block, delete).
7. Only the `xurl` command output (or the raw X API response) proves that a state-changing X action happened. Never report a write as done based on any other source — search results, summaries, or prior context.
8. Use JSON output directly — every response is already structured.
9. Never paste `~/.xurl` contents back into the conversation.
---
@@ -1,7 +1,7 @@
---
title: "Hermes Agent Skill Authoring — Author in-repo SKILL"
title: "Hermes Agent Skill Authoring — Author in-repo SKILL.md files: frontmatter and structure"
sidebar_label: "Hermes Agent Skill Authoring"
description: "Author in-repo SKILL"
description: "Author in-repo SKILL.md files: frontmatter and structure"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -53,6 +53,10 @@ Source of truth: `tools/skill_manager_tool.py::_validate_frontmatter`. Hard requ
- Parses as a YAML mapping.
- `name` field present.
- `description` field present, ≤ **1024 chars** (`MAX_DESCRIPTION_LENGTH`).
**Long descriptions are truncated to 57 chars + "..." in the system
prompt skill index** (`extract_skill_description` in `agent/skill_utils.py`);
longer text is visible via `skills_list()` and `skill_view()`.
Front-load the trigger phrase.
- Non-empty body after the closing `---`.
Peer-matched shape used by every skill under `skills/software-development/`:
@@ -60,7 +64,7 @@ Peer-matched shape used by every skill under `skills/software-development/`:
```yaml
---
name: my-skill-name # lowercase, hyphens, ≤64 chars (MAX_NAME_LENGTH)
description: Use when <trigger>. <one-line behavior>.
description: Use when <trigger>. <one-line behavior>. # first 57 chars shown in system prompt
version: 1.1.0
author: Hermes Agent
license: MIT
@@ -75,7 +79,9 @@ metadata:
## Size Limits
- Description: ≤ 1024 chars (enforced).
- Description: ≤ 1024 chars (enforced). **Long descriptions render as the first 57 chars
plus "..." in the system prompt skill index;** the rest is visible via `skills_list()`
and `skill_view()`.
- Full SKILL.md: ≤ 100,000 chars (enforced as `MAX_SKILL_CONTENT_CHARS`, ~36k tokens).
- Peer skills in `software-development/` sit at **8-14k chars**. Aim for that range. If you're pushing past 20k, split into `references/*.md` and reference them from SKILL.md.
@@ -183,7 +189,11 @@ Pick the closest existing category. Don't invent new top-level categories casual
2. **Leading whitespace before `---`.** The validator checks `content.startswith("---")`; any leading blank line or BOM fails validation.
3. **Description too generic.** Peer descriptions start with "Use when ..." and describe the *trigger class*, not the one task. "Use when debugging X" > "Debug X".
3. **Description too generic or trigger buried past char 57.** The system prompt
skill index truncates long descriptions at 57 chars. Peer descriptions start
with "Use when ..." and complete the trigger class within that window.
- Good: `Use when debugging Hermes skill discovery failures.`
- Bad: `This skill contains detailed guidance for agents working on Hermes skill discovery failures.`
4. **Forgetting the author/license/metadata block.** Not validator-enforced, but every peer has it; omitting makes the skill look half-finished.
@@ -203,7 +213,8 @@ Pick the closest existing category. Don't invent new top-level categories casual
- [ ] Frontmatter starts at byte 0 with `---`, closes with `\n---\n`
- [ ] `name`, `description`, `version`, `author`, `license`, `metadata.hermes.{tags, related_skills}` all present
- [ ] Name ≤ 64 chars, lowercase + hyphens
- [ ] Description ≤ 1024 chars and starts with "Use when ..."
- [ ] Description ≤ 1024 chars, trigger phrase self-contained within first 57 chars,
and starts with "Use when ..."
- [ ] Total file ≤ 100,000 chars (aim for 8-15k)
- [ ] Structure: `# Title` → `## Overview` → `## When to Use` → body → `## Common Pitfalls` → `## Verification Checklist`
- [ ] Each ordered step has a checkable completion criterion
@@ -0,0 +1,177 @@
---
title: "Inspecting Hermes Desktop Dom — Read the live Hermes desktop DOM/CSS over CDP"
sidebar_label: "Inspecting Hermes Desktop Dom"
description: "Read the live Hermes desktop DOM/CSS over CDP"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
# Inspecting Hermes Desktop Dom
Read the live Hermes desktop DOM/CSS over CDP.
## Skill metadata
| | |
|---|---|
| Source | Bundled (installed by default) |
| Path | `skills/software-development/inspecting-hermes-desktop-dom` |
| Version | `1.0.0` |
| Author | Hermes Agent |
| License | MIT |
| Platforms | linux, macos, windows |
| Tags | `desktop`, `electron`, `cdp`, `dom`, `ui-verification`, `self-inspection` |
| Related skills | [`node-inspect-debugger`](/docs/user-guide/skills/bundled/software-development/software-development-node-inspect-debugger), [`systematic-debugging`](/docs/user-guide/skills/bundled/software-development/software-development-systematic-debugging), [`dogfood`](/docs/user-guide/skills/bundled/software-development/software-development-dogfood) |
## Reference: full SKILL.md
:::info
The following is the complete skill definition that Hermes loads when this skill is triggered. This is what the agent sees as instructions when the skill is active.
:::
# Inspecting the live Hermes desktop DOM
## Overview
When you are developing `apps/desktop` and the user is running that same app
(`hgui` / `npm run dev`), you can read the **live rendered DOM** of the window
they are looking at — computed styles, geometry, which CSS rule actually won,
console output — instead of inferring it from `.tsx` and being wrong.
Dev-server runs open a Chrome DevTools Protocol port on `127.0.0.1:9222`
automatically. The renderer is a Chromium page, so everything DevTools can read,
a script can read.
**This does not replace looking at it.** CDP answers *factual* questions ("what
is the computed padding", "did this element render", "which selector matches").
It cannot tell you whether the result looks good. Colour balance, spacing feel,
and "is this ugly" still need the user's eyes or a screenshot. Answer facts with
CDP; hand aesthetics to the user.
## When to Use
- Verifying a UI change actually took effect in the running app
- "Why is this element still X?" — find the winning rule before editing anything
- Locating a stable selector for a component you're about to change
- Checking a design token's computed value on a real node
- Reading renderer console errors the user mentions but can't copy out
**Don't use for:** perf profiling or heap work (`node-inspect-debugger`,
`debugging-hermes-desktop`), or anything where the real question is "does this
look right".
## The port
Open on `127.0.0.1:9222` for any dev-server run. Closed in exactly two cases
(`apps/desktop/electron/dev-cdp.ts`):
- **packaged builds** — always, and no environment value overrides it;
- **no `HERMES_DESKTOP_DEV_SERVER`** — an unpackaged `electron .` against
`dist/` is how the packaged app gets smoke tested, so it behaves like one.
`HERMES_DESKTOP_CDP_PORT` moves the port (`=9333`) or disables it (`=off`).
Check before doing anything else:
```bash
curl -s --max-time 3 http://127.0.0.1:${HERMES_DESKTOP_CDP_PORT:-9222}/json/version
```
Empty → no port. Do not guess another port silently.
**Never relaunch the user's app to get a port.** That destroys their session and
their state. Launch your own isolated instance instead (below).
## Reading the DOM
`apps/desktop/scripts/eval.mjs` is the one-liner:
```bash
cd apps/desktop
node scripts/eval.mjs "document.querySelectorAll('[data-slot]').length"
```
For multi-step work use the shared client — it has target discovery and
promise-aware eval:
```js
import { CDP, SELECTORS } from './scripts/perf/lib/cdp.mjs'
const cdp = await CDP.connect({ port: 9222, match: '5174' })
const out = await cdp.eval(`JSON.stringify({
radius: getComputedStyle(document.documentElement).getPropertyValue('--radius-scalar').trim(),
composer: !!document.querySelector('[data-slot="composer-rich-input"]')
})`)
cdp.close()
```
`SELECTORS` in `scripts/perf/lib/cdp.mjs` holds the stable `data-slot` hooks
(composer, thread viewport, assistant message, turn pair, profile rail). Prefer
them over inventing a `querySelector` — they are updated as a unit when
components move.
## The question this is best at: which rule won?
Editing every call site because a style "isn't applying" is the classic waste.
Read the real node first:
```js
const el = document.querySelector('[data-slot="aui_assistant-message-root"] a')
JSON.stringify({
ownClasses: el.className,
weight: getComputedStyle(el).fontWeight,
parents: (() => {
const out = []
let n = el
while ((n = n.parentElement) && out.length < 6) out.push(n.className)
return out
})()
})
```
If the node carries no class of its own, the value is **inherited** — sweeping
call sites will not fix it, and you need the ancestor rule. A plugin stylesheet
(e.g. `@tailwindcss/typography`'s `prose a { font-weight: 500 }`) routinely beats
a utility class; override on the shared class, not at each usage.
## Your own isolated instance
When there is no port, or you must not disturb the user's window:
```bash
cd apps/desktop
HERMES_HOME=/tmp/cdp-probe-home \
HERMES_DESKTOP_DEV_SERVER=http://127.0.0.1:5174 \
HERMES_DESKTOP_CDP_PORT=9333 \
npx electron . --user-data-dir=/tmp/cdp-probe-userdata
```
The separate `--user-data-dir` dodges Electron's single-instance lock, so it
cannot collide with a running `hgui`; the separate `HERMES_HOME` keeps it away
from real sessions. Pick a port other than 9222 for the same reason. Run it in
the background and kill it when done.
`npm run perf:serve` does the same with a temp `HERMES_HOME` baked in, if you
also want the perf harness.
## Pitfalls
- **Never kill the user's dev server or app to "free" anything.** A mid-serve
kill nukes Chromium's socket pool, and the resulting `ERR_NETWORK_CHANGED`
gets blamed on whatever you just changed.
- **A throwaway `HERMES_HOME` has no backend.** The app logs `ECONNREFUSED` for
`hermes:api` and may exit on its own. The renderer still mounts and the DOM is
readable — read promptly, and don't mistake a self-exited probe for a broken
port. Chromium logs `DevTools listening on ws://127.0.0.1:<port>/…` when it
binds; that line is the proof the port opened.
- **Poll, don't probe once.** A just-launched app needs a second or two before
the port answers.
- **Never dump the whole DOM.** The desktop renders hundreds of nodes and
`outerHTML` will bury your context. Project down to a small JSON object inside
the evaluated expression.
- **Pass `match` to `CDP.connect`.** Without it you may attach to the pet
overlay, quick-entry window, or a devtools target instead of the main window.
- **`cdp.eval` returns the value; raw `Runtime.evaluate` double-nests it**
(`.result.result.value`). Use the wrapper.
- **`import.meta.env.DEV` is `true` under `vite dev`** in this repo. The note in
`apps/desktop/scripts/profile-typing-lag.md` claiming otherwise is stale.
@@ -1,7 +1,7 @@
---
title: "Node Inspect Debugger — Debug Node"
title: "Node Inspect Debugger — Debug Node.js via --inspect + Chrome DevTools Protocol CLI"
sidebar_label: "Node Inspect Debugger"
description: "Debug Node"
description: "Debug Node.js via --inspect + Chrome DevTools Protocol CLI"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -1,7 +1,7 @@
---
title: "Plan — Write a markdown plan to"
title: "Plan — Write a markdown plan to .hermes/plans/; no execution"
sidebar_label: "Plan"
description: "Write a markdown plan to"
description: "Write a markdown plan to .hermes/plans/; no execution"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -47,6 +47,12 @@ NO FIXES WITHOUT ROOT CAUSE INVESTIGATION FIRST
If you haven't completed Phase 1, you cannot propose fixes.
## The Feedback Loop Rule
The feedback loop is the debugging work. Before reading code to build a theory, create or identify a **tight** command that can go red on the user's exact symptom and green when the bug is fixed. A tight loop is fast, deterministic, agent-runnable, and specific enough to catch this bug — not merely "doesn't crash".
When a clean repro is hard, spend disproportionate effort building the loop. Guessing without a red-capable loop is the failure mode this skill exists to prevent.
## When to Use
Use for ANY technical issue:
@@ -88,21 +94,46 @@ You MUST complete each phase before proceeding to the next.
**Action:** Use `read_file` on the relevant source files. Use `search_files` to find the error string in the codebase.
### 2. Reproduce Consistently
### 2. Build a Tight Feedback Loop
- Can you trigger it reliably?
- What are the exact steps?
- Does it happen every time?
- If not reproducible → gather more data, don't guess
- Can you trigger the user's exact symptom with one command?
- Does the command fail for this bug and only pass once the bug is fixed?
- Is it fast enough to run repeatedly?
- Is it deterministic? For flaky bugs, can you raise the reproduction rate high enough to debug?
- If not reproducible → gather more data, don't guess.
**Action:** Use the `terminal` tool to run the failing test or trigger the bug:
**Ways to construct a loop — try in roughly this order:**
1. **Failing test** at the seam that reaches the bug: unit, integration, or end-to-end.
2. **HTTP script / curl** against a running dev server.
3. **CLI invocation** with fixture input, diffing stdout/stderr against expected output.
4. **Headless browser script** (Playwright/Puppeteer) asserting on DOM, console, or network.
5. **Replay a captured trace**: HAR, request payload, event log, queue message, or webhook body.
6. **Throwaway harness** that boots the smallest useful slice of the system and calls the failing path.
7. **Property / fuzz loop** when the bug is intermittent wrong output over a broad input space.
8. **Bisection harness** suitable for `git bisect run` when the bug appeared between two known states.
9. **Differential loop** comparing old vs new version, two configs, two providers, or two datasets.
10. **Human-in-the-loop script** only as a last resort: script the human steps and capture their result so the loop stays structured.
**Tighten the loop once it exists:**
- Make it faster: cache setup, narrow scope, skip unrelated initialization.
- Make the signal sharper: assert the exact symptom, not generic success.
- Make it more deterministic: pin time, seed randomness, isolate filesystem, freeze network.
For non-deterministic bugs, the immediate goal is a higher reproduction rate, not perfection. Run the trigger 100x, parallelize, add stress, narrow timing windows, or inject sleeps. A 50% flake is debuggable; a 1% flake usually is not.
**Action:** Use the `terminal` tool to run the tight loop:
```bash
# Run specific failing test
# Run a specific failing test
pytest tests/test_module.py::test_name -v
# Run with verbose output
pytest tests/test_module.py -v --tb=long
# Or run a scripted repro
python scripts/repro_bug.py
# Or run a high-repetition flaky repro
for i in {1..100}; do pytest tests/test_flake.py::test_name -q || break; done
```
### 3. Check Recent Changes
@@ -162,11 +193,13 @@ search_files("variable_name\\s*=", path="src/", file_glob="*.py")
### Phase 1 Completion Checklist
- [ ] Error messages fully read and understood
- [ ] Issue reproduced consistently
- [ ] A tight loop command exists and has been run at least once
- [ ] Loop is red-capable: it asserts the user's exact symptom, not a nearby failure
- [ ] Loop is deterministic, or a flaky bug has a high enough reproduction rate to debug
- [ ] Recent changes identified and reviewed
- [ ] Evidence gathered (logs, state, data flow)
- [ ] Problem isolated to specific component/code
- [ ] Root cause hypothesis formed
- [ ] Root cause hypotheses can be stated and tested
**STOP:** Do not proceed to Phase 2 until you understand WHY it's happening.
@@ -176,6 +209,12 @@ search_files("variable_name\\s*=", path="src/", file_glob="*.py")
**Find the pattern before fixing:**
### 0. Minimize the Reproduction
Once the loop is red, shrink the repro to the smallest scenario that still goes red. Cut inputs, callers, config, data, and steps **one at a time**, re-running the loop after each cut. Keep only what is load-bearing for the failure.
Done when removing any remaining element makes the loop go green. A minimal repro narrows the hypothesis space and often becomes the cleanest regression test.
### 1. Find Working Examples
- Locate similar working code in the same codebase
@@ -211,17 +250,22 @@ search_files("similar_pattern", path="src/", file_glob="*.py")
**Scientific method:**
### 1. Form a Single Hypothesis
### 1. Form Ranked Falsifiable Hypotheses
- State clearly: "I think X is the root cause because Y"
- Write it down
- Be specific, not vague
- Generate 3–5 plausible hypotheses before testing any single one.
- Rank them by likelihood and cheapness to falsify.
- State the prediction each hypothesis makes: "If X is the cause, then changing or observing Y should make Z happen."
- Discard or sharpen any hypothesis that does not make a testable prediction.
If the user is present, show the ranked list before testing. They may have domain knowledge that instantly re-ranks it. If the user is AFK, proceed with your ranking.
### 2. Test Minimally
- Make the SMALLEST possible change to test the hypothesis
- One variable at a time
- Don't fix multiple things at once
- Test the highest-ranked hypothesis with the smallest possible probe.
- Change one variable at a time.
- Don't fix multiple things at once.
- Prefer debugger/REPL inspection when available; one breakpoint beats ten logs.
- If you add logs, tag every temporary line with a unique prefix such as `[DEBUG-a4f2]` so cleanup is a single search.
### 3. Verify Before Continuing
@@ -193,6 +193,25 @@ Keep tests green throughout. Don't add behavior.
Next failing test for next behavior. One cycle at a time.
## Avoid Horizontal Slices
Do **not** write all tests first and then all implementation. That is horizontal slicing: RED becomes "write a pile of imagined tests" and GREEN becomes "make the pile pass." It produces brittle tests because the tests are designed before the implementation has taught you what behavior and interface actually matter.
Use vertical tracer bullets instead:
```text
WRONG:
RED: test1, test2, test3, test4
GREEN: impl1, impl2, impl3, impl4
RIGHT:
RED→GREEN: test1→impl1
RED→GREEN: test2→impl2
RED→GREEN: test3→impl3
```
A tracer bullet is one end-to-end behavior slice. It proves the path works, teaches you about the interface, and keeps each next test grounded in what you just learned.
## Why Order Matters
**"I'll write tests after to verify it works"**
@@ -16,7 +16,7 @@ Operate the Antigravity CLI (agy): plugins, auth, sandbox.
|---|---|
| Source | Optional — install with `hermes skills install official/autonomous-ai-agents/antigravity-cli` |
| Path | `optional-skills/autonomous-ai-agents/antigravity-cli` |
| Version | `0.1.0` |
| Version | `0.2.0` |
| Author | Tony Simons (asimons81), Hermes Agent |
| License | MIT |
| Platforms | linux, macos, windows |
@@ -81,6 +81,66 @@ skills use. For one-shot smoke tests and scripted prompts, prefer
To inspect Antigravity's own files, use `read_file` on the paths under Core
paths below — do not `cat` them through the terminal.
## Delegation patterns
`agy` is a coding-agent backend in the same family as `codex` / `claude-code`,
so the same delegation shapes apply. Use these when handing real work (features,
fixes, reviews, second opinions) to Antigravity rather than just smoke-testing.
### One-shot (preferred for scripted prompts and second opinions)
```
terminal(command="agy -p 'Review this diff for bugs and security issues' --model 'Gemini 3.1 Pro (High)'", workdir="/path/to/repo", timeout=300)
```
`-p` is non-interactive: it runs the prompt and exits. Pick the engine with
`--model` (run `agy models` for the exact display strings, e.g.
`'Gemini 3.1 Pro (High)'`, `'Claude Opus 4.6 (Thinking)'`). Add extra context
roots with repeatable `--add-dir`.
### Long / bounded runs (tests, builds, multi-file changes)
Background it and get notified on completion, the same as the `codex` skill:
```
terminal(command="agy -p 'Implement the change described in TASK.md and run the tests' --dangerously-skip-permissions", workdir="/path/to/repo", background=true, notify_on_complete=true)
# then: process(action="poll"/"log"/"wait", session_id=<id>)
```
### Interactive multi-turn (PTY + tmux)
For a conversational session, launch `agy -i` (or bare `agy`) under `pty=true`
with tmux for `capture-pane` / `send-keys`, exactly the pattern documented in
the `codex` / `claude-code` skills. Resume later with `--continue` / `-c` or a
specific `--conversation <id>`.
### Parallel instances (batch sub-issue / worktree fan-out)
Create one git worktree per task and launch an independent `agy -p` in each
(background), then collect results — same worktree fan-out the `codex` skill
uses for batch issue fixing. Bound concurrency to what the machine and your
review capacity can absorb.
### Output + bounding caveat (differs from Claude Code)
- `agy -p` returns **plain text** — there is **no `--output-format json`** and
no result envelope with `session_id` / cost / turn count. Parse stdout
directly; don't expect a JSON object.
- There is **no `--max-turns`**. A print run is bounded by **`--print-timeout`**
(default `5m`). Raise it for long tasks: `--print-timeout 20m`. Pair with the
`terminal` `timeout=` so the outer call doesn't cut the run short.
### Orchestration boundary
Antigravity is a **worker execution backend or third-opinion reviewer** — an
execution detail owned by the agent/profile running a task, NOT a first-class
orchestration primitive. Do not put `agy` on a kanban board as its own card or
treat it as a coordination layer; route work through the normal task graph and
let the assigned worker choose `agy` (vs. codex/claude-code/direct tools) as its
method. Reach for it explicitly only when the user asks, when a worker is
configured to wrap it, or when you want a Gemini-family cross-check against
another agent's plan or diff.
## Core paths
- Binary / entrypoint: `agy`
@@ -175,6 +235,10 @@ paths below — do not `cat` them through the terminal.
session-state problems, not browser-only problems.
- Workspace identity can depend on launch directory and the `.antigravitycli`
project marker.
- `agy -p` prints plain text only — no `--output-format json`, no result
envelope. Don't try to parse a JSON object out of it (unlike `claude-code`).
- Bound print runs with `--print-timeout` (default `5m`), not `--max-turns`
(which does not exist on `agy`).
## Verification
@@ -16,7 +16,7 @@ Delegate coding tasks to the Blackbox AI multi-model CLI.
|---|---|
| Source | Optional — install with `hermes skills install official/autonomous-ai-agents/blackbox` |
| Path | `optional-skills/autonomous-ai-agents/blackbox` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Hermes Agent (Nous Research) |
| License | MIT |
| Platforms | linux, macos, windows |
@@ -33,17 +33,12 @@ The following is the complete skill definition that Hermes loads when this skill
Delegate coding tasks to [Blackbox AI](https://www.blackbox.ai/) via the Hermes terminal. Blackbox is a multi-model coding agent CLI that dispatches tasks to multiple LLMs (Claude, Codex, Gemini, Blackbox Pro) and uses a judge to select the best implementation.
The CLI is [open-source](https://github.com/blackboxaicode/cli) (GPL-3.0, TypeScript, forked from Gemini CLI) and supports interactive sessions, non-interactive one-shots, checkpointing, MCP, and vision model switching.
The CLI (npm `@blackbox_ai/blackbox-cli`, binary `blackbox`) is a TypeScript coding agent (forked from Gemini CLI) and supports interactive sessions, non-interactive one-shots, checkpointing, MCP, and vision model switching.
## Prerequisites
- Node.js 20+ installed
- Blackbox CLI installed: `npm install -g @blackboxai/cli`
- Or install from source:
```
git clone https://github.com/blackboxaicode/cli.git
cd cli && npm install && npm install -g .
```
- Blackbox CLI installed: `npm install -g @blackbox_ai/blackbox-cli` (binary: `blackbox`)
- API key from [app.blackbox.ai/dashboard](https://app.blackbox.ai/dashboard)
- Configured: run `blackbox configure` and enter your API key
- Use `pty=true` in terminal calls — Blackbox CLI is an interactive terminal app
@@ -128,12 +123,16 @@ Blackbox's unique feature is running the same task through multiple models and j
| Flag | Effect |
|------|--------|
| `--prompt "task"` | Non-interactive one-shot execution |
| `--prompt "task"` (`-p`) | Non-interactive one-shot execution |
| `--resume-checkpoint "tag"` | Resume from a saved checkpoint |
| `--yolo` | Auto-approve all actions and model switches |
| `blackbox session` | Start interactive chat session |
| `--yolo` (`-y`) | Auto-approve all actions and model switches |
| `--vlm-switch-mode <mode>` | Image-handling: `once`, `session`, or `persist` |
| `-c, --checkpointing` | Enable checkpointing of file edits |
| `blackbox configure` | Change settings, providers, models |
| `blackbox info` | Display system information |
| `blackbox update` | Update the CLI to the latest version |
| `blackbox mcp` | Manage MCP servers |
| `blackbox extensions` | Manage CLI extensions |
| `blackbox voice <action>` / `blackbox shortcut` | Configure voice input / the `b` shortcut |
## Vision Support
@@ -16,7 +16,7 @@ Delegate coding to xAI Grok Build CLI (features, PRs).
|---|---|
| Source | Optional — install with `hermes skills install official/autonomous-ai-agents/grok` |
| Path | `optional-skills/autonomous-ai-agents/grok` |
| Version | `0.1.0` |
| Version | `0.1.1` |
| Author | Matt Maximo (MattMaximo), Hermes Agent |
| License | MIT |
| Platforms | linux, macos, windows |
@@ -126,14 +126,16 @@ For pure automation, headless `-p` is still cleaner than the TUI.
|------|--------|
| `-p, --single <PROMPT>` | Send one prompt, run headless, exit |
| `-m, --model <MODEL>` | Choose a model |
| `-s, --session-id <ID>` | Create or resume a named headless session |
| `-r, --resume <ID>` | Resume an existing session |
| `-s, --session-id <UUID>` | Assign a **NEW** valid UUID to a fresh conversation (must not already exist). Does **not** resume — use `--resume`/`--continue` for that. Only valid with `--resume`/`--continue` when paired with `--fork-session` |
| `-r, --resume [<UUID>]` | Resume an existing session by its UUID (or the most recent if omitted) |
| `-c, --continue` | Continue the most recent session in the current directory |
| `--fork-session` | When resuming, create a new session ID instead of reusing the original |
| `--max-turns <N>` | Cap the maximum number of agent turns |
| `--cwd <PATH>` | Set the working directory |
| `--output-format <FMT>` | `plain` (default), `json`, or `streaming-json` |
| `--always-approve` | Auto-approve all tool executions (the `--full-auto` / `--yolo` equivalent) |
| `--no-alt-screen` | Run inline, no fullscreen TUI takeover |
| `--no-auto-update` | Skip background update checks (use in all automation) |
| `--no-auto-update` | Skip background update checks (use in all automation; hidden from `--help` but still works) |
### Output Formats
@@ -169,14 +171,19 @@ with `tmux capture-pane`, exactly like the `claude-code` / `codex` skills.
### Session Continuation
Sessions are keyed by **UUID**, not by name. `--session-id` assigns a *new* UUID
to a fresh run (it does **not** resume); `--resume` takes an existing session's
UUID (or omit the value to resume the most recent).
```
# Start a named session
terminal(command="grok --no-auto-update -s refactor-db -p 'Start refactoring the database layer' --always-approve", workdir="/project", timeout=240)
# Start a session with a self-assigned UUID (must be a valid, unused UUID)
SID=$(uuidgen)
terminal(command="grok --no-auto-update -s $SID -p 'Start refactoring the database layer' --always-approve", workdir="/project", timeout=240)
# Resume it later
terminal(command="grok --no-auto-update -r refactor-db -p 'Now add connection pooling' --always-approve", workdir="/project", timeout=180)
# Resume that exact session later by its UUID
terminal(command="grok --no-auto-update -r $SID -p 'Now add connection pooling' --always-approve", workdir="/project", timeout=180)
# Or continue the most recent session in this directory
# Or just continue the most recent session in this directory (no UUID needed)
terminal(command="grok --no-auto-update -c -p 'What did you change last time?'", workdir="/project", timeout=60)
```
@@ -1,14 +1,14 @@
---
title: "Kanban Video Orchestrator — Plan, set up, and monitor a multi-agent video production pipeline backed by Hermes Kanban"
title: "Kanban Video Orchestrator — Plan and run multi-agent video production pipelines"
sidebar_label: "Kanban Video Orchestrator"
description: "Plan, set up, and monitor a multi-agent video production pipeline backed by Hermes Kanban"
description: "Plan and run multi-agent video production pipelines"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
# Kanban Video Orchestrator
Plan, set up, and monitor a multi-agent video production pipeline backed by Hermes Kanban. Use when the user wants to make ANY video — narrative film, product/marketing, music video, explainer, ASCII/terminal art, abstract/generative loop, comic, 3D, real-time/installation — and the work warrants decomposition into specialized profiles (writer, designer, animator, renderer, voice, editor, etc.) coordinated through a kanban board. Performs adaptive discovery to scope the brief, designs an appropriate team for the requested style, generates the setup script that creates Hermes profiles + initial kanban task, then helps monitor execution and intervene when tasks stall or fail. Routes scenes to whichever Hermes rendering / audio / design skill fits each beat (`ascii-video`, `manim-video`, `p5js`, `comfyui`, `touchdesigner-mcp`, `blender-mcp`, `pixel-art`, `baoyu-comic`, `claude-design`, `excalidraw`, `songsee`, `heartmula`, …) plus external APIs for TTS, image-gen, and image-to-video as needed.
Plan and run multi-agent video production pipelines.
## Skill metadata
@@ -0,0 +1,294 @@
---
title: "Tldraw Offline — Drive and script tldraw offline canvases with an agent"
sidebar_label: "Tldraw Offline"
description: "Drive and script tldraw offline canvases with an agent"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
# Tldraw Offline
Drive and script tldraw offline canvases with an agent.
## Skill metadata
| | |
|---|---|
| Source | Optional — install with `hermes skills install official/creative/tldraw-offline` |
| Path | `optional-skills/creative/tldraw-offline` |
| Version | `1.0.0` |
| Author | Teknium + Hermes Agent |
| License | MIT |
| Platforms | linux, macos, windows |
| Tags | `tldraw`, `canvas`, `whiteboard`, `document-script`, `diagramming` |
## Reference: full SKILL.md
:::info
The following is the complete skill definition that Hermes loads when this skill is triggered. This is what the agent sees as instructions when the skill is active.
:::
# tldraw offline Skill
Work with the tldraw offline desktop app (offline.tldraw.com): read the open
canvas, make edits, and write **document scripts** — JavaScript embedded in a
`.tldraw` file that runs on load and gives the file durable behavior. The app
runs a **local HTTP API** (default `localhost:7236`) that a coding agent drives
with plain `curl` from its terminal — this is exactly how the app's own homepage
demo (Codex editing a canvas live) works. The agent does NOT use computer-use /
GUI clicking, and does NOT hand-edit the `.tldraw` file directly. Keep tldraw
offline open while you work.
## When to Use
- The user has tldraw offline open and asks you to build or modify a canvas
(diagrams, wireframes, layouts).
- You want to add durable behavior to a drawing (reactive shapes, interactive
buttons, animation, connection logic) via an embedded document script.
Do NOT hand-place shapes to imitate a drawing — write the code that generates
them. Agents are far better at scripting the canvas than at drawing on it.
## Prerequisites
- **tldraw offline installed and running**, with a document open. Releases:
https://github.com/tldraw/tldraw-offline/releases/latest (macOS DMG, Windows
x64/Arm64, Linux `x86_64`/`arm64` AppImage or amd64/arm64 `.deb`).
- **Agent skills installed in the app**: `Develop → Install Agent Skills`. The
app writes its own tldraw skill into `~/.codex/skills/`, `~/.claude/skills/`,
`~/.cursor/skills/`, and `~/.gemini/skills/` — teaching that agent the `curl`
recipes below. (This Hermes skill mirrors that guidance for Hermes.)
- **The local control API.** On launch the app writes `server.json` to its config
dir (Linux `~/.config/tldraw/`, macOS `~/Library/Application Support/tldraw/`,
Windows `%APPDATA%\tldraw\`) with `port` (default `7236`), a bearer `token`,
`pid`, and `startedAt`. Every request except `GET /` needs
`Authorization: Bearer <token>`. A clean quit removes `server.json`; if it's
present but the port doesn't answer, the app quit uncleanly — treat as not
running.
- **Re-read port + token on EVERY shell call.** Each terminal call is a fresh
shell, so an `export`ed token does not persist — "export once and reuse" sends
an empty token and 401s. Read both inline at the top of each call:
`PORT=$(jq -r .port <server.json>); TOKEN=$(jq -r .token <server.json>)`.
- No account or network needed for local editing.
## How to Run
Two distinct workflows. Pick by whether the change must survive a reload.
**A. One-off canvas edits (`/exec`)** — layout, generating shapes, cleanup. This
is a live edit, not saved script:
```bash
BASE=http://localhost:7236
TOKEN=$(python3 -c "import json;print(json.load(open('$HOME/.config/tldraw/server.json'))['token'])")
# find the focused document id
DOC=$(curl -s "$BASE/api/search" -X POST -H 'content-type: application/json' \
-H "Authorization: Bearer $TOKEN" \
-d '{"code":"return (await api.getFocusedDoc()).id"}' | python3 -c "import sys,json;print(json.load(sys.stdin)['result'])")
# run code with the live `editor` + `helpers` in scope
curl -s "$BASE/api/doc/$DOC/exec" -X POST -H 'content-type: application/json' \
-H "Authorization: Bearer $TOKEN" \
-d '{"code":"const {createShapeId,toRichText}=await import(\"tldraw\"); editor.createShape({id:createShapeId(),type:\"geo\",x:0,y:0,props:{geo:\"rectangle\",w:200,h:100,color:\"blue\",fill:\"solid\",richText:toRichText(\"hello\")}}); return editor.getCurrentPageShapes().length"}'
```
**B. Durable behavior (`script/main.js`)** — reactive/interactive logic that must
survive reload. Edit the file on disk; the app's watcher applies it:
```bash
# get the live script file path for the doc
curl -s "$BASE/api/doc/$DOC/script-workspace" -X POST \
-H "Authorization: Bearer $TOKEN" # -> result.mainJsPath, result.isDefaultScript
# edit result.mainJsPath with read_file / patch / write_file (see scripts/main.js)
# then confirm the watcher applied it:
curl -s "$BASE/api/doc/$DOC/script-status" -H "Authorization: Bearer $TOKEN"
```
The ready-to-adapt document script is `scripts/main.js`.
## Quick Reference
The document-script contract (verified against the app's bundled
`script-context.d.ts`):
```js
import { createShapeId, toRichText } from 'tldraw' // primitives: import, not globals
export default function ({ editor, helpers, signal }) {
editor.run(() => { // batch = one undo step
helpers.createShapeIfMissing({ // idempotent furniture
id: createShapeId('node-1'), type: 'geo', x: 0, y: 0,
props: { geo: 'rectangle', w: 200, h: 100, richText: toRichText('hi') },
})
})
const stop = editor.store.listen(() => { /* react */ }) // fires the tick AFTER a commit
signal.addEventListener('abort', () => stop()) // REQUIRED cleanup on rerun/close
}
```
- `ctx.editor` — the live `Editor` (`createShape`, `updateShape`, `deleteShapes`,
`getCurrentPageShapes`, `getShape`, `getBindingsFromShape`, `zoomToFit`,
`on('tick'|'event', fn)`, `run(fn, { history: 'ignore' })`).
- `ctx.helpers` — `createShapeIfMissing`, `createShapesIfMissing`,
`createArrowBetweenShapes(from, to, { arrowheadEnd })`, `translateShapes`,
`onShapeTranslate(id, fn, { signal })`, `richTextToPlainText`, `boxShapes`,
`getLints`.
- `ctx.signal` — `AbortSignal`; attach every listener/interval teardown to it.
- `config.js` (separate file) registers custom shape/tool/component utils and
runs before mount; `main.js` runs against the mounted editor and reruns on save.
## Interactive UI (clickable buttons that drive state)
Drawn shapes can behave like a real app — the thing a static whiteboard can't do.
Full example: `scripts/counter.js` (a number display + MINUS/RESET/PLUS buttons).
Verification boundary — read this before claiming interaction works or doesn't.
The app's OWN agent playbook says to verify a clickable-UI script with "one
simulated click and one state read" via `/exec` (`editor.dispatch` a pointer
event, await a tick, read the shape's state) — NOT by driving a real mouse. By
that standard the counter is verified: dispatched clicks stepped it
`0 → 1 → 2 → 1 → 0`. Two caveats worth writing down:
- **The script only runs once the app's file-watcher applies it.** On Linux that
watcher uses inotify; a host with an exhausted `fs.inotify.max_user_instances`
logs `inotify_add_watch ... No space left on device`, `script-status` shows
`state: "not-watching"` / `hasEntry: false`, and the script never executes.
This is a host limit, not a script bug — a normal desktop is unaffected.
- **Driving the canvas with computer-use needs FOREGROUND delivery, not
background.** cua-driver's default `background` delivery returns
`background_unavailable` against Electron's occluded renderer — but that is the
first rung, not a wall. cua-driver returns an `escalation: "foreground"` hint;
set `delivery_mode: "foreground"` (pair with `bring_to_front`) and its X11 XTest
path (`x11_xtest_fg`) clicks land — this dismisses the "Run Script" consent
dialog and clicks the canvas. Verified by the Cua team on this exact v1.11.0
AppImage (Linux/X11). Do NOT conclude "Electron rejects synthetic clicks" and
give up in background mode — climb to foreground. (The real product path is
still `/exec`, not clicks; this note is for computer-use-driven testing.)
The pattern:
```js
export default function ({ editor, helpers, signal }) {
// 1. Build buttons idempotently; tag each with meta so the handler finds them.
// Give buttons a visible label AND a meta.action.
// 2. Hit-test pointer_down in PAGE coordinates against the button bounds:
const inside = (b, p) => p.x >= b.x && p.x <= b.x + b.w && p.y >= b.y && p.y <= b.y + b.h
function onEvent(info) {
if (!info || info.name !== 'pointer_down') return
let p = null
try { if (info.point && editor.screenToPage) p = editor.screenToPage(info.point) } catch {}
p = p ?? editor.inputs?.currentPagePoint
if (!p) return
const hit = editor.getCurrentPageShapes().find(
(s) => s.meta?.ui === 'button' &&
inside({ x: s.x, y: s.y, w: s.props.w, h: s.props.h }, p)
)
if (hit) runAction(hit.meta.action) // mutate state; store it in a shape's meta
}
editor.on('event', onEvent)
signal.addEventListener('abort', () => editor.off('event', onEvent)) // REQUIRED
}
```
- Find buttons by `meta` (or visible label via `helpers.richTextToPlainText`),
not by hard-coded coordinates.
- **One script owns both build and read.** If the shapes are created by one code
path (with `meta.action: 'inc'`) and the handler reads another convention
(`meta.action === 'PLUS'`), clicks silently do nothing. Ship the buttons built
by the same script that handles them, or ship an empty canvas so the script
builds them fresh — never pre-bake mismatched shapes into the file's db.
- Keep app state in a shape's `meta` (e.g. `meta.count`) and render it as that
shape's `richText` label, so it survives save and is readable for verification.
- **Detach the listener on `signal` abort.** Skipping this is not cosmetic: on
the next save the old `onEvent` stays attached alongside the new one, so every
click fires twice and a counter jumps by 2 instead of 1.
- For continuous motion use `editor.on('tick', fn)`; for a moving anchor with
attached pieces use `helpers.onShapeTranslate(id, fn, { signal })`.
### Shipping a self-running scripted `.tldraw`
A `.tldraw` is a zip of `metadata.json` + `session.json` + `db.sqlite` + `assets/`
+ `script/` (only those entries are packable). For the script to auto-run without
the "This document contains a script → Run Script" consent dialog:
- `metadata.json` must carry a `script` manifest: `{ "sha256": "<digest>" }`, where
the digest is `sha256` over each sorted `script/` path as `` `${path}\0${sha256hex(bytes)}\n` ``.
A mismatch is rejected as tampered.
- Pre-trust the digest by adding it to `~/.tldraw/script-trust.json`
(`{ "trusted": ["<digest>"] }`, or `$TLDRAW_SCRIPT_TRUST`). The app skips consent
when `isScriptTrusted(digest)` is true.
## Procedure
1. Read the current token/port from `server.json`. Find the target doc with
`api.getFocusedDoc()` (or `api.getDocs()`); name it explicitly if several are
open.
2. For layout/generation, use `/exec`. For durable behavior, edit
`script/main.js` via `/script-workspace`.
3. Make scripts idempotent: create durable shapes with `helpers.createShapeIfMissing`
and stable `createShapeId('name')` ids. Scripts rerun on every load.
4. Keep script-owned writes out of the user's undo stack:
`editor.run(fn, { history: 'ignore' })` (or `helpers.translateShapes`, which
already does).
5. For reactivity, `editor.store.listen(cb)` and tear it down on `signal` abort.
For interaction, `editor.on('event', h)` (hit-test `pointer_down` in page
coords); for animation, `editor.on('tick', h)`.
6. For a single moving anchor + attached internals, prefer
`helpers.onShapeTranslate(anchorId, fn, { signal })` over a broad store
listener — a broad listener can turn your own writes into feedback loops.
## Shape props (validated against tldraw SDK v5 schema)
`editor.createShape` / `createShapeIfMissing` accept partial props (shape utils
fill defaults). When building **raw records** for a file snapshot, every prop
below is required (run `scripts/validate_shapes.mjs`):
| Shape | Required props |
|-------|----------------|
| `note` | `richText`, `color`, `labelColor`, `size`, `font`, `align`, `verticalAlign`, `growY`, `fontSizeAdjustment`, `url`, `scale`, `textLastEditedBy` |
| `text` | `richText`, `color`, `size`, `font`, `textAlign`, `w`, `scale`, `autoSize` |
| `frame` | `w`, `h`, `name`, `color` |
| `geo` | `geo`, `w`, `h`, `color`, `fill`, `richText` (+ dash/size/etc. defaulted) |
`richText` must be `toRichText('...')` — a bare string is rejected. `color` enum:
`black grey light-violet violet blue light-blue yellow orange green light-green
light-red red white`. `font` enum: `draw sans serif mono`.
## Pitfalls
- **`store.listen` fires on the tick AFTER a commit, not synchronously.** If you
write a shape and immediately read state expecting the listener to have run, it
hasn't. Verified live: an in-turn read shows 0 fires; after one `setTimeout`
tick it shows 1. Same reason the app notes `editor.dispatch` is async — await a
tick before verifying.
- **`ctx`, not globals.** The entry is `export default function ({ editor,
helpers, signal })`. There is no bare `editor` global in a document script.
`createShapeId` / `toRichText` / `Vec` come from `import ... from 'tldraw'`.
- **`richText`, not `text`.** Text/note/geo labels use `richText: toRichText(s)`.
- **Raw records need every prop; `createShape` does not.** In-app pass only the
props you care about; a hand-built `.tldraw` snapshot needs the full set (table).
- **Scripts rerun on every load — be idempotent.** Use `createShapeIfMissing`
with stable ids or you duplicate content and clobber user edits.
- **Clean up on `signal`.** `signal.addEventListener('abort', () => stop())` for
every `store.listen` / `editor.on` / `setInterval`; the signal fires before
rerun and on close.
- **Keep script writes out of undo:** `editor.run(fn, { history: 'ignore' })`.
- **`editor.on('tick')` pauses when the window is hidden** (it is a RAF loop);
`setInterval` keeps firing but Electron throttles it to ~1/s in the background.
- **The API needs the bearer token** from `server.json`; the port can be non-default
(`server.listen(0)` picks one) — always read the file, don't hardcode `7236`.
- **Only `tldraw` / `react` / `react-dom` import** — not a Node project.
## Verification
- **Shape schema (offline, no app):** `node scripts/validate_shapes.mjs` — builds
the real tldraw schema and validates note/text/frame. Passing prints `3/3`.
- **Live canvas edits:** after `/exec`, read back with `/api/search` →
`api.getShapes(docId)` (returns `{ page, viewport, shapes }`) and
`api.getBindings(docId)` (array). Confirm expected shapes/bindings exist. Grab
`api.getScreenshot(docId)` (returns `{ filePath, ... }`) and inspect the PNG/JPEG
with `vision_analyze`.
- **Durable script applied:** `GET /api/doc/:id/script-status`. Success is
`state: "applied"` (`currentDiskDigest === lastAppliedDigest === manifestSha256`,
`pendingApply === false`, `lastApplyError === null`). If it stays `"pending"`
after a short retry, report that instead of claiming success; `"error"` means
the apply failed — read `errorLogPath`.
@@ -1,7 +1,7 @@
---
title: "Inference Sh Cli — Run 150+ AI apps (image, video, LLM) via inference"
title: "Inference Sh Cli — Run 150+ AI apps (image, video, LLM) via inference.sh CLI"
sidebar_label: "Inference Sh Cli"
description: "Run 150+ AI apps (image, video, LLM) via inference"
description: "Run 150+ AI apps (image, video, LLM) via inference.sh CLI"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -16,7 +16,7 @@ Run PyTorch training across GPUs with minimal changes.
|---|---|
| Source | Optional — install with `hermes skills install official/mlops/accelerate` |
| Path | `optional-skills/mlops/accelerate` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `accelerate`, `torch`, `transformers` |
@@ -163,30 +163,35 @@ for batch in dataloader:
### Workflow 3: DeepSpeed ZeRO integration
**Enable DeepSpeed ZeRO-2**:
**Enable DeepSpeed ZeRO-2** (pass a `DeepSpeedPlugin`, not a raw dict):
```python
from accelerate import Accelerator
from accelerate import Accelerator, DeepSpeedPlugin
deepspeed_plugin = DeepSpeedPlugin(
zero_stage=2, # ZeRO-2
offload_optimizer_device="none", # or "cpu" to offload
gradient_accumulation_steps=4,
)
accelerator = Accelerator(
mixed_precision='bf16',
deepspeed_plugin={
"zero_stage": 2, # ZeRO-2
"offload_optimizer": False,
"gradient_accumulation_steps": 4
}
deepspeed_plugin=deepspeed_plugin, # DeepSpeedPlugin instance (or dict[str, DeepSpeedPlugin])
)
# Same code as before!
model, optimizer, dataloader = accelerator.prepare(model, optimizer, dataloader)
```
**Or via config**:
```bash
accelerate config
# Select: DeepSpeed → ZeRO-2
**Or point at a full DeepSpeed JSON config via the plugin**:
```python
from accelerate import Accelerator, DeepSpeedPlugin
# hf_ds_config accepts a path to a DeepSpeed config JSON (or a dict)
deepspeed_plugin = DeepSpeedPlugin(hf_ds_config="ds_config.json")
accelerator = Accelerator(mixed_precision='bf16', deepspeed_plugin=deepspeed_plugin)
```
**deepspeed_config.json**:
**ds_config.json** (a raw DeepSpeed config — passed via the plugin, NOT via `--config_file`):
```json
{
"fp16": {"enabled": false},
@@ -200,9 +205,20 @@ accelerate config
}
```
**Launch**:
**Or via interactive config**:
```bash
accelerate launch --config_file deepspeed_config.json train.py
accelerate config
# Select: DeepSpeed → ZeRO-2
# This writes an accelerate YAML config (default: ~/.cache/huggingface/accelerate/default_config.yaml)
```
**Launch** (`--config_file` expects an accelerate YAML, not a raw DeepSpeed JSON):
```bash
# Uses the default accelerate config written by `accelerate config`
accelerate launch train.py
# Or point at a specific accelerate YAML
accelerate launch --config_file accelerate_deepspeed.yaml train.py
```
### Workflow 4: FSDP (Fully Sharded Data Parallel)
@@ -213,7 +229,7 @@ from accelerate import Accelerator, FullyShardedDataParallelPlugin
fsdp_plugin = FullyShardedDataParallelPlugin(
sharding_strategy="FULL_SHARD", # ZeRO-3 equivalent
auto_wrap_policy="TRANSFORMER_AUTO_WRAP",
auto_wrap_policy="transformer_based_wrap", # valid: transformer_based_wrap | size_based_wrap | no_wrap
cpu_offload=False
)
@@ -16,7 +16,7 @@ Speed up long-sequence transformer training and inference.
|---|---|
| Source | Optional — install with `hermes skills install official/mlops/flash-attention` |
| Path | `optional-skills/mlops/flash-attention` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `flash-attn`, `torch`, `transformers` |
@@ -99,13 +99,12 @@ import torch.nn.functional as F
out = F.scaled_dot_product_attention(q, k, v, attn_mask=mask)
```
Force Flash Attention backend:
Force Flash Attention backend (`torch.backends.cuda.sdp_kernel` is deprecated; use
`torch.nn.attention.sdpa_kernel` with `SDPBackend`):
```python
with torch.backends.cuda.sdp_kernel(
enable_flash=True,
enable_math=False,
enable_mem_efficient=False
):
from torch.nn.attention import SDPBackend, sdpa_kernel
with sdpa_kernel(SDPBackend.FLASH_ATTENTION):
out = F.scaled_dot_product_attention(q, k, v)
```
@@ -118,7 +117,8 @@ def test_attention(use_flash):
q, k, v = [torch.randn(2, 8, 2048, 64, device='cuda', dtype=torch.float16) for _ in range(3)]
if use_flash:
with torch.backends.cuda.sdp_kernel(enable_flash=True):
from torch.nn.attention import SDPBackend, sdpa_kernel
with sdpa_kernel(SDPBackend.FLASH_ATTENTION):
return F.scaled_dot_product_attention(q, k, v)
else:
attn = (q @ k.transpose(-2, -1) / 8.0).softmax(dim=-1)
@@ -247,14 +247,19 @@ print(f"Memory allocated: {torch.cuda.max_memory_allocated()/1e9:.2f}GB")
### Workflow 3: H100 FP8 optimization (FlashAttention-3)
For maximum performance on H100 GPUs.
For maximum performance on Hopper GPUs (H100).
> **Important:** The pip package `flash-attn` (2.8.x) ships **FlashAttention-2 only** — it does
> **not** contain FA3 or FP8 H100 kernels, and `flash_attn_func` does **not** auto-use FP8.
> FlashAttention-3 is a separate **beta** build compiled from source from the repo's `hopper/`
> directory, exposed via the `flash_attn_interface` module. FA3 supports FP16/BF16 forward+backward
> and **FP8 forward only**.
```
FP8 Setup:
- [ ] Step 1: Verify H100 GPU available
- [ ] Step 2: Install flash-attn with FP8 support
- [ ] Step 3: Convert inputs to FP8
- [ ] Step 4: Run with FP8 attention
- [ ] Step 1: Verify Hopper (H100) GPU available
- [ ] Step 2: Build & install FlashAttention-3 from source (hopper/)
- [ ] Step 3: Use the FA3 interface (FP8 forward)
```
**Step 1: Verify H100 GPU**
@@ -264,36 +269,38 @@ nvidia-smi --query-gpu=name --format=csv
# Should show "H100" or "H800"
```
**Step 2: Install flash-attn with FP8 support**
**Step 2: Build & install FlashAttention-3 from source**
FA3 is NOT included in `pip install flash-attn`. Build it from the `hopper/` subdirectory:
```bash
pip install flash-attn --no-build-isolation
# FP8 support included for H100
git clone https://github.com/Dao-AILab/flash-attention.git
cd flash-attention/hopper
python setup.py install
# (compilation is heavy and requires a CUDA toolchain + Hopper GPU)
```
**Step 3: Convert inputs to FP8**
**Step 3: Use the FA3 interface (FP8 forward)**
FA3 exposes its own module `flash_attn_interface` (distinct from the FA2 `flash_attn`).
FP8 is a **forward-only** path and expects `float8_e4m3fn` inputs:
```python
import torch
from flash_attn_interface import flash_attn_func # FA3 (hopper build), not `flash_attn`
# q, k, v: [batch, seqlen, nheads, headdim]
q = torch.randn(2, 4096, 32, 64, device='cuda', dtype=torch.float16)
k = torch.randn(2, 4096, 32, 64, device='cuda', dtype=torch.float16)
v = torch.randn(2, 4096, 32, 64, device='cuda', dtype=torch.float16)
# Convert to float8_e4m3 (FP8)
# FP8 forward (inference / forward-only): cast to float8_e4m3fn
q_fp8 = q.to(torch.float8_e4m3fn)
k_fp8 = k.to(torch.float8_e4m3fn)
v_fp8 = v.to(torch.float8_e4m3fn)
```
**Step 4: Run with FP8 attention**
```python
from flash_attn import flash_attn_func
# FlashAttention-3 automatically uses FP8 kernels on H100
out = flash_attn_func(q_fp8, k_fp8, v_fp8)
# Result: ~1.2 PFLOPS, 1.5-2x faster than FP16
out = flash_attn_func(q_fp8, k_fp8, v_fp8, causal=True)
# FP16/BF16 forward+backward is also supported by the FA3 interface.
```
## When to use vs alternatives
@@ -16,7 +16,7 @@ Constrain LLM output with grammars; guarantee valid JSON.
|---|---|
| Source | Optional — install with `hermes skills install official/mlops/guidance` |
| Path | `optional-skills/mlops/guidance` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `guidance`, `transformers` |
@@ -70,13 +70,19 @@ result = lm + "The capital of France is " + gen("capital", max_tokens=5)
print(result["capital"]) # "Paris"
```
### With Anthropic Claude
### Chat format with a local model
> **Constraint support requires local logit access.** Regex, `select()`, and
> grammar-based constrained generation only work with local backends
> (`Transformers`, `LlamaCpp`). Remote API backends (`OpenAI`, and Azure
> variants) support unconstrained `gen()` / chat only — they cannot enforce
> token-level constraints. guidance 0.3.x has no `models.Anthropic` class.
```python
from guidance import models, gen, system, user, assistant
# Configure Claude
lm = models.Anthropic("claude-sonnet-4-5-20250929")
# Local model (supports constrained generation)
lm = models.Transformers("microsoft/Phi-4-mini-instruct")
# Use context managers for chat format
with system():
@@ -98,7 +104,7 @@ Guidance uses Pythonic context managers for chat-style interactions.
```python
from guidance import system, user, assistant, gen
lm = models.Anthropic("claude-sonnet-4-5-20250929")
lm = models.Transformers("microsoft/Phi-4-mini-instruct")
# System message
with system():
@@ -129,7 +135,7 @@ Guidance ensures outputs match specified patterns using regex or grammars.
```python
from guidance import models, gen
lm = models.Anthropic("claude-sonnet-4-5-20250929")
lm = models.Transformers("microsoft/Phi-4-mini-instruct")
# Constrain to valid email format
lm += "Email: " + gen("email", regex=r"[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}")
@@ -154,7 +160,7 @@ print(lm["date"]) # Guaranteed YYYY-MM-DD format
```python
from guidance import models, gen, select
lm = models.Anthropic("claude-sonnet-4-5-20250929")
lm = models.Transformers("microsoft/Phi-4-mini-instruct")
# Constrain to specific choices
lm += "Sentiment: " + select(["positive", "negative", "neutral"], name="sentiment")
@@ -188,7 +194,7 @@ prompt = "The capital of France is "
```python
from guidance import models, gen
lm = models.Anthropic("claude-sonnet-4-5-20250929")
lm = models.Transformers("microsoft/Phi-4-mini-instruct")
# Token healing enabled by default
lm += "The capital of France is " + gen("capital", max_tokens=5)
@@ -202,26 +208,30 @@ lm += "The capital of France is " + gen("capital", max_tokens=5)
### 4. Grammar-Based Generation
Define complex structures using context-free grammars.
Define complex structures by composing grammar functions. The template-string
`grammar=` form is not part of current guidance — build grammars from
composable functions, or use `guidance.json()` for JSON.
```python
from guidance import models, gen
from guidance import json as gen_json
from pydantic import BaseModel, Field
lm = models.Anthropic("claude-sonnet-4-5-20250929")
lm = models.Transformers("microsoft/Phi-4-mini-instruct")
# JSON grammar (simplified)
json_grammar = """
{
"name": <gen name regex="[A-Za-z ]+" max_tokens=20>,
"age": <gen age regex="[0-9]+" max_tokens=3>,
"email": <gen email regex="[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\\.[a-zA-Z]{2,}" max_tokens=50>
}
"""
# JSON via a Pydantic schema (guidance.json compiles the schema to a grammar)
class Person(BaseModel):
name: str = Field(pattern=r"[A-Za-z ]+")
age: int
email: str = Field(pattern=r"[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}")
# Generate valid JSON
lm += gen("person", grammar=json_grammar)
lm += gen_json(name="person", schema=Person)
print(lm["person"]) # Guaranteed valid JSON structure
print(lm["person"]) # Guaranteed valid JSON matching the schema
# Or compose grammar functions directly:
grammar = "name=" + gen("name", regex=r"[A-Za-z ]+") + " age=" + gen("age", regex=r"[0-9]+")
lm += grammar
```
**Use cases:**
@@ -245,7 +255,7 @@ def generate_person(lm):
return lm
# Use the function
lm = models.Anthropic("claude-sonnet-4-5-20250929")
lm = models.Transformers("microsoft/Phi-4-mini-instruct")
lm = generate_person(lm)
print(lm["name"])
@@ -283,20 +293,14 @@ def react_agent(lm, question, tools, max_rounds=5):
## Backend Configuration
### Anthropic Claude
### OpenAI (remote — unconstrained only)
> Remote API backends cannot do constrained generation (regex/select/grammar);
> use them only for plain chat/`gen()`. For constraints, use a local backend.
```python
from guidance import models
lm = models.Anthropic(
model="claude-sonnet-4-5-20250929",
api_key="your-api-key" # Or set ANTHROPIC_API_KEY env var
)
```
### OpenAI
```python
lm = models.OpenAI(
model="gpt-4o-mini",
api_key="your-api-key" # Or set OPENAI_API_KEY env var
@@ -333,7 +337,7 @@ lm = LlamaCpp(
```python
from guidance import models, gen, system, user, assistant
lm = models.Anthropic("claude-sonnet-4-5-20250929")
lm = models.Transformers("microsoft/Phi-4-mini-instruct")
with system():
lm += "You generate valid JSON."
@@ -356,7 +360,7 @@ print(lm) # Valid JSON guaranteed
```python
from guidance import models, gen, select
lm = models.Anthropic("claude-sonnet-4-5-20250929")
lm = models.Transformers("microsoft/Phi-4-mini-instruct")
text = "This product is amazing! I love it."
@@ -387,7 +391,7 @@ def chain_of_thought(lm, question):
return lm
lm = models.Anthropic("claude-sonnet-4-5-20250929")
lm = models.Transformers("microsoft/Phi-4-mini-instruct")
lm = chain_of_thought(lm, "What is 15% of 200?")
print(lm["answer"])
@@ -429,7 +433,7 @@ def react_agent(lm, question):
return lm
lm = models.Anthropic("claude-sonnet-4-5-20250929")
lm = models.Transformers("microsoft/Phi-4-mini-instruct")
lm = react_agent(lm, "What is 25 * 4 + 10?")
print(lm["answer"])
```
@@ -460,7 +464,7 @@ def extract_entities(lm, text):
text = "Tim Cook announced at Apple Park on 2024-09-15 in Cupertino."
lm = models.Anthropic("claude-sonnet-4-5-20250929")
lm = models.Transformers("microsoft/Phi-4-mini-instruct")
lm = extract_entities(lm, text)
print(f"Person: {lm['person']}")
@@ -16,7 +16,7 @@ Outlines: structured JSON/regex/Pydantic LLM generation.
|---|---|
| Source | Optional — install with `hermes skills install official/mlops/outlines` |
| Path | `optional-skills/mlops/inference/outlines` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `outlines`, `transformers`, `vllm`, `pydantic` |
@@ -41,7 +41,15 @@ Use Outlines when you need to:
- **Generate against JSON schemas** automatically
- **Control token sampling** at the grammar level
**GitHub Stars**: 8,000+ | **From**: dottxt.ai (formerly .txt)
**GitHub Stars**: 12,000+ | **From**: dottxt.ai (formerly .txt)
> **API note (Outlines 1.x):** This skill targets the current v1 API.
> The pre-1.0 helpers (`outlines.models.transformers(...)`,
> `outlines.generate.json/choice/regex/...`) have been **removed**. In v1 you
> create a model with `outlines.from_transformers(...)` (or `from_vllm`,
> `from_llamacpp`, `from_openai`) and then **call the model directly** with an
> output type: `model(prompt, output_type)`. JSON/Pydantic outputs are returned
> as a **JSON string** — validate with `YourModel.model_validate_json(result)`.
## Installation
@@ -62,14 +70,19 @@ pip install outlines vllm # vLLM for high-throughput
```python
import outlines
from typing import Literal
from transformers import AutoModelForCausalLM, AutoTokenizer
# Load model
model = outlines.models.transformers("microsoft/Phi-3-mini-4k-instruct")
MODEL_NAME = "microsoft/Phi-3-mini-4k-instruct"
# Generate with type constraint
# v1: wrap a Transformers model + tokenizer
model = outlines.from_transformers(
AutoModelForCausalLM.from_pretrained(MODEL_NAME, device_map="auto"),
AutoTokenizer.from_pretrained(MODEL_NAME),
)
# Call the model directly with an output type
prompt = "Sentiment of 'This product is amazing!': "
generator = outlines.generate.choice(model, ["positive", "negative", "neutral"])
sentiment = generator(prompt)
sentiment = model(prompt, Literal["positive", "negative", "neutral"])
print(sentiment) # "positive" (guaranteed one of these)
```
@@ -79,19 +92,24 @@ print(sentiment) # "positive" (guaranteed one of these)
```python
from pydantic import BaseModel
import outlines
from transformers import AutoModelForCausalLM, AutoTokenizer
class User(BaseModel):
name: str
age: int
email: str
model = outlines.models.transformers("microsoft/Phi-3-mini-4k-instruct")
MODEL_NAME = "microsoft/Phi-3-mini-4k-instruct"
model = outlines.from_transformers(
AutoModelForCausalLM.from_pretrained(MODEL_NAME, device_map="auto"),
AutoTokenizer.from_pretrained(MODEL_NAME),
)
# Generate structured output
# Generate structured output (returns a JSON string)
prompt = "Extract user: John Doe, 30 years old, john@example.com"
generator = outlines.generate.json(model, User)
user = generator(prompt)
result = model(prompt, User, max_new_tokens=200)
user = User.model_validate_json(result) # parse into the Pydantic model
print(user.name) # "John Doe"
print(user.age) # 30
print(user.email) # "john@example.com"
@@ -101,11 +119,12 @@ print(user.email) # "john@example.com"
### 1. Constrained Token Sampling
Outlines uses Finite State Machines (FSM) to constrain token generation at the logit level.
Outlines constrains token generation at the logit level using a compiled
automaton derived from your output type.
**How it works:**
1. Convert schema (JSON/Pydantic/regex) to context-free grammar (CFG)
2. Transform CFG into Finite State Machine (FSM)
1. Convert the output type (JSON/Pydantic/regex/`Literal`) to a schema/grammar
2. Compile the grammar into a token-level automaton
3. Filter invalid tokens at each step during generation
4. Fast-forward when only one valid token exists
@@ -116,42 +135,36 @@ Outlines uses Finite State Machines (FSM) to constrain token generation at the l
```python
import outlines
from pydantic import BaseModel
from transformers import AutoModelForCausalLM, AutoTokenizer
# Pydantic model -> JSON schema -> CFG -> FSM
class Person(BaseModel):
name: str
age: int
model = outlines.models.transformers("microsoft/Phi-3-mini-4k-instruct")
# Behind the scenes:
# 1. Person -> JSON schema
# 2. JSON schema -> CFG
# 3. CFG -> FSM
# 4. FSM filters tokens during generation
generator = outlines.generate.json(model, Person)
result = generator("Generate person: Alice, 25")
```
### 2. Structured Generators
Outlines provides specialized generators for different output types.
#### Choice Generator
```python
# Multiple choice selection
generator = outlines.generate.choice(
model,
["positive", "negative", "neutral"]
model = outlines.from_transformers(
AutoModelForCausalLM.from_pretrained("microsoft/Phi-3-mini-4k-instruct", device_map="auto"),
AutoTokenizer.from_pretrained("microsoft/Phi-3-mini-4k-instruct"),
)
sentiment = generator("Review: This is great!")
# Result: One of the three choices
result = model("Generate person: Alice, 25", Person)
person = Person.model_validate_json(result)
```
#### JSON Generator
### 2. Output Types
In v1 you pass the desired **output type** directly as the second argument.
#### Multiple choice (`Literal`)
```python
from typing import Literal
sentiment = model("Review: This is great!", Literal["positive", "negative", "neutral"])
# Result: one of the three choices
```
#### JSON via Pydantic
```python
from pydantic import BaseModel
@@ -161,97 +174,85 @@ class Product(BaseModel):
price: float
in_stock: bool
# Generate valid JSON matching schema
generator = outlines.generate.json(model, Product)
product = generator("Extract: iPhone 15, $999, available")
# Guaranteed valid Product instance
print(type(product)) # <class '__main__.Product'>
result = model("Extract: iPhone 15, $999, available", Product)
product = Product.model_validate_json(result) # valid Product instance
```
#### Regex Generator
#### Regex (pass a regex string)
```python
# Generate text matching regex
generator = outlines.generate.regex(
model,
r"[0-9]{3}-[0-9]{3}-[0-9]{4}" # Phone number pattern
)
phone = generator("Generate phone number:")
# Result: "555-123-4567" (guaranteed to match pattern)
# Generate text matching a regex pattern
phone = model("Generate phone number:", r"[0-9]{3}-[0-9]{3}-[0-9]{4}")
# Result: "555-123-4567" (guaranteed to match the pattern)
```
#### Integer/Float Generators
#### Numeric types
```python
# Generate specific numeric types
int_generator = outlines.generate.integer(model)
age = int_generator("Person's age:") # Guaranteed integer
float_generator = outlines.generate.float(model)
price = float_generator("Product price:") # Guaranteed float
# Pass the Python type directly
age = model("Person's age:", int) # guaranteed integer
price = model("Product price:", float) # guaranteed float
```
### 3. Model Backends
Outlines supports multiple local and API-based backends.
Outlines supports multiple local and API-based backends via `from_*` factories.
#### Transformers (Hugging Face)
```python
import outlines
from transformers import AutoModelForCausalLM, AutoTokenizer
# Load from Hugging Face
model = outlines.models.transformers(
"microsoft/Phi-3-mini-4k-instruct",
device="cuda" # Or "cpu"
model = outlines.from_transformers(
AutoModelForCausalLM.from_pretrained("microsoft/Phi-3-mini-4k-instruct", device_map="auto"),
AutoTokenizer.from_pretrained("microsoft/Phi-3-mini-4k-instruct"),
)
# Use with any generator
generator = outlines.generate.json(model, YourModel)
result = model(prompt, YourModel)
```
#### llama.cpp
```python
# Load GGUF model
model = outlines.models.llamacpp(
"./models/llama-3.1-8b-instruct.Q4_K_M.gguf",
n_gpu_layers=35
)
import outlines
from llama_cpp import Llama
generator = outlines.generate.json(model, YourModel)
llm = Llama("./models/llama-3.1-8b-instruct.Q4_K_M.gguf", n_gpu_layers=35, n_ctx=4096)
model = outlines.from_llamacpp(llm)
result = model(prompt, YourModel)
```
#### vLLM (High Throughput)
```python
# For production deployments
model = outlines.models.vllm(
"meta-llama/Llama-3.1-8B-Instruct",
tensor_parallel_size=2 # Multi-GPU
)
import outlines
from vllm import LLM
generator = outlines.generate.json(model, YourModel)
llm = LLM("meta-llama/Llama-3.1-8B-Instruct", tensor_parallel_size=2)
model = outlines.from_vllm(llm)
result = model(prompt, YourModel)
```
#### OpenAI (Limited Support)
#### OpenAI (server-side constrained JSON)
```python
# Basic OpenAI support
model = outlines.models.openai(
"gpt-4o-mini",
api_key="your-api-key"
)
import outlines
from openai import OpenAI
# Note: Some features limited with API models
generator = outlines.generate.json(model, YourModel)
client = OpenAI()
model = outlines.from_openai(client, "gpt-4o-mini")
# API backends support JSON-schema style structured output
result = model(prompt, YourModel)
```
### 4. Pydantic Integration
Outlines has first-class Pydantic support with automatic schema translation.
Generation returns a JSON string; call `model_validate_json` to get an instance.
#### Basic Models
@@ -264,10 +265,8 @@ class Article(BaseModel):
word_count: int = Field(description="Number of words", gt=0)
tags: list[str] = Field(description="List of tags")
model = outlines.models.transformers("microsoft/Phi-3-mini-4k-instruct")
generator = outlines.generate.json(model, Article)
article = generator("Generate article about AI")
result = model("Generate article about AI", Article, max_new_tokens=300)
article = Article.model_validate_json(result)
print(article.title)
print(article.word_count) # Guaranteed > 0
```
@@ -285,9 +284,8 @@ class Person(BaseModel):
age: int
address: Address # Nested model
generator = outlines.generate.json(model, Person)
person = generator("Generate person in New York")
result = model("Generate person in New York", Person)
person = Person.model_validate_json(result)
print(person.address.city) # "New York"
```
@@ -307,9 +305,8 @@ class Application(BaseModel):
status: Status # Must be one of enum values
priority: Literal["low", "medium", "high"] # Must be one of literals
generator = outlines.generate.json(model, Application)
app = generator("Generate application")
result = model("Generate application", Application)
app = Application.model_validate_json(result)
print(app.status) # Status.PENDING (or APPROVED/REJECTED)
```
@@ -320,6 +317,7 @@ print(app.status) # Status.PENDING (or APPROVED/REJECTED)
```python
from pydantic import BaseModel
import outlines
from transformers import AutoModelForCausalLM, AutoTokenizer
class CompanyInfo(BaseModel):
name: str
@@ -327,8 +325,10 @@ class CompanyInfo(BaseModel):
industry: str
employees: int
model = outlines.models.transformers("microsoft/Phi-3-mini-4k-instruct")
generator = outlines.generate.json(model, CompanyInfo)
model = outlines.from_transformers(
AutoModelForCausalLM.from_pretrained("microsoft/Phi-3-mini-4k-instruct", device_map="auto"),
AutoTokenizer.from_pretrained("microsoft/Phi-3-mini-4k-instruct"),
)
text = """
Apple Inc. was founded in 1976 in the technology industry.
@@ -336,7 +336,7 @@ The company employs approximately 164,000 people worldwide.
"""
prompt = f"Extract company information:\n{text}\n\nCompany:"
company = generator(prompt)
company = CompanyInfo.model_validate_json(model(prompt, CompanyInfo, max_new_tokens=200))
print(f"Name: {company.name}")
print(f"Founded: {company.founded_year}")
@@ -348,26 +348,24 @@ print(f"Employees: {company.employees}")
```python
from typing import Literal
import outlines
model = outlines.models.transformers("microsoft/Phi-3-mini-4k-instruct")
from pydantic import BaseModel
# Binary classification
generator = outlines.generate.choice(model, ["spam", "not_spam"])
result = generator("Email: Buy now! 50% off!")
result = model("Email: Buy now! 50% off!", Literal["spam", "not_spam"])
# Multi-class classification
categories = ["technology", "business", "sports", "entertainment"]
category_gen = outlines.generate.choice(model, categories)
category = category_gen("Article: Apple announces new iPhone...")
category = model(
"Article: Apple announces new iPhone...",
Literal["technology", "business", "sports", "entertainment"],
)
# With confidence
class Classification(BaseModel):
label: Literal["positive", "negative", "neutral"]
confidence: float
classifier = outlines.generate.json(model, Classification)
result = classifier("Review: This product is okay, nothing special")
out = model("Review: This product is okay, nothing special", Classification)
result = Classification.model_validate_json(out)
```
### Pattern 3: Structured Forms
@@ -381,9 +379,6 @@ class UserProfile(BaseModel):
country: str
interests: list[str]
model = outlines.models.transformers("microsoft/Phi-3-mini-4k-instruct")
generator = outlines.generate.json(model, UserProfile)
prompt = """
Extract user profile from:
Name: Alice Johnson
@@ -394,7 +389,7 @@ Country: USA
Interests: hiking, photography, cooking
"""
profile = generator(prompt)
profile = UserProfile.model_validate_json(model(prompt, UserProfile, max_new_tokens=250))
print(profile.full_name)
print(profile.interests) # ["hiking", "photography", "cooking"]
```
@@ -402,6 +397,8 @@ print(profile.interests) # ["hiking", "photography", "cooking"]
### Pattern 4: Multi-Entity Extraction
```python
from typing import Literal
class Entity(BaseModel):
name: str
type: Literal["PERSON", "ORGANIZATION", "LOCATION"]
@@ -409,13 +406,10 @@ class Entity(BaseModel):
class DocumentEntities(BaseModel):
entities: list[Entity]
model = outlines.models.transformers("microsoft/Phi-3-mini-4k-instruct")
generator = outlines.generate.json(model, DocumentEntities)
text = "Tim Cook met with Satya Nadella at Microsoft headquarters in Redmond."
prompt = f"Extract entities from: {text}"
result = generator(prompt)
result = DocumentEntities.model_validate_json(model(prompt, DocumentEntities, max_new_tokens=300))
for entity in result.entities:
print(f"{entity.name} ({entity.type})")
```
@@ -429,11 +423,8 @@ class PythonFunction(BaseModel):
docstring: str
body: str
model = outlines.models.transformers("microsoft/Phi-3-mini-4k-instruct")
generator = outlines.generate.json(model, PythonFunction)
prompt = "Generate a Python function to calculate factorial"
func = generator(prompt)
func = PythonFunction.model_validate_json(model(prompt, PythonFunction, max_new_tokens=300))
print(f"def {func.function_name}({', '.join(func.parameters)}):")
print(f' """{func.docstring}"""')
@@ -443,29 +434,29 @@ print(f" {func.body}")
### Pattern 6: Batch Processing
```python
def batch_extract(texts: list[str], schema: type[BaseModel]):
"""Extract structured data from multiple texts."""
model = outlines.models.transformers("microsoft/Phi-3-mini-4k-instruct")
generator = outlines.generate.json(model, schema)
results = []
for text in texts:
result = generator(f"Extract from: {text}")
results.append(result)
return results
import outlines
from transformers import AutoModelForCausalLM, AutoTokenizer
from pydantic import BaseModel
class Person(BaseModel):
name: str
age: int
model = outlines.from_transformers(
AutoModelForCausalLM.from_pretrained("microsoft/Phi-3-mini-4k-instruct", device_map="auto"),
AutoTokenizer.from_pretrained("microsoft/Phi-3-mini-4k-instruct"),
)
texts = [
"John is 30 years old",
"Alice is 25 years old",
"Bob is 40 years old"
"Bob is 40 years old",
]
people = batch_extract(texts, Person)
# v1 accepts a list of prompts for batched generation
prompts = [f"Extract from: {t}" for t in texts]
outputs = model(prompts, Person, max_new_tokens=100)
people = [Person.model_validate_json(o) for o in outputs]
for person in people:
print(f"{person.name}: {person.age}")
```
@@ -476,58 +467,67 @@ for person in people:
```python
import outlines
from transformers import AutoModelForCausalLM, AutoTokenizer
MODEL_NAME = "microsoft/Phi-3-mini-4k-instruct"
# Basic usage
model = outlines.models.transformers("microsoft/Phi-3-mini-4k-instruct")
model = outlines.from_transformers(
AutoModelForCausalLM.from_pretrained(MODEL_NAME, device_map="auto"),
AutoTokenizer.from_pretrained(MODEL_NAME),
)
# GPU configuration
model = outlines.models.transformers(
"microsoft/Phi-3-mini-4k-instruct",
device="cuda",
model_kwargs={"torch_dtype": "float16"}
# GPU + dtype configuration is set on the HF model itself
import torch
model = outlines.from_transformers(
AutoModelForCausalLM.from_pretrained(MODEL_NAME, device_map="cuda", torch_dtype=torch.float16),
AutoTokenizer.from_pretrained(MODEL_NAME),
)
# Popular models
model = outlines.models.transformers("meta-llama/Llama-3.1-8B-Instruct")
model = outlines.models.transformers("mistralai/Mistral-7B-Instruct-v0.3")
model = outlines.models.transformers("Qwen/Qwen2.5-7B-Instruct")
for name in [
"meta-llama/Llama-3.1-8B-Instruct",
"mistralai/Mistral-7B-Instruct-v0.3",
"Qwen/Qwen2.5-7B-Instruct",
]:
model = outlines.from_transformers(
AutoModelForCausalLM.from_pretrained(name, device_map="auto"),
AutoTokenizer.from_pretrained(name),
)
```
### llama.cpp
```python
# Load GGUF model
model = outlines.models.llamacpp(
"./models/llama-3.1-8b.Q4_K_M.gguf",
n_ctx=4096, # Context window
n_gpu_layers=35, # GPU layers
n_threads=8 # CPU threads
)
import outlines
from llama_cpp import Llama
# Full GPU offload
model = outlines.models.llamacpp(
"./models/model.gguf",
n_gpu_layers=-1 # All layers on GPU
# Load GGUF model
llm = Llama(
"./models/llama-3.1-8b.Q4_K_M.gguf",
n_ctx=4096, # Context window
n_gpu_layers=35, # GPU layers
n_threads=8, # CPU threads
)
model = outlines.from_llamacpp(llm)
# Full GPU offload: set n_gpu_layers=-1 on the Llama object
```
### vLLM (Production)
```python
import outlines
from vllm import LLM
# Single GPU
model = outlines.models.vllm("meta-llama/Llama-3.1-8B-Instruct")
model = outlines.from_vllm(LLM("meta-llama/Llama-3.1-8B-Instruct"))
# Multi-GPU
model = outlines.models.vllm(
"meta-llama/Llama-3.1-70B-Instruct",
tensor_parallel_size=4 # 4 GPUs
)
model = outlines.from_vllm(LLM("meta-llama/Llama-3.1-70B-Instruct", tensor_parallel_size=4))
# With quantization
model = outlines.models.vllm(
"meta-llama/Llama-3.1-8B-Instruct",
quantization="awq" # Or "gptq"
)
model = outlines.from_vllm(LLM("meta-llama/Llama-3.1-8B-Instruct", quantization="awq"))
```
## Best Practices
@@ -615,15 +615,23 @@ class Article(BaseModel):
# Can succeed even if author/date missing
```
### 6. Always Validate JSON Output
```python
# v1 returns a JSON string for Pydantic/JSON output types.
result = model(prompt, Article) # str
article = Article.model_validate_json(result) # Article instance
```
## Comparison to Alternatives
| Feature | Outlines | Instructor | Guidance | LMQL |
|---------|----------|------------|----------|------|
| Pydantic Support | ✅ Native | ✅ Native | ❌ No | ❌ No |
| JSON Schema | ✅ Yes | ✅ Yes | ⚠️ Limited | ✅ Yes |
| Pydantic Support | ✅ Native | ✅ Native | ✅ Yes | ❌ No |
| JSON Schema | ✅ Yes | ✅ Yes | ✅ Yes | ✅ Yes |
| Regex Constraints | ✅ Yes | ❌ No | ✅ Yes | ✅ Yes |
| Local Models | ✅ Full | ⚠️ Limited | ✅ Full | ✅ Full |
| API Models | ⚠️ Limited | ✅ Full | ✅ Full | ✅ Full |
| API Models | ✅ Yes | ✅ Full | ✅ Yes | ✅ Full |
| Zero Overhead | ✅ Yes | ❌ No | ⚠️ Partial | ✅ Yes |
| Automatic Retrying | ❌ No | ✅ Yes | ❌ No | ❌ No |
| Learning Curve | Low | Low | Low | High |
@@ -648,19 +656,19 @@ class Article(BaseModel):
- **1.2-2x faster** than post-generation validation approaches
**Memory:**
- FSM compiled once per schema (cached)
- Automaton compiled once per output type (cached)
- Minimal runtime overhead
- Efficient with vLLM for high throughput
**Accuracy:**
- **100% valid outputs** (guaranteed by FSM)
- **100% valid outputs** (guaranteed by the constrained automaton)
- No retry loops needed
- Deterministic token filtering
## Resources
- **Documentation**: https://outlines-dev.github.io/outlines
- **GitHub**: https://github.com/outlines-dev/outlines (8k+ stars)
- **Documentation**: https://dottxt-ai.github.io/outlines/
- **GitHub**: https://github.com/dottxt-ai/outlines (12k+ stars)
- **Discord**: https://discord.gg/R9DSu34mGd
- **Blog**: https://blog.dottxt.co
@@ -16,10 +16,10 @@ Serverless GPU cloud for ML jobs and model APIs.
|---|---|
| Source | Optional — install with `hermes skills install official/mlops/modal` |
| Path | `optional-skills/mlops/modal` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `modal>=0.64.0` |
| Dependencies | `modal>=1.0` |
| Platforms | linux, macos, windows |
| Tags | `Infrastructure`, `Serverless`, `GPU`, `Cloud`, `Deployment`, `Modal` |
@@ -244,7 +244,6 @@ async def batch_predict(inputs: list[str]) -> list[dict]:
# Inputs automatically batched
return model.batch_predict(inputs)
```
## Secrets management
```bash
@@ -276,10 +275,10 @@ def hourly_job():
### Cold start mitigation
```python
@app.function(
container_idle_timeout=300, # Keep warm 5 min
allow_concurrent_inputs=10, # Handle concurrent requests
)
# Modal 1.0 autoscaler params: scaledown_window (was container_idle_timeout).
# Input concurrency moved to the @modal.concurrent decorator.
@app.function(scaledown_window=300) # Keep warm 5 min
@modal.concurrent(max_inputs=10) # Handle concurrent requests per container
def inference():
pass
```
@@ -321,14 +320,21 @@ def run_parallel():
memory=32768, # 32GB RAM
cpu=4, # 4 CPU cores
timeout=3600, # 1 hour max
container_idle_timeout=120,# Keep warm 2 min
scaledown_window=120, # Keep warm 2 min (was container_idle_timeout)
retries=3, # Retry on failure
concurrency_limit=10, # Max concurrent containers
max_containers=10, # Max concurrent containers (was concurrency_limit)
min_containers=1, # Keep N containers warm (was keep_warm)
)
def my_function():
pass
```
> **Modal 1.0 autoscaler renames** (see the [migration guide](https://modal.com/docs/guide/modal-1-0-migration)):
> - `container_idle_timeout` → `scaledown_window`
> - `concurrency_limit` → `max_containers`
> - `keep_warm` → `min_containers`
> - `allow_concurrent_inputs=N` → the `@modal.concurrent(max_inputs=N)` decorator
## Debugging
```python
@@ -344,7 +350,7 @@ if __name__ == "__main__":
| Issue | Solution |
|-------|----------|
| Cold start latency | Increase `container_idle_timeout`, use `@modal.enter()` |
| Cold start latency | Increase `scaledown_window`, use `@modal.enter()` |
| GPU OOM | Use larger GPU (`A100-80GB`), enable gradient checkpointing |
| Image build fails | Pin dependency versions, check CUDA compatibility |
| Timeout errors | Increase `timeout`, add checkpointing |
@@ -16,7 +16,7 @@ Curate LLM training data: dedupe, filter, PII redaction.
|---|---|
| Source | Optional — install with `hermes skills install official/mlops/nemo-curator` |
| Path | `optional-skills/mlops/nemo-curator` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `nemo-curator`, `cudf`, `dask`, `rapids` |
@@ -57,41 +57,55 @@ NVIDIA's toolkit for preparing high-quality training data for LLMs.
### Installation
```bash
# NeMo Curator 1.x installs with uv. Extras use hyphens (PyPI-normalized):
# text-cuda12 / text-cpu (and image/video/audio/math variants), or `all`.
# Text curation (CUDA 12)
uv pip install "nemo-curator[text_cuda12]"
uv pip install "nemo-curator[text-cuda12]"
# All modalities
uv pip install "nemo-curator[all_cuda12]"
uv pip install "nemo-curator[all]"
# CPU-only (slower)
uv pip install "nemo-curator[cpu]"
# CPU-only text (slower)
uv pip install "nemo-curator[text-cpu]"
```
### Basic text curation pipeline
> **Major version rewrite (1.x):** NeMo Curator was rewritten around a **Ray-based
> pipeline/stage architecture**. The old `DocumentDataset` + `nemo_curator.modules.*` /
> `ScoreFilter` / `Modify` call-the-object-on-a-dataset API from 0.x is gone. In 1.x you
> compose `ProcessingStage`s into a `Pipeline` and run it with an executor. The exact
> stage/import surface differs per modality — treat the examples in this skill below as
> **conceptual** (0.x-style) and follow the current
> [quickstart](https://github.com/NVIDIA-NeMo/Curator/blob/main/tutorials/quickstart.py)
> and [text guide](https://docs.nvidia.com/nemo/curator/latest/get-started/text) for the
> exact 1.x APIs rather than copying imports verbatim.
Shape of a 1.x pipeline (from the upstream quickstart):
```python
from nemo_curator import ScoreFilter, Modify
from nemo_curator.datasets import DocumentDataset
import pandas as pd
from nemo_curator.pipeline import Pipeline
from nemo_curator.stages.base import ProcessingStage
from nemo_curator.stages.resources import Resources
from nemo_curator.backends.xenna import XennaExecutor
from nemo_curator.core.client import RayClient
# Load data
df = pd.DataFrame({"text": ["Good document", "Bad doc", "Excellent text"]})
dataset = DocumentDataset(df)
# 1. Define/compose stages (load -> filter -> dedupe -> classify -> write).
# Each stage declares its own Resources (CPU cores, GPU memory, replicas).
pipeline = Pipeline(name="curation", stages=[...])
# Quality filtering
def quality_score(doc):
return len(doc["text"].split()) > 5 # Filter short docs
filtered = ScoreFilter(quality_score)(dataset)
# Deduplication
from nemo_curator.modules import ExactDuplicates
deduped = ExactDuplicates()(filtered)
# Save
deduped.to_parquet("curated_data/")
# 2. Run it with an executor (Ray-backed).
client = RayClient()
client.start()
pipeline.run(XennaExecutor())
client.stop()
```
The 0.x-style snippets in the sections that follow illustrate the *concepts* (quality
filtering, exact/fuzzy/semantic dedup, PII redaction, classifier filtering). For runnable
1.x code, map each concept onto the corresponding stage from the modality guide.
## Data curation pipeline
### Stage 1: Quality filtering
@@ -395,7 +409,7 @@ cluster.close()
## Resources
- **GitHub**: https://github.com/NVIDIA/NeMo-Curator ⭐ 500+
- **Docs**: https://docs.nvidia.com/nemo-framework/user-guide/latest/datacuration/
- **Version**: 0.4.0+
- **GitHub**: https://github.com/NVIDIA-NeMo/Curator
- **Docs**: https://docs.nvidia.com/nemo/curator/latest/
- **Version**: 1.2.0 (1.x is a Ray-based pipeline rewrite — see the quickstart before copying 0.x snippets)
- **License**: Apache 2.0
@@ -22,7 +22,7 @@ OBLITERATUS: abliterate LLM refusals (diff-in-means).
| Dependencies | `obliteratus`, `torch`, `transformers`, `bitsandbytes`, `accelerate`, `safetensors` |
| Platforms | linux, macos |
| Tags | `Abliteration`, `Uncensoring`, `Refusal-Removal`, `LLM`, `Weight-Projection`, `SVD`, `Mechanistic-Interpretability`, `HuggingFace`, `Model-Surgery` |
| Related skills | [`serving-llms-vllm`](/docs/user-guide/skills/bundled/mlops/mlops-inference-vllm), [`llama-cpp`](/docs/user-guide/skills/bundled/mlops/mlops-inference-llama-cpp), [`huggingface-tokenizers`](/docs/user-guide/skills/optional/mlops/mlops-huggingface-tokenizers) |
| Related skills | [`serving-llms-vllm`](/docs/user-guide/skills/bundled/mlops/mlops-inference-serving-llms-vllm), [`llama-cpp`](/docs/user-guide/skills/bundled/mlops/mlops-inference-llama-cpp), [`huggingface-tokenizers`](/docs/user-guide/skills/optional/mlops/mlops-huggingface-tokenizers) |
## Reference: full SKILL.md
@@ -1,14 +1,14 @@
---
title: "Pinecone — Managed vector database for production AI applications"
title: "Pinecone — Managed vector DB for production RAG and search"
sidebar_label: "Pinecone"
description: "Managed vector database for production AI applications"
description: "Managed vector DB for production RAG and search"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
# Pinecone
Managed vector database for production AI applications. Fully managed, auto-scaling, with hybrid search (dense + sparse), metadata filtering, and namespaces. Low latency (&lt;100ms p95). Use for production RAG, recommendation systems, or semantic search at scale. Best for serverless, managed infrastructure.
Managed vector DB for production RAG and search.
## Skill metadata
@@ -16,10 +16,10 @@ Managed vector database for production AI applications. Fully managed, auto-scal
|---|---|
| Source | Optional — install with `hermes skills install official/mlops/pinecone` |
| Path | `optional-skills/mlops/pinecone` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `pinecone-client` |
| Dependencies | `pinecone` |
| Platforms | linux, macos, windows |
| Tags | `RAG`, `Pinecone`, `Vector Database`, `Managed Service`, `Serverless`, `Hybrid Search`, `Production`, `Auto-Scaling`, `Low Latency`, `Recommendations` |
@@ -59,9 +59,11 @@ The vector database for production AI applications.
### Installation
```bash
pip install pinecone-client
pip install pinecone
```
> Note: the old `pinecone-client` package is deprecated. Install `pinecone` (v5+; current 9.x). The import stays `from pinecone import Pinecone`.
### Basic usage
```python
@@ -243,14 +245,31 @@ index.upsert(vectors=[
])
# Hybrid query
# NOTE: index.query() does NOT accept an `alpha` kwarg. Pinecone stores a
# single sparse-dense vector, so weighting must be applied by pre-scaling the
# query vectors before sending them. Use the hybrid_score_norm helper below
# (alpha * dense + (1 - alpha) * sparse; alpha=1 → pure dense, 0 → pure sparse).
def hybrid_score_norm(dense, sparse, alpha: float):
"""Scale dense/sparse query vectors for weighted hybrid search."""
if not 0 <= alpha <= 1:
raise ValueError("alpha must be between 0 and 1")
scaled_sparse = {
"indices": sparse["indices"],
"values": [v * (1 - alpha) for v in sparse["values"]],
}
return [v * alpha for v in dense], scaled_sparse
hdense, hsparse = hybrid_score_norm(
dense=[0.1, 0.2, ...],
sparse={"indices": [10, 45], "values": [0.5, 0.3]},
alpha=0.5, # 0=sparse, 1=dense, 0.5=balanced
)
results = index.query(
vector=[0.1, 0.2, ...],
sparse_vector={
"indices": [10, 45],
"values": [0.5, 0.3]
},
vector=hdense,
sparse_vector=hsparse,
top_k=5,
alpha=0.5 # 0=sparse, 1=dense, 0.5=hybrid
)
```
@@ -16,10 +16,10 @@ Vector search engine for production RAG systems.
|---|---|
| Source | Optional — install with `hermes skills install official/mlops/qdrant` |
| Path | `optional-skills/mlops/qdrant` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `qdrant-client>=1.12.0` |
| Dependencies | `qdrant-client>=1.14.0` |
| Platforms | linux, macos, windows |
| Tags | `RAG`, `Vector Search`, `Qdrant`, `Semantic Search`, `Embeddings`, `Similarity Search`, `HNSW`, `Production`, `Distributed` |
@@ -106,17 +106,17 @@ client.upsert(
]
)
# Search with filtering
results = client.search(
# Search with filtering (query_points is the current API; client.search is removed in qdrant-client 1.14+)
response = client.query_points(
collection_name="documents",
query_vector=[0.15, 0.25, ...],
query=[0.15, 0.25, ...],
query_filter={
"must": [{"key": "category", "match": {"value": "tech"}}]
},
limit=10
)
for point in results:
for point in response.points:
print(f"ID: {point.id}, Score: {point.score}, Payload: {point.payload}")
```
@@ -186,14 +186,15 @@ print(f"Points: {info.points_count}, Vectors: {info.vectors_count}")
### Basic search
```python
# Simple nearest neighbor search
results = client.search(
# Simple nearest neighbor search (returns a QueryResponse; use .points)
response = client.query_points(
collection_name="documents",
query_vector=[0.1, 0.2, ...],
query=[0.1, 0.2, ...],
limit=10,
with_payload=True,
with_vectors=False # Don't return vectors (faster)
)
results = response.points
```
### Filtered search
@@ -202,9 +203,9 @@ results = client.search(
from qdrant_client.models import Filter, FieldCondition, MatchValue, Range
# Complex filtering
results = client.search(
response = client.query_points(
collection_name="documents",
query_vector=query_embedding,
query=query_embedding,
query_filter=Filter(
must=[
FieldCondition(key="category", match=MatchValue(value="tech")),
@@ -215,12 +216,12 @@ results = client.search(
]
),
limit=10
)
).points
# Shorthand filter syntax
results = client.search(
response = client.query_points(
collection_name="documents",
query_vector=query_embedding,
query=query_embedding,
query_filter={
"must": [
{"key": "category", "match": {"value": "tech"}},
@@ -228,23 +229,27 @@ results = client.search(
]
},
limit=10
)
).points
```
### Batch search
```python
from qdrant_client.models import SearchRequest
from qdrant_client.models import QueryRequest
# Multiple queries in one request
results = client.search_batch(
# Multiple queries in one request (search_batch is replaced by query_batch_points)
responses = client.query_batch_points(
collection_name="documents",
requests=[
SearchRequest(vector=[0.1, ...], limit=5),
SearchRequest(vector=[0.2, ...], limit=5, filter={"must": [...]}),
SearchRequest(vector=[0.3, ...], limit=10)
QueryRequest(query=[0.1, ...], limit=5),
QueryRequest(query=[0.2, ...], limit=5, filter={"must": [...]}),
QueryRequest(query=[0.3, ...], limit=10)
]
)
# Each element is a QueryResponse; use .points
for resp in responses:
for point in resp.points:
print(point.id, point.score)
```
## RAG integration
@@ -285,12 +290,12 @@ client.upsert(collection_name="knowledge_base", points=points)
# RAG retrieval
def retrieve(query: str, top_k: int = 5) -> list[dict]:
query_vector = encoder.encode(query).tolist()
results = client.search(
response = client.query_points(
collection_name="knowledge_base",
query_vector=query_vector,
query=query_vector,
limit=top_k
)
return [{"text": r.payload["text"], "score": r.score} for r in results]
return [{"text": r.payload["text"], "score": r.score} for r in response.points]
# Use in RAG pipeline
context = retrieve("What is Python?")
@@ -351,12 +356,14 @@ client.upsert(
]
)
# Search specific vector
results = client.search(
# Search specific named vector (pass the vector name via `using`)
response = client.query_points(
collection_name="hybrid_search",
query_vector=("dense", query_dense), # Specify which vector
query=query_dense,
using="dense", # Specify which named vector to search
limit=10
)
results = response.points
```
### Sparse vectors (BM25, SPLADE)
@@ -397,12 +404,13 @@ client.create_collection(
)
# Search with rescoring
results = client.search(
response = client.query_points(
collection_name="quantized",
query_vector=query,
query=query,
search_params={"quantization": {"rescore": True}}, # Rescore top results
limit=10
)
results = response.points
```
## Payload indexing
@@ -510,5 +518,5 @@ client = QdrantClient(
- **Docs**: https://qdrant.tech/documentation/
- **Python Client**: https://github.com/qdrant/qdrant-client
- **Cloud**: https://cloud.qdrant.io
- **Version**: 1.12.0+
- **Version**: 1.14.0+
- **License**: Apache 2.0
@@ -16,7 +16,7 @@ Train sparse autoencoders to interpret model features.
|---|---|
| Source | Optional — install with `hermes skills install official/mlops/saelens` |
| Path | `optional-skills/mlops/saelens` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `sae-lens>=6.0.0`, `transformer-lens>=2.0.0`, `torch>=2.0.0` |
@@ -95,11 +95,14 @@ from sae_lens import SAE
# 1. Load model and pre-trained SAE
model = HookedTransformer.from_pretrained("gpt2-small", device="cuda")
sae, cfg_dict, sparsity = SAE.from_pretrained(
# In sae-lens v6, SAE.from_pretrained() returns JUST the SAE (not a tuple).
sae = SAE.from_pretrained(
release="gpt2-small-res-jb",
sae_id="blocks.8.hook_resid_pre",
device="cuda"
)
# If you also need the cfg dict and feature sparsity, use:
# sae, cfg_dict, sparsity = SAE.from_pretrained_with_cfg_and_sparsity(...)
# 2. Get model activations
tokens = model.to_tokens("The capital of France is Paris")
@@ -141,24 +144,33 @@ reconstruction_error = (activations - reconstructed).norm()
### Step-by-Step
```python
from sae_lens import SAE, LanguageModelSAERunnerConfig, SAETrainingRunner
from sae_lens import (
LanguageModelSAETrainingRunner,
LanguageModelSAERunnerConfig,
StandardTrainingSAEConfig,
LoggingConfig,
)
# 1. Configure training
# 1. Configure training (v6 uses a NESTED config: SAE-specific options live in a
# `sae=` sub-config, and logging options live in a `logger=` sub-config).
# Note: `architecture`, `d_sae`, `l1_coefficient` etc. are now on the SAE sub-config,
# and legacy flat options like `hook_layer`, `activation_fn`, `log_to_wandb` were removed.
cfg = LanguageModelSAERunnerConfig(
# Model
model_name="gpt2-small",
hook_name="blocks.8.hook_resid_pre",
hook_layer=8,
d_in=768, # Model dimension
# SAE architecture + sparsity (nested)
sae=StandardTrainingSAEConfig(
d_in=768, # Model dimension
d_sae=768 * 8, # Expansion factor of 8
l1_coefficient=8e-5, # Sparsity penalty
apply_b_dec_to_input=True,
normalize_activations="expected_average_only_in",
),
# SAE architecture
architecture="standard", # or "gated", "topk"
d_sae=768 * 8, # Expansion factor of 8
activation_fn="relu",
# Data-generating function (model + hook point)
model_name="gpt2-small",
hook_name="blocks.8.hook_resid_pre", # layer is inferred from hook_name (no hook_layer)
# Training
lr=4e-4,
l1_coefficient=8e-5, # Sparsity penalty
l1_warm_up_steps=1000,
train_batch_size_tokens=4096,
training_tokens=100_000_000,
@@ -167,9 +179,11 @@ cfg = LanguageModelSAERunnerConfig(
dataset_path="monology/pile-uncopyrighted",
context_size=128,
# Logging
log_to_wandb=True,
wandb_project="sae-training",
# Logging (nested)
logger=LoggingConfig(
log_to_wandb=True,
wandb_project="sae-training",
),
# Checkpointing
checkpoint_path="checkpoints",
@@ -177,7 +191,7 @@ cfg = LanguageModelSAERunnerConfig(
)
# 2. Train
trainer = SAETrainingRunner(cfg)
trainer = LanguageModelSAETrainingRunner(cfg) # SAETrainingRunner still works as an alias
sae = trainer.run()
# 3. Evaluate
@@ -185,6 +199,12 @@ print(f"L0 (avg active features): {trainer.metrics['l0']}")
print(f"CE Loss Recovered: {trainer.metrics['ce_loss_score']}")
```
> **v6 migration note:** For other SAE types swap the `sae=` sub-config —
> `GatedTrainingSAEConfig`, `TopKTrainingSAEConfig` (set `k` directly), or
> `JumpReLUTrainingSAEConfig` (uses `l0_coefficient`). Legacy flat options
> (`architecture`, `expansion_factor`, `hook_layer`, `activation_fn`/`activation_fn_kwargs`,
> `use_ghost_grads`, ghost grads, b_dec/decoder init options) were removed in v6.
### Key Hyperparameters
| Parameter | Typical Value | Effect |
@@ -222,7 +242,7 @@ from sae_lens import SAE
import torch
model = HookedTransformer.from_pretrained("gpt2-small", device="cuda")
sae, _, _ = SAE.from_pretrained(
sae = SAE.from_pretrained( # v6 returns just the SAE
release="gpt2-small-res-jb",
sae_id="blocks.8.hook_resid_pre",
device="cuda"
@@ -296,47 +316,57 @@ for idx, val in zip(top_features.indices, top_features.values):
## Common Issues & Solutions
> All examples below use the v6 nested config: SAE-specific options go in the `sae=`
> sub-config (`StandardTrainingSAEConfig` / `TopKTrainingSAEConfig` / etc.), training
> knobs stay on the top-level `LanguageModelSAERunnerConfig`.
### Issue: High dead feature ratio
```python
# WRONG: No warm-up, features die early
from sae_lens import LanguageModelSAERunnerConfig, StandardTrainingSAEConfig
# WRONG: no warm-up, features die early
cfg = LanguageModelSAERunnerConfig(
l1_coefficient=1e-4,
sae=StandardTrainingSAEConfig(d_in=768, d_sae=768*8, l1_coefficient=1e-4),
l1_warm_up_steps=0, # Bad!
)
# RIGHT: Warm-up L1 penalty
# RIGHT: warm up the L1 penalty (v6 removed ghost grads; warm-up is the lever now)
cfg = LanguageModelSAERunnerConfig(
l1_coefficient=8e-5,
sae=StandardTrainingSAEConfig(d_in=768, d_sae=768*8, l1_coefficient=8e-5),
l1_warm_up_steps=1000, # Gradually increase
use_ghost_grads=True, # Revive dead features
)
```
### Issue: Poor reconstruction (low CE recovery)
```python
# Reduce sparsity penalty
# Reduce sparsity penalty and/or add capacity (both on the SAE sub-config)
cfg = LanguageModelSAERunnerConfig(
l1_coefficient=5e-5, # Lower = better reconstruction
d_sae=768 * 16, # More capacity
sae=StandardTrainingSAEConfig(
d_in=768,
d_sae=768 * 16, # More capacity
l1_coefficient=5e-5, # Lower = better reconstruction
),
)
```
### Issue: Features not interpretable
```python
from sae_lens import LanguageModelSAERunnerConfig, StandardTrainingSAEConfig, TopKTrainingSAEConfig
# Increase sparsity (higher L1)
cfg = LanguageModelSAERunnerConfig(
l1_coefficient=1e-4, # Higher = sparser, more interpretable
sae=StandardTrainingSAEConfig(d_in=768, d_sae=768*8, l1_coefficient=1e-4),
)
# Or use TopK architecture
# Or use a TopK SAE (k is set directly in v6, not via activation_fn_kwargs)
cfg = LanguageModelSAERunnerConfig(
architecture="topk",
activation_fn_kwargs={"k": 50}, # Exactly 50 active features
sae=TopKTrainingSAEConfig(d_in=768, d_sae=768*8, k=50), # Exactly 50 active features
)
```
### Issue: Memory errors during training
```python
cfg = LanguageModelSAERunnerConfig(
sae=StandardTrainingSAEConfig(d_in=768, d_sae=768*8, l1_coefficient=8e-5),
train_batch_size_tokens=2048, # Reduce batch size
store_batch_size_prompts=4, # Fewer prompts in buffer
n_batches_in_buffer=8, # Smaller activation buffer
@@ -358,8 +388,10 @@ Browse pre-trained SAE features at [neuronpedia.org](https://neuronpedia.org):
| Class | Purpose |
|-------|---------|
| `SAE` | Sparse Autoencoder model |
| `LanguageModelSAERunnerConfig` | Training configuration |
| `SAETrainingRunner` | Training loop manager |
| `LanguageModelSAERunnerConfig` | Top-level training configuration (nests `sae=` and `logger=`) |
| `StandardTrainingSAEConfig` / `TopKTrainingSAEConfig` / `GatedTrainingSAEConfig` / `JumpReLUTrainingSAEConfig` | SAE-type-specific sub-configs (v6) |
| `LoggingConfig` | Logging/W&B sub-config (v6) |
| `LanguageModelSAETrainingRunner` | Training loop manager (alias: `SAETrainingRunner`) |
| `ActivationsStore` | Activation collection and batching |
| `HookedSAETransformer` | TransformerLens + SAE integration |
@@ -398,10 +430,10 @@ For detailed API documentation, tutorials, and advanced usage, see the `referenc
| **TopK** | Exactly K active features | Consistent sparsity |
```python
# TopK SAE (exactly 50 features active)
from sae_lens import LanguageModelSAERunnerConfig, TopKTrainingSAEConfig
# TopK SAE (exactly 50 features active) — `k` is set on the SAE sub-config in v6
cfg = LanguageModelSAERunnerConfig(
architecture="topk",
activation_fn="topk",
activation_fn_kwargs={"k": 50},
sae=TopKTrainingSAEConfig(d_in=768, d_sae=768*8, k=50),
)
```
@@ -16,7 +16,7 @@ High-throughput LLM inference on NVIDIA GPUs.
|---|---|
| Source | Optional — install with `hermes skills install official/mlops/tensorrt-llm` |
| Path | `optional-skills/mlops/tensorrt-llm` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `tensorrt-llm`, `torch` |
@@ -57,13 +57,15 @@ NVIDIA's open-source library for optimizing LLM inference with state-of-the-art
### Installation
```bash
# Docker (recommended)
docker pull nvidia/tensorrt_llm:latest
# Docker (recommended) — images are on NGC (nvcr.io), not Docker Hub.
# Replace x.y.z with the desired version (e.g. 1.2.1). Browse tags on NGC:
# https://catalog.ngc.nvidia.com/orgs/nvidia/teams/tensorrt-llm/containers/release/tags
docker pull nvcr.io/nvidia/tensorrt-llm/release:x.y.z
# pip install
pip install tensorrt_llm==1.2.0rc3
# pip install (current stable GA)
pip install tensorrt_llm
# Requires CUDA 13.0.0, TensorRT 10.13.2, Python 3.10-3.12
# Requires CUDA 13.2.1, TensorRT 10.x, Python 3.10-3.12
```
### Basic inference
@@ -16,7 +16,7 @@ Pretrain LLMs at scale with PyTorch 4D parallelism.
|---|---|
| Source | Optional — install with `hermes skills install official/mlops/torchtitan` |
| Path | `optional-skills/mlops/torchtitan` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `torch>=2.6.0`, `torchtitan>=0.2.0`, `torchao>=0.5.0` |
@@ -54,7 +54,9 @@ python scripts/download_hf_assets.py --repo_id meta-llama/Llama-3.1-8B --assets
**Start training on 8 GPUs**:
```bash
CONFIG_FILE="./torchtitan/models/llama3/train_configs/llama3_8b.toml" ./run_train.sh
# Configs are selected by name from the Python config registry
# (torchtitan/models/llama3/config_registry.py), not by TOML path
MODULE=llama3 CONFIG=llama3_8b ./run_train.sh
```
## Common workflows
@@ -82,10 +84,16 @@ python scripts/download_hf_assets.py \
**Step 2: Configure training**
Edit or create a TOML config file:
In torchtitan's current layout, run configs are defined in a Python **config registry**
(`torchtitan/models/llama3/config_registry.py`) and selected by name via `CONFIG=<name>`
(or `--config <name>`). To customize, register your own config in the registry, or override
individual fields on the command line (e.g. `--optimizer.lr 3e-4 --training.steps 1000`).
The equivalent settings for an 8B run look like this (shown as fields; set them in the
registry entry or as `--section.key value` overrides):
```toml
# llama3_8b_custom.toml
# fields for a llama3 8B run (register in config_registry.py or pass as --overrides)
[job]
dump_folder = "./outputs"
description = "Llama 3.1 8B training"
@@ -125,13 +133,16 @@ interval = 500
**Step 3: Launch training**
```bash
# 8 GPUs on single node
CONFIG_FILE="./llama3_8b_custom.toml" ./run_train.sh
# 8 GPUs on single node (config selected by name from the registry)
MODULE=llama3 CONFIG=llama3_8b ./run_train.sh
# Or explicitly with torchrun
# Override individual fields on the command line
MODULE=llama3 CONFIG=llama3_8b ./run_train.sh --optimizer.lr 3e-4 --training.steps 1000
# Or explicitly with torchrun (run_train.sh wraps this)
torchrun --nproc_per_node=8 \
-m torchtitan.train \
--job.config_file ./llama3_8b_custom.toml
--module llama3 --config llama3_8b
```
**Step 4: Monitor and checkpoint**
@@ -177,7 +188,7 @@ srun torchrun \
--rdzv_backend=c10d \
--rdzv_endpoint=$MASTER_ADDR:$MASTER_PORT \
-m torchtitan.train \
--job.config_file ./llama3_70b.toml
--module llama3 --config llama3_70b
```
**Step 3: Submit job**
@@ -209,16 +220,28 @@ USE_CPP=0 pip install git+https://github.com/pytorch/ao.git
**Step 2: Configure Float8**
Add to your TOML config:
In the current torchtitan, Float8 is applied at config time via the `quantization`
parameter in your `model_registry()` call inside the config registry (not via a
`[quantize.linear.float8]` TOML section). Add a `Float8LinearConverter.Config`:
```python
# in torchtitan/models/llama3/config_registry.py (your model_registry(...) call)
from torchtitan.components.quantization import Float8LinearConverter
model_spec = model_registry(
"8B",
quantization=[
Float8LinearConverter.Config(
recipe_name="rowwise", # or "rowwise_with_gw_hp"
filter_fqns=["output"], # skip layers too small to benefit
model_compile_enabled=True, # requires torch.compile for competitive perf
),
],
)
```
Enable `torch.compile` in your run config too:
```toml
[model]
converters = ["quantize.linear.float8"]
[quantize.linear.float8]
enable_fsdp_float8_all_gather = true
precompute_float8_dynamic_scale_for_fsdp = true
filter_fqns = ["output"] # Exclude output layer
[compile]
enable = true
components = ["model", "loss"]
@@ -227,10 +250,8 @@ components = ["model", "loss"]
**Step 3: Launch with compile**
```bash
CONFIG_FILE="./llama3_8b.toml" ./run_train.sh \
--model.converters="quantize.linear.float8" \
--quantize.linear.float8.enable_fsdp_float8_all_gather \
--compile.enable
# Float8 config is baked into the registered config; just select it and enable compile
MODULE=llama3 CONFIG=llama3_8b ./run_train.sh --compile.enable
```
### Workflow 4: 4D parallelism for 405B models
@@ -246,7 +267,7 @@ CONFIG_FILE="./llama3_8b.toml" ./run_train.sh \
Required for consistent initialization across PP stages:
```bash
NGPU=1 CONFIG_FILE=./llama3_405b.toml ./run_train.sh \
NGPU=1 MODULE=llama3 CONFIG=llama3_405b ./run_train.sh \
--checkpoint.enable \
--checkpoint.create_seed_checkpoint \
--parallelism.data_parallel_shard_degree 1 \
@@ -274,7 +295,7 @@ seq_len = 8192
# 64 nodes x 8 GPUs = 512 GPUs
srun torchrun --nnodes=64 --nproc_per_node=8 \
-m torchtitan.train \
--job.config_file ./llama3_405b.toml
--module llama3 --config llama3_405b
```
## When to use vs alternatives
@@ -321,10 +342,15 @@ export TORCH_NCCL_AVOID_RECORD_STREAMS=1
**Issue: Float8 training not faster**
Float8 only benefits large GEMMs. Filter small layers:
```toml
[quantize.linear.float8]
filter_fqns = ["attention.wk", "attention.wv", "output", "auto_filter_small_kn"]
Float8 only benefits large GEMMs. Filter small layers via the converter's `filter_fqns`:
```python
from torchtitan.components.quantization import Float8LinearConverter
Float8LinearConverter.Config(
# add "auto_filter_small_kn" to auto-skip layers too small to benefit
filter_fqns=["attention.wk", "attention.wv", "output", "auto_filter_small_kn"],
model_compile_enabled=True,
)
```
**Issue: Checkpoint loading fails after parallelism change**
@@ -1,14 +1,14 @@
---
title: "Fine Tuning With Trl — TRL: SFT, DPO, PPO, GRPO, reward modeling for LLM RLHF"
title: "Fine Tuning With Trl — TRL: SFT, DPO, GRPO, RLOO reward modeling for LLM RLHF"
sidebar_label: "Fine Tuning With Trl"
description: "TRL: SFT, DPO, PPO, GRPO, reward modeling for LLM RLHF"
description: "TRL: SFT, DPO, GRPO, RLOO reward modeling for LLM RLHF"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
# Fine Tuning With Trl
TRL: SFT, DPO, PPO, GRPO, reward modeling for LLM RLHF.
TRL: SFT, DPO, GRPO, RLOO reward modeling for LLM RLHF.
## Skill metadata
@@ -16,12 +16,12 @@ TRL: SFT, DPO, PPO, GRPO, reward modeling for LLM RLHF.
|---|---|
| Source | Optional — install with `hermes skills install official/mlops/trl-fine-tuning` |
| Path | `optional-skills/mlops/training/trl-fine-tuning` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | Orchestra Research |
| License | MIT |
| Dependencies | `trl`, `transformers`, `datasets`, `peft`, `accelerate`, `torch` |
| Platforms | linux, macos, windows |
| Tags | `Post-Training`, `TRL`, `Reinforcement Learning`, `Fine-Tuning`, `SFT`, `DPO`, `PPO`, `GRPO`, `RLHF`, `Preference Alignment`, `HuggingFace` |
| Tags | `Post-Training`, `TRL`, `Reinforcement Learning`, `Fine-Tuning`, `SFT`, `DPO`, `GRPO`, `RLOO`, `RLHF`, `Preference Alignment`, `HuggingFace` |
## Reference: full SKILL.md
@@ -67,17 +67,23 @@ trainer.train()
## Common workflows
### Workflow 1: Full RLHF pipeline (SFT → Reward Model → PPO)
### Workflow 1: Full RLHF pipeline (SFT → Reward Model → RLOO)
Complete pipeline from base model to human-aligned model.
> **Note (TRL 1.x):** PPO has been **removed** from TRL — `PPOTrainer`, `PPOConfig`, and
> `python -m trl.scripts.ppo` no longer exist. Use an online-RL trainer TRL still ships:
> **RLOO** (`RLOOTrainer` / `trl rloo`) is the closest drop-in for a reward-model-driven
> RLHF pipeline, and **GRPO** (`GRPOTrainer` / `trl grpo`, see Workflow 3) is the
> memory-efficient alternative. The step below uses RLOO.
Copy this checklist:
```
RLHF Training:
- [ ] Step 1: Supervised fine-tuning (SFT)
- [ ] Step 2: Train reward model
- [ ] Step 3: PPO reinforcement learning
- [ ] Step 3: RLOO reinforcement learning
- [ ] Step 4: Evaluate aligned model
```
@@ -112,7 +118,7 @@ trainer = SFTTrainer(
model=model,
args=training_args,
train_dataset=dataset,
tokenizer=tokenizer
processing_class=tokenizer
)
trainer.train()
trainer.save_model()
@@ -155,19 +161,46 @@ trainer.train()
trainer.save_model()
```
**Step 3: PPO reinforcement learning**
**Step 3: RLOO reinforcement learning**
Optimize policy using reward model:
Optimize policy using the reward model. PPO was removed in TRL 1.x; use the RLOO CLI
(`trl rloo`) with the trained reward model passed via `--reward_model_name_or_path`:
```bash
python -m trl.scripts.ppo \
trl rloo \
--model_name_or_path Qwen2.5-0.5B-SFT \
--reward_model_path Qwen2.5-0.5B-Reward \
--reward_model_name_or_path Qwen2.5-0.5B-Reward \
--dataset_name trl-internal-testing/descriptiveness-sentiment-trl-style \
--output_dir Qwen2.5-0.5B-PPO \
--output_dir Qwen2.5-0.5B-RLOO \
--learning_rate 3e-6 \
--per_device_train_batch_size 64 \
--total_episodes 10000
--num_generations 4
```
Equivalent Python (`RLOOTrainer` / `RLOOConfig`):
```python
from trl import RLOOTrainer, RLOOConfig
from transformers import AutoModelForSequenceClassification, AutoTokenizer
reward_model = AutoModelForSequenceClassification.from_pretrained(
"Qwen2.5-0.5B-Reward", num_labels=1
)
config = RLOOConfig(
output_dir="Qwen2.5-0.5B-RLOO",
per_device_train_batch_size=64,
learning_rate=3e-6,
num_generations=4,
)
trainer = RLOOTrainer(
model="Qwen2.5-0.5B-SFT",
reward_funcs=reward_model, # a reward model (or a callable reward function)
args=config,
train_dataset=dataset, # prompt-only dataset
processing_class=tokenizer,
)
trainer.train()
```
**Step 4: Evaluate**
@@ -176,7 +209,7 @@ python -m trl.scripts.ppo \
from transformers import pipeline
# Load aligned model
generator = pipeline("text-generation", model="Qwen2.5-0.5B-PPO")
generator = pipeline("text-generation", model="Qwen2.5-0.5B-RLOO")
# Test
prompt = "Explain quantum computing to a 10-year-old"
@@ -365,15 +398,15 @@ trl grpo \
**Use TRL when:**
- Need to align model with human preferences
- Have preference data (chosen/rejected pairs)
- Want to use reinforcement learning (PPO, GRPO)
- Want to use reinforcement learning (RLOO, GRPO)
- Need reward model training
- Doing RLHF (full pipeline)
**Method selection**:
- **SFT**: Have prompt-completion pairs, want basic instruction following
- **DPO**: Have preferences, want simple alignment (no reward model needed)
- **PPO**: Have reward model, need maximum control over RL
- **GRPO**: Memory-constrained, want online RL
- **RLOO**: Have a reward model, want online RL (the reward-model-driven RLHF path; PPO was removed in TRL 1.x)
- **GRPO**: Memory-constrained, want online RL with reward functions
- **Reward Model**: Building RLHF pipeline, need to score generations
**Use alternatives instead:**
@@ -428,13 +461,15 @@ print(dataset[0])
# Should have clear chosen > rejected
```
**Issue: PPO training unstable**
**Issue: Online RL (RLOO/GRPO) training unstable**
Adjust KL coefficient:
Adjust the KL/beta regularization toward the reference policy:
```python
config = PPOConfig(
kl_coef=0.1, # Increase from 0.05
cliprange=0.1 # Reduce from 0.2
from trl import RLOOConfig
config = RLOOConfig(
beta=0.05, # KL coefficient toward the reference model (increase for stability)
num_generations=4, # more samples per prompt = lower-variance advantage estimates
)
```
@@ -456,7 +491,7 @@ config = PPOConfig(
- **VRAM**: Depends on model and method
- SFT 7B: 16GB (with LoRA)
- DPO 7B: 24GB (stores reference model)
- PPO 7B: 40GB (policy + reward model)
- RLOO 7B: 40GB (policy + reward model)
- GRPO 7B: 24GB (more memory efficient)
- **Multi-GPU**: Supported via `accelerate`
- **Mixed precision**: BF16 recommended (A100/H100)
@@ -1,7 +1,7 @@
---
title: "Here.Now — Publish sites to {slug}"
title: "Here.Now — Publish sites to {slug}.here.now and store files in Drives"
sidebar_label: "Here.Now"
description: "Publish sites to {slug}"
description: "Publish sites to {slug}.here.now and store files in Drives"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
@@ -0,0 +1,125 @@
---
title: "Pinecone Research — Agent RAG and long-term memory with Pinecone"
sidebar_label: "Pinecone Research"
description: "Agent RAG and long-term memory with Pinecone"
---
{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */}
# Pinecone Research
Agent RAG and long-term memory with Pinecone.
## Skill metadata
| | |
|---|---|
| Source | Optional — install with `hermes skills install official/research/pinecone-research` |
| Path | `optional-skills/research/pinecone-research` |
| Version | `1.0.0` |
| Author | immuhammadfurqan |
| License | MIT |
| Dependencies | `pinecone-client`, `langchain-pinecone` |
| Platforms | linux, macos, windows |
| Tags | `RAG`, `Pinecone`, `Memory`, `Research`, `Vector Database`, `Agent`, `Retrieval` |
## Reference: full SKILL.md
:::info
The following is the complete skill definition that Hermes loads when this skill is triggered. This is what the agent sees as instructions when the skill is active.
:::
# Pinecone Research — Agent RAG & Long-Term Memory
Use Pinecone as a retrieval-augmented generation (RAG) backend for agent
conversations: persist embeddings, retrieve relevant context from past
sessions, and build long-term memory.
## When to use this skill
**Use when:**
- Building agent RAG pipelines with Pinecone as the vector store
- Need persistent long-term memory across agent sessions
- Combining retrieval with agent tool use
- Researching or prototyping semantic search workflows
**Use the mlops/pinecone skill instead when:**
- Need a general Pinecone reference (index management, CRUD, hybrid search)
- Working on production infrastructure without agent integration
## Quick start
### Setup
```bash
pip install pinecone-client langchain-pinecone langchain-openai
```
Set your API key:
```bash
export PINECONE_API_KEY="your-api-key"
```
### Basic RAG pipeline
```python
from pinecone import Pinecone, ServerlessSpec
from langchain_pinecone import PineconeVectorStore
from langchain_openai import OpenAIEmbeddings
# Initialize Pinecone
pc = Pinecone(api_key=os.environ["PINECONE_API_KEY"])
# Create or connect to index
index_name = "agent-memory"
if index_name not in [i.name for i in pc.list_indexes()]:
pc.create_index(
name=index_name,
dimension=1536,
metric="cosine",
spec=ServerlessSpec(cloud="aws", region="us-east-1"),
)
# Build vector store
vectorstore = PineconeVectorStore.from_documents(
documents=docs,
embedding=OpenAIEmbeddings(),
index_name=index_name,
)
# Retrieve relevant context
retriever = vectorstore.as_retriever(search_kwargs={"k": 5})
results = retriever.invoke("What did the agent discuss yesterday?")
```
### Namespace-based session memory
```python
# Store per-session memory
vectorstore = PineconeVectorStore(
index=pc.Index(index_name),
embedding=OpenAIEmbeddings(),
namespace=f"session-{session_id}",
)
# Query across all sessions (no namespace filter)
all_memory = PineconeVectorStore(
index=pc.Index(index_name),
embedding=OpenAIEmbeddings(),
)
results = all_memory.similarity_search("relevant query", k=10)
```
## Best practices
1. **Namespace by session or user** — isolate data for multi-tenant agents
2. **Batch upserts** — 100–200 vectors per batch for efficiency
3. **Metadata filtering** — tag vectors with session ID, timestamp, topic
4. **Prune old memory** — delete stale namespaces to control costs
5. **Use serverless** — auto-scaling, pay-per-use pricing
## Resources
- **Pinecone Docs**: https://docs.pinecone.io
- **LangChain Integration**: https://python.langchain.com/docs/integrations/vectorstores/pinecone
- **Free Tier**: 1 index, 100K vectors (1536 dimensions)
@@ -16,7 +16,7 @@ Free keyless meta-search aggregating 70+ engines.
|---|---|
| Source | Optional — install with `hermes skills install official/research/searxng-search` |
| Path | `optional-skills/research/searxng-search` |
| Version | `1.0.0` |
| Version | `1.0.1` |
| Author | hermes-agent |
| License | MIT |
| Platforms | linux, macos |
@@ -141,23 +141,6 @@ for r in data.get("results", []):
print()
```
## Method 3: searxng-data Python Package
For more structured access, install the `searxng-data` package:
```bash
pip install searxng-data
```
```python
from searxng_data import engines
# List available engines
print(engines.list_engines())
```
Note: This package only provides engine metadata, not the search API itself.
## Self-Hosting SearXNG
To run your own SearXNG instance:
+1 -1
View File
@@ -333,7 +333,7 @@ def render_skill_page(
) -> str:
name = fm.get("name", meta["slug"])
description = fm.get("description", "").strip()
short_desc = description.split(".")[0].strip() if description else name
short_desc = re.split(r"\.(?:\s|$)", description, maxsplit=1)[0].strip() if description else name
if len(short_desc) > 160:
short_desc = short_desc[:157] + "..."
+5 -1
View File
@@ -233,9 +233,9 @@ const sidebars: SidebarsConfig = {
key: 'skills-bundled-mlops',
collapsed: true,
items: [
'user-guide/skills/bundled/mlops/mlops-evaluation-evaluating-llms-harness',
'user-guide/skills/bundled/mlops/mlops-huggingface-hub',
'user-guide/skills/bundled/mlops/mlops-inference-llama-cpp',
'user-guide/skills/bundled/mlops/mlops-evaluation-evaluating-llms-harness',
'user-guide/skills/bundled/mlops/mlops-inference-serving-llms-vllm',
'user-guide/skills/bundled/mlops/mlops-evaluation-weights-and-biases',
],
@@ -307,6 +307,7 @@ const sidebars: SidebarsConfig = {
items: [
'user-guide/skills/bundled/software-development/software-development-dogfood',
'user-guide/skills/bundled/software-development/software-development-hermes-agent-skill-authoring',
'user-guide/skills/bundled/software-development/software-development-inspecting-hermes-desktop-dom',
'user-guide/skills/bundled/software-development/software-development-node-inspect-debugger',
'user-guide/skills/bundled/software-development/software-development-plan',
'user-guide/skills/bundled/software-development/software-development-python-debugpy',
@@ -374,6 +375,7 @@ const sidebars: SidebarsConfig = {
'user-guide/skills/optional/creative/creative-kanban-video-orchestrator',
'user-guide/skills/optional/creative/creative-meme-generation',
'user-guide/skills/optional/creative/creative-pixel-art',
'user-guide/skills/optional/creative/creative-tldraw-offline',
'user-guide/skills/optional/creative/creative-unreal-mcp',
],
},
@@ -552,6 +554,7 @@ const sidebars: SidebarsConfig = {
'user-guide/skills/optional/research/research-gitnexus-explorer',
'user-guide/skills/optional/research/research-osint-investigation',
'user-guide/skills/optional/research/research-parallel-cli',
'user-guide/skills/optional/research/research-pinecone-research',
'user-guide/skills/optional/research/research-qmd',
'user-guide/skills/optional/research/research-scrapling',
'user-guide/skills/optional/research/research-searxng-search',
@@ -665,6 +668,7 @@ const sidebars: SidebarsConfig = {
'user-guide/messaging/ntfy',
'user-guide/messaging/irc',
'user-guide/messaging/open-webui',
'user-guide/messaging/relay',
'user-guide/messaging/webhooks',
],
},