diff --git a/.env.example b/.env.example
new file mode 100644
index 0000000..927ff2d
--- /dev/null
+++ b/.env.example
@@ -0,0 +1,8 @@
+# API Keys for Deep Research Agent Example
+# Copy this file to .env and fill in your actual API keys
+
+# Anthropic API Key (for Claude Sonnet 4)
+ANTHROPIC_API_KEY=your_anthropic_api_key_here
+
+# Tavily API Key (for web search)
+TAVILY_API_KEY=your_tavily_api_key_here
\ No newline at end of file
diff --git a/.gitignore b/.gitignore
index 4dfeffd..7dd83c4 100644
--- a/.gitignore
+++ b/.gitignore
@@ -1,3 +1,6 @@
+# macOS
+.DS_Store
+
# Python
__pycache__/
*.py[cod]
@@ -23,3 +26,5 @@ venv/
.langgraph_api/
workspace/
.deno_cache/
+*test.ipynb
+*CLAUDE.md
\ No newline at end of file
diff --git a/EvoScientist/EvoScientist.py b/EvoScientist/EvoScientist.py
new file mode 100644
index 0000000..50e0d09
--- /dev/null
+++ b/EvoScientist/EvoScientist.py
@@ -0,0 +1,157 @@
+"""EvoScientist Agent graph construction.
+
+This module creates and exports the compiled agent graph.
+Usage:
+ from EvoScientist import agent
+
+ # Notebook / programmatic usage
+ for state in agent.stream(
+ {"messages": [HumanMessage(content="your question")]},
+ config={"configurable": {"thread_id": "1"}},
+ stream_mode="values",
+ ):
+ ...
+"""
+
+import os
+from datetime import datetime
+from pathlib import Path
+
+from deepagents import create_deep_agent
+from deepagents.backends import FilesystemBackend, CompositeBackend
+from langchain.chat_models import init_chat_model
+
+from .backends import CustomSandboxBackend, ReadOnlyFilesystemBackend
+from .middleware import create_skills_middleware
+from .prompts import RESEARCHER_INSTRUCTIONS, get_system_prompt
+from .utils import load_subagents
+from .tools import tavily_search, think_tool
+
+# =============================================================================
+# Configuration
+# =============================================================================
+
+# Backend mode: "sandbox" (with execute) or "filesystem" (read/write only)
+BACKEND_MODE = "sandbox"
+
+# Research limits
+MAX_CONCURRENT = 3 # Max parallel sub-agents
+MAX_ITERATIONS = 3 # Max delegation rounds
+
+# Workspace settings
+WORKSPACE_DIR = "./workspace/"
+SKILLS_DIR = "./skills/"
+SUBAGENTS_CONFIG = Path(__file__).parent / "subagent.yaml"
+
+# =============================================================================
+# Initialization
+# =============================================================================
+
+# Get current date
+current_date = datetime.now().strftime("%Y-%m-%d")
+
+# Generate system prompt with limits
+SYSTEM_PROMPT = get_system_prompt(
+ max_concurrent=MAX_CONCURRENT,
+ max_iterations=MAX_ITERATIONS,
+)
+
+# Initialize chat model
+chat_model = init_chat_model(
+ model="claude-sonnet-4-5-20250929",
+ model_provider="anthropic",
+ # thinking={"type": "enabled", "budget_tokens": 2000},
+)
+
+# Initialize workspace backend based on mode
+if BACKEND_MODE == "sandbox":
+ _workspace_backend = CustomSandboxBackend(
+ root_dir=WORKSPACE_DIR,
+ virtual_mode=True,
+ timeout=300,
+ )
+else:
+ _workspace_backend = FilesystemBackend(
+ root_dir=WORKSPACE_DIR,
+ virtual_mode=True,
+ )
+
+# Skills backend: read-only access to ./skills/
+_skills_backend = ReadOnlyFilesystemBackend(
+ root_dir=SKILLS_DIR,
+ virtual_mode=True,
+)
+
+# Composite backend: workspace as default, skills mounted at /skills/
+backend = CompositeBackend(
+ default=_workspace_backend,
+ routes={"/skills/": _skills_backend},
+)
+
+tool_registry = {
+ "think_tool": think_tool,
+ "tavily_search": tavily_search,
+}
+
+prompt_refs = {
+ "RESEARCHER_INSTRUCTIONS": RESEARCHER_INSTRUCTIONS.format(date=current_date),
+}
+
+subagents = load_subagents(
+ SUBAGENTS_CONFIG,
+ tool_registry=tool_registry,
+ prompt_refs=prompt_refs,
+)
+
+# Shared kwargs for agent creation
+_AGENT_KWARGS = dict(
+ name="EvoScientist",
+ model=chat_model,
+ tools=[think_tool],
+ backend=backend,
+ subagents=subagents,
+ middleware=[create_skills_middleware(SKILLS_DIR, WORKSPACE_DIR)],
+ system_prompt=SYSTEM_PROMPT,
+)
+
+# Default agent (no checkpointer) — used by langgraph dev / LangSmith / notebooks
+EvoScientist_agent = create_deep_agent(**_AGENT_KWARGS).with_config({"recursion_limit": 500})
+
+
+def create_cli_agent(workspace_dir: str | None = None):
+ """Create agent with InMemorySaver checkpointer for CLI multi-turn support.
+
+ Args:
+ workspace_dir: Optional per-session workspace directory. If provided,
+ creates a fresh backend rooted at this path. If None, uses the
+ module-level default backend (./workspace/).
+ """
+ from langgraph.checkpoint.memory import InMemorySaver # type: ignore[import-untyped]
+
+ if workspace_dir:
+ ws_backend = CustomSandboxBackend(
+ root_dir=workspace_dir,
+ virtual_mode=True,
+ timeout=300,
+ )
+ sk_backend = ReadOnlyFilesystemBackend(
+ root_dir=SKILLS_DIR,
+ virtual_mode=True,
+ )
+ be = CompositeBackend(
+ default=ws_backend,
+ routes={"/skills/": sk_backend},
+ )
+ mw = [create_skills_middleware(SKILLS_DIR, workspace_dir)]
+ kwargs = dict(
+ _AGENT_KWARGS,
+ backend=be,
+ middleware=mw,
+ )
+ else:
+ kwargs = dict(_AGENT_KWARGS)
+
+ return create_deep_agent(
+ **kwargs,
+ checkpointer=InMemorySaver(),
+ ).with_config({"recursion_limit": 500})
diff --git a/EvoScientist/__init__.py b/EvoScientist/__init__.py
new file mode 100644
index 0000000..3e9461b
--- /dev/null
+++ b/EvoScientist/__init__.py
@@ -0,0 +1,24 @@
+"""EvoScientist Agent — AI-powered research & code execution."""
+
+from .backends import CustomSandboxBackend, ReadOnlyFilesystemBackend
+from .middleware import create_skills_middleware
+from .prompts import get_system_prompt, RESEARCHER_INSTRUCTIONS
+from .tools import tavily_search, think_tool
+from .EvoScientist import EvoScientist_agent, create_cli_agent
+
+__all__ = [
+ # Agent graph (main export)
+ "EvoScientist_agent",
+ "create_cli_agent",
+ # Backends
+ "CustomSandboxBackend",
+ "ReadOnlyFilesystemBackend",
+ # Middleware
+ "create_skills_middleware",
+ # Prompts
+ "get_system_prompt",
+ "RESEARCHER_INSTRUCTIONS",
+ # Tools
+ "tavily_search",
+ "think_tool",
+]
diff --git a/EvoScientist/__main__.py b/EvoScientist/__main__.py
new file mode 100644
index 0000000..9895412
--- /dev/null
+++ b/EvoScientist/__main__.py
@@ -0,0 +1,4 @@
+"""Enable `python -m EvoScientist` execution."""
+from EvoScientist.cli import main
+
+main()
diff --git a/EvoScientist/backends.py b/EvoScientist/backends.py
new file mode 100644
index 0000000..dbf97bb
--- /dev/null
+++ b/EvoScientist/backends.py
@@ -0,0 +1,366 @@
+"""Custom backends for EvoScientist agent."""
+
+import os
+import re
+import subprocess
+from pathlib import Path
+
+from deepagents.backends import FilesystemBackend
+from deepagents.backends.filesystem import WriteResult, EditResult
+from deepagents.backends.protocol import (
+ ExecuteResponse,
+ SandboxBackendProtocol,
+)
+
+# System path prefixes that should never appear in virtual paths.
+# If the agent hallucinates an absolute system path, we block it.
+_SYSTEM_PATH_PREFIXES = (
+ "/Users/", "/home/", "/tmp/", "/var/", "/etc/",
+ "/opt/", "/usr/", "/bin/", "/sbin/", "/dev/",
+ "/proc/", "/sys/", "/root/",
+)
+
+# Dangerous patterns that could escape the workspace
+BLOCKED_PATTERNS = [
+ r'\.\.', # ../ directory traversal
+ r'~/', # home directory
+ r'\bcd\s+/', # cd to absolute path
+ r'\brm\s+-rf\s+/', # rm -rf with absolute path
+]
+
+# Dangerous commands that should never be executed
+BLOCKED_COMMANDS = [
+ 'sudo',
+ 'chmod',
+ 'chown',
+ 'mkfs',
+ 'dd',
+ 'shutdown',
+ 'reboot',
+]
+
+
+def validate_command(command: str) -> str | None:
+ """
+ Validate a shell command for safety.
+
+ Returns:
+ None if command is safe, error message string if blocked.
+ """
+ # Check for directory traversal and dangerous patterns
+ for pattern in BLOCKED_PATTERNS:
+ if re.search(pattern, command):
+ return (
+ f"Command blocked: contains forbidden pattern '{pattern}'. "
+ f"All commands must operate within the workspace directory. "
+ f"Use relative paths (e.g., './file.py') instead."
+ )
+
+ # Check for dangerous commands
+ for cmd in BLOCKED_COMMANDS:
+ if re.search(rf'\b{cmd}\b', command):
+ return (
+ f"Command blocked: '{cmd}' is not allowed in sandbox mode. "
+ f"Only standard development commands are permitted."
+ )
+
+ return None
+
+
+def convert_virtual_paths_in_command(command: str) -> str:
+ """
+ Convert virtual paths (starting with /) in commands to relative paths.
+
+ Examples:
+ - "python /main.py" -> "python ./main.py"
+ - "cat /data/file.txt" -> "cat ./data/file.txt"
+ - "ls /" -> "ls ."
+ - "python main.py" -> "python main.py" (unchanged)
+
+ Args:
+ command: Original command
+
+ Returns:
+ Converted command
+ """
+
+ def replace_virtual_path(match):
+ path = match.group(0)
+
+ # Skip content that looks like a URL
+ if '://' in command[max(0, match.start() - 10):match.end() + 10]:
+ return path
+
+ # Convert virtual path
+ if path == '/':
+ return '.'
+ else:
+ return '.' + path
+
+ # Match pattern: paths starting with / (but not URLs)
+ pattern = r'(?<=\s)/[^\s;|&<>\'"`]*|^/[^\s;|&<>\'"`]*'
+ converted = re.sub(pattern, replace_virtual_path, command)
+
+ return converted
+
+
+class ReadOnlyFilesystemBackend(FilesystemBackend):
+ """
+ Read-only filesystem backend.
+
+ Allows read, ls, grep, glob operations but blocks write and edit.
+ Used for skills directory — agent can read skill definitions but cannot
+ modify them.
+ """
+
+ def write(self, file_path: str, content: str) -> WriteResult:
+ return WriteResult(
+ error="This directory is read-only. Write operations are not permitted here."
+ )
+
+ def edit(
+ self,
+ file_path: str,
+ old_string: str,
+ new_string: str,
+ replace_all: bool = False,
+ ) -> EditResult:
+ return EditResult(
+ error="This directory is read-only. Edit operations are not permitted here."
+ )
+
+
+class MergedReadOnlyBackend:
+ """Read-only backend that merges two directories.
+
+ Reads from *primary* first (user skills in workspace/skills/),
+ falls back to *secondary* (system skills in ./skills/).
+ User skills override system skills with the same name.
+
+ Both directories share the same virtual path namespace — the agent
+ sees all skills under /skills/ regardless of which backend serves them.
+ """
+
+ def __init__(self, primary_dir: str, secondary_dir: str):
+ self._primary = ReadOnlyFilesystemBackend(root_dir=primary_dir, virtual_mode=True)
+ self._secondary = ReadOnlyFilesystemBackend(root_dir=secondary_dir, virtual_mode=True)
+
+ # -- read: try primary first, fall back to secondary --
+
+ def read(self, file_path: str, offset: int = 0, limit: int = 2000) -> str:
+ try:
+ result = self._primary.read(file_path, offset, limit)
+ if not result.startswith("Error:"):
+ return result
+ except (ValueError, FileNotFoundError, OSError):
+ pass
+ return self._secondary.read(file_path, offset, limit)
+
+ # -- ls_info: merge both, primary wins on name conflicts --
+
+ def ls_info(self, path: str = "/") -> list:
+ secondary_items = {item["path"]: item for item in self._secondary.ls_info(path)}
+ primary_items = {item["path"]: item for item in self._primary.ls_info(path)}
+ secondary_items.update(primary_items) # primary overrides
+ return sorted(secondary_items.values(), key=lambda x: x["path"])
+
+ # -- grep_raw: search both, deduplicate --
+
+ def grep_raw(self, pattern: str, path: str | None = None, glob: str | None = None) -> list:
+ results = self._secondary.grep_raw(pattern, path, glob)
+ try:
+ results += self._primary.grep_raw(pattern, path, glob)
+ except Exception:
+ pass
+ return results
+
+ # -- glob_info: merge both --
+
+ def glob_info(self, pattern: str, path: str = "/") -> list:
+ secondary = {item["path"]: item for item in self._secondary.glob_info(pattern, path)}
+ try:
+ primary = {item["path"]: item for item in self._primary.glob_info(pattern, path)}
+ secondary.update(primary)
+ except Exception:
+ pass
+ return sorted(secondary.values(), key=lambda x: x["path"])
+
+ # -- write / edit: blocked --
+
+ def write(self, file_path: str, content: str) -> WriteResult:
+ return WriteResult(
+ error="This directory is read-only. Write operations are not permitted here."
+ )
+
+ def edit(
+ self,
+ file_path: str,
+ old_string: str,
+ new_string: str,
+ replace_all: bool = False,
+ ) -> EditResult:
+ return EditResult(
+ error="This directory is read-only. Edit operations are not permitted here."
+ )
+
+ # -- async variants (required by middleware) --
+
+ async def aread(self, file_path: str, offset: int = 0, limit: int = 2000) -> str:
+ return self.read(file_path, offset, limit)
+
+ async def als_info(self, path: str = "/") -> list:
+ return self.ls_info(path)
+
+ async def agrep_raw(self, pattern: str, path: str | None = None, glob: str | None = None) -> list:
+ return self.grep_raw(pattern, path, glob)
+
+ async def aglob_info(self, pattern: str, path: str = "/") -> list:
+ return self.glob_info(pattern, path)
+
+ async def awrite(self, file_path: str, content: str) -> WriteResult:
+ return self.write(file_path, content)
+
+ async def aedit(self, file_path: str, old_string: str, new_string: str, replace_all: bool = False) -> EditResult:
+ return self.edit(file_path, old_string, new_string, replace_all)
+
+
+class CustomSandboxBackend(FilesystemBackend, SandboxBackendProtocol):
+ """
+ Custom sandbox backend - inherits FilesystemBackend and implements execute method.
+
+ Features:
+ - Inherits all file operations (ls, read, write, edit, grep, glob)
+ - Adds shell command execution capability
+ - Command validation prevents directory traversal and dangerous operations
+ - Runs commands in specified working directory
+ - Compatible with LangGraph checkpointer (no thread locks)
+ """
+
+ def __init__(
+ self,
+ root_dir: str = ".",
+ virtual_mode: bool = True,
+ working_dir: str | None = None,
+ timeout: int = 300,
+ shell: str = "/bin/bash",
+ ):
+ """
+ Initialize custom sandbox backend.
+
+ Args:
+ root_dir: File system root directory
+ virtual_mode: Whether to enable virtual path mode
+ working_dir: Working directory for command execution (defaults to root_dir)
+ timeout: Command execution timeout in seconds
+ shell: Shell program to use
+ """
+ super().__init__(root_dir=root_dir, virtual_mode=virtual_mode)
+
+ self.working_dir = working_dir or root_dir
+ self.timeout = timeout
+ self.shell = shell
+ self.virtual_mode = virtual_mode
+
+ # Ensure working directory exists
+ os.makedirs(self.working_dir, exist_ok=True)
+
+ def _resolve_path(self, key: str) -> Path:
+ """Resolve path with sanitization to prevent nested directories.
+
+ Intercepts all file operations (read, write, edit, ls, grep, glob).
+ Auto-corrects common LLM path mistakes instead of crashing:
+ 1. /workspace/file.py → /file.py
+ 2. /Users/name/.../workspace/f → /f (strip up to workspace/)
+ 3. /Users/name/file.py → /file.py (keep basename)
+ """
+ # Auto-strip /workspace/ prefix to prevent nesting
+ if key.startswith("/workspace/"):
+ key = key[len("/workspace"):] # "/workspace/main.py" → "/main.py"
+ elif key == "/workspace":
+ key = "/"
+
+ # Auto-correct system absolute paths
+ for prefix in _SYSTEM_PATH_PREFIXES:
+ if key.startswith(prefix):
+ # Try to extract path after "workspace/" or "workspace" at end
+ marker = "/workspace/"
+ idx = key.find(marker)
+ if idx != -1:
+ key = "/" + key[idx + len(marker):]
+ elif key.endswith("/workspace"):
+ key = "/"
+ else:
+ # Fall back to basename
+ key = "/" + Path(key).name
+ break
+
+ return super()._resolve_path(key)
+
+ def execute(self, command: str) -> ExecuteResponse:
+ """
+ Execute shell command in sandbox environment.
+
+ Commands are validated before execution to prevent:
+ - Directory traversal (../)
+ - Access to paths outside workspace
+ - Dangerous system commands
+
+ Args:
+ command: Command string to execute
+
+ Returns:
+ ExecuteResponse containing output, exit_code, and truncated flag
+ """
+ try:
+ # Validate command safety
+ error = validate_command(command)
+ if error:
+ return ExecuteResponse(
+ output=error,
+ exit_code=1,
+ truncated=False,
+ )
+
+ # Convert virtual paths to relative paths
+ if self.virtual_mode:
+ command = convert_virtual_paths_in_command(command=command)
+
+ result = subprocess.run(
+ command,
+ shell=True,
+ executable=self.shell,
+ cwd=self.working_dir,
+ capture_output=True,
+ text=True,
+ timeout=self.timeout,
+ )
+
+ output = ""
+ if result.stdout:
+ output += result.stdout
+ if result.stderr:
+ output += result.stderr
+
+ return ExecuteResponse(
+ output=output,
+ exit_code=result.returncode,
+ truncated=False,
+ )
+
+ except subprocess.TimeoutExpired:
+ return ExecuteResponse(
+ output=f"Command timed out after {self.timeout} seconds",
+ exit_code=-1,
+ truncated=False,
+ )
+ except Exception as e:
+ return ExecuteResponse(
+ output=f"Error executing command: {str(e)}",
+ exit_code=-1,
+ truncated=False,
+ )
+
+ async def aexecute(self, command: str) -> ExecuteResponse:
+ """Async version of execute (runs sync version in thread)."""
+ import asyncio
+ return await asyncio.to_thread(self.execute, command)
diff --git a/EvoScientist/cli.py b/EvoScientist/cli.py
new file mode 100644
index 0000000..cc2af80
--- /dev/null
+++ b/EvoScientist/cli.py
@@ -0,0 +1,1281 @@
+"""
+EvoScientist Agent CLI
+
+Command-line interface with streaming output for the EvoScientist research agent.
+
+Features:
+- Thinking panel (blue) - shows model reasoning
+- Tool calls with status indicators (green/yellow/red dots)
+- Tool results in tree format with folding
+- Response panel (green) - shows final response
+- Thread ID support for multi-turn conversations
+- Interactive mode with prompt_toolkit
+"""
+
+import argparse
+import asyncio
+import os
+import sys
+import uuid
+from datetime import datetime
+from typing import Any, AsyncIterator
+
+from dotenv import load_dotenv # type: ignore[import-untyped]
+from prompt_toolkit import PromptSession # type: ignore[import-untyped]
+from prompt_toolkit.history import FileHistory # type: ignore[import-untyped]
+from prompt_toolkit.auto_suggest import AutoSuggestFromHistory # type: ignore[import-untyped]
+from prompt_toolkit.formatted_text import HTML # type: ignore[import-untyped]
+from rich.console import Console, Group # type: ignore[import-untyped]
+from rich.panel import Panel # type: ignore[import-untyped]
+from rich.markdown import Markdown # type: ignore[import-untyped]
+from rich.live import Live # type: ignore[import-untyped]
+from rich.text import Text # type: ignore[import-untyped]
+from rich.spinner import Spinner # type: ignore[import-untyped]
+from langchain_core.messages import AIMessage, AIMessageChunk # type: ignore[import-untyped]
+
+from .stream import (
+ StreamEventEmitter,
+ ToolCallTracker,
+ ToolResultFormatter,
+ DisplayLimits,
+ ToolStatus,
+ format_tool_compact,
+ is_success,
+)
+
+load_dotenv(override=True)
+
+console = Console(
+ legacy_windows=(sys.platform == 'win32'),
+ no_color=os.getenv('NO_COLOR') is not None,
+)
+
+formatter = ToolResultFormatter()
+
+
+# =============================================================================
+# Stream event generator
+# =============================================================================
+
+async def stream_agent_events(agent: Any, message: str, thread_id: str) -> AsyncIterator[dict]:
+ """Stream events from the agent graph using async iteration.
+
+ Uses agent.astream() with subgraphs=True to see sub-agent activity.
+
+ Args:
+ agent: Compiled state graph from create_deep_agent()
+ message: User message
+ thread_id: Thread ID for conversation persistence
+
+ Yields:
+ Event dicts: thinking, text, tool_call, tool_result,
+ subagent_start, subagent_tool_call, subagent_tool_result, subagent_end,
+ done, error
+ """
+ config = {"configurable": {"thread_id": thread_id}}
+ emitter = StreamEventEmitter()
+ tracker = ToolCallTracker()
+ full_response = ""
+
+ # Track sub-agent names by root namespace element
+ _subagent_names: dict[str, str] = {} # root_ns_element → display name
+ # Track which task tool_call_ids have been announced
+ _announced_tasks: set[str] = set()
+
+ def _get_subagent_name(namespace: tuple) -> str | None:
+ """Get sub-agent name from namespace, or None if main agent.
+
+ Any non-empty namespace is a sub-agent. Name is resolved by checking
+ all registered names for a prefix match against namespace elements.
+ """
+ if not namespace:
+ return None
+ root = str(namespace[0]) if namespace else ""
+ # Exact match
+ if root in _subagent_names:
+ return _subagent_names[root]
+ # Prefix match: namespace root might be "task:abc123" and we
+ # registered "task:call_xyz" — check if any registered key
+ # appears as a substring of the root or vice versa
+ for key, name in _subagent_names.items():
+ if key in root or root in key:
+ _subagent_names[root] = name # cache for next lookup
+ return name
+ # Auto-register: infer from namespace string
+ if ":" in root:
+ inferred = root.split(":")[0]
+ else:
+ inferred = root
+ name = inferred or "sub-agent"
+ _subagent_names[root] = name
+ return name
+
+ try:
+ async for chunk in agent.astream(
+ {"messages": [{"role": "user", "content": message}]},
+ config=config,
+ stream_mode="messages",
+ subgraphs=True,
+ ):
+ # With subgraphs=True, event is (namespace, (message, metadata))
+ namespace: tuple = ()
+ data: Any = chunk
+
+ if isinstance(chunk, tuple) and len(chunk) >= 2:
+ first = chunk[0]
+ if isinstance(first, tuple):
+ # (namespace_tuple, (message, metadata))
+ namespace = first
+ data = chunk[1]
+ else:
+ # (message, metadata) — no namespace
+ data = chunk
+
+ # Unpack message from data
+ msg: Any
+ if isinstance(data, tuple) and len(data) >= 2:
+ msg = data[0]
+ else:
+ msg = data
+
+ subagent = _get_subagent_name(namespace)
+
+ # Process AIMessageChunk / AIMessage
+ if isinstance(msg, (AIMessageChunk, AIMessage)):
+ if subagent:
+ # Sub-agent content — emit sub-agent events
+ for ev in _process_chunk_content(msg, emitter, tracker):
+ if ev.type == "tool_call":
+ yield emitter.subagent_tool_call(
+ subagent, ev.data["name"], ev.data["args"], ev.data.get("id", "")
+ ).data
+ # Skip text/thinking from sub-agents (too noisy)
+
+ if hasattr(msg, "tool_calls") and msg.tool_calls:
+ for tc in msg.tool_calls:
+ name = tc.get("name", "")
+ args = tc.get("args", {})
+ tool_id = tc.get("id", "")
+ # Skip empty-name chunks (incomplete streaming fragments)
+ if not name and not tool_id:
+ continue
+ yield emitter.subagent_tool_call(
+ subagent, name, args if isinstance(args, dict) else {}, tool_id
+ ).data
+ else:
+ # Main agent content
+ for ev in _process_chunk_content(msg, emitter, tracker):
+ if ev.type == "text":
+ full_response += ev.data.get("content", "")
+ yield ev.data
+
+ if hasattr(msg, "tool_calls") and msg.tool_calls:
+ for ev in _process_tool_calls(msg.tool_calls, emitter, tracker):
+ yield ev.data
+ # Detect task tool calls → announce sub-agent
+ tc_data = ev.data
+ if tc_data.get("name") == "task":
+ tool_id = tc_data.get("id", "")
+ if tool_id and tool_id not in _announced_tasks:
+ _announced_tasks.add(tool_id)
+ args = tc_data.get("args", {})
+ sa_name = args.get("subagent_type", "").strip()
+ desc = args.get("description", "").strip()
+ # Use subagent_type as name; fall back to description snippet
+ if not sa_name:
+ sa_name = desc[:30] + "..." if len(desc) > 30 else desc
+ if not sa_name:
+ sa_name = "sub-agent"
+ # Pre-register name so namespace lookup finds it
+ _subagent_names[f"task:{tool_id}"] = sa_name
+ yield emitter.subagent_start(sa_name, desc).data
+
+ # Process ToolMessage (tool execution result)
+ elif hasattr(msg, "type") and msg.type == "tool":
+ if subagent:
+ name = getattr(msg, "name", "unknown")
+ raw_content = str(getattr(msg, "content", ""))
+ content = raw_content[:DisplayLimits.TOOL_RESULT_MAX]
+ success = is_success(content)
+ yield emitter.subagent_tool_result(subagent, name, content, success).data
+ else:
+ for ev in _process_tool_result(msg, emitter, tracker):
+ yield ev.data
+ # Check if this is a task result → sub-agent ended
+ name = getattr(msg, "name", "")
+ if name == "task":
+ tool_call_id = getattr(msg, "tool_call_id", "")
+ # Find the sub-agent name for this task
+ sa_key = f"task:{tool_call_id}"
+ sa_name = _subagent_names.get(sa_key, "sub-agent")
+ yield emitter.subagent_end(sa_name).data
+
+ except Exception as e:
+ yield emitter.error(str(e)).data
+ raise
+
+ yield emitter.done(full_response).data
+
+
+def _process_chunk_content(chunk, emitter: StreamEventEmitter, tracker: ToolCallTracker):
+ """Process content blocks from an AI message chunk."""
+ content = chunk.content
+
+ if isinstance(content, str):
+ if content:
+ yield emitter.text(content)
+ return
+
+ blocks = None
+ if hasattr(chunk, "content_blocks"):
+ try:
+ blocks = chunk.content_blocks
+ except Exception:
+ blocks = None
+
+ if blocks is None:
+ if isinstance(content, dict):
+ blocks = [content]
+ elif isinstance(content, list):
+ blocks = content
+ else:
+ return
+
+ for raw_block in blocks:
+ block = raw_block
+ if not isinstance(block, dict):
+ if hasattr(block, "model_dump"):
+ block = block.model_dump()
+ elif hasattr(block, "dict"):
+ block = block.dict()
+ else:
+ continue
+
+ block_type = block.get("type")
+
+ if block_type in ("thinking", "reasoning"):
+ thinking_text = block.get("thinking") or block.get("reasoning") or ""
+ if thinking_text:
+ yield emitter.thinking(thinking_text)
+
+ elif block_type == "text":
+ text = block.get("text") or block.get("content") or ""
+ if text:
+ yield emitter.text(text)
+
+ elif block_type in ("tool_use", "tool_call"):
+ tool_id = block.get("id", "")
+ name = block.get("name", "")
+ args = block.get("input") if block_type == "tool_use" else block.get("args")
+ args_payload = args if isinstance(args, dict) else {}
+
+ if tool_id:
+ tracker.update(tool_id, name=name, args=args_payload)
+ if tracker.is_ready(tool_id):
+ tracker.mark_emitted(tool_id)
+ yield emitter.tool_call(name, args_payload, tool_id)
+
+ elif block_type == "input_json_delta":
+ partial_json = block.get("partial_json", "")
+ if partial_json:
+ tracker.append_json_delta(partial_json, block.get("index", 0))
+
+ elif block_type == "tool_call_chunk":
+ tool_id = block.get("id", "")
+ name = block.get("name", "")
+ if tool_id:
+ tracker.update(tool_id, name=name)
+ partial_args = block.get("args", "")
+ if isinstance(partial_args, str) and partial_args:
+ tracker.append_json_delta(partial_args, block.get("index", 0))
+
+
+def _process_tool_calls(tool_calls: list, emitter: StreamEventEmitter, tracker: ToolCallTracker):
+ """Process tool_calls from chunk.tool_calls attribute."""
+ for tc in tool_calls:
+ tool_id = tc.get("id", "")
+ if tool_id:
+ name = tc.get("name", "")
+ args = tc.get("args", {})
+ args_payload = args if isinstance(args, dict) else {}
+
+ tracker.update(tool_id, name=name, args=args_payload)
+ if tracker.is_ready(tool_id):
+ tracker.mark_emitted(tool_id)
+ yield emitter.tool_call(name, args_payload, tool_id)
+
+
+def _process_tool_result(chunk, emitter: StreamEventEmitter, tracker: ToolCallTracker):
+ """Process a ToolMessage result."""
+ tracker.finalize_all()
+
+ # Re-emit all tool calls with complete args
+ for info in tracker.get_all():
+ yield emitter.tool_call(info.name, info.args, info.id)
+
+ name = getattr(chunk, "name", "unknown")
+ raw_content = str(getattr(chunk, "content", ""))
+ content = raw_content[:DisplayLimits.TOOL_RESULT_MAX]
+ if len(raw_content) > DisplayLimits.TOOL_RESULT_MAX:
+ content += "\n... (truncated)"
+
+ success = is_success(content)
+ yield emitter.tool_result(name, content, success)
+
+
+# =============================================================================
+# Stream state
+# =============================================================================
+
+class SubAgentState:
+ """Tracks a single sub-agent's activity."""
+
+ def __init__(self, name: str, description: str = ""):
+ self.name = name
+ self.description = description
+ self.tool_calls: list[dict] = []
+ self.tool_results: list[dict] = []
+ self._result_map: dict[str, dict] = {} # tool_call_id → result
+ self.is_active = True
+
+ def add_tool_call(self, name: str, args: dict, tool_id: str = ""):
+ # Skip empty-name calls without an id (incomplete streaming chunks)
+ if not name and not tool_id:
+ return
+ tc_data = {"id": tool_id, "name": name, "args": args}
+ if tool_id:
+ for i, tc in enumerate(self.tool_calls):
+ if tc.get("id") == tool_id:
+ # Merge: keep the non-empty name/args
+ if name:
+ self.tool_calls[i]["name"] = name
+ if args:
+ self.tool_calls[i]["args"] = args
+ return
+ # Skip if name is empty and we can't deduplicate by id
+ if not name:
+ return
+ self.tool_calls.append(tc_data)
+
+ def add_tool_result(self, name: str, content: str, success: bool = True):
+ result = {"name": name, "content": content, "success": success}
+ self.tool_results.append(result)
+ # Try to match result to the first unmatched tool call with same name
+ for tc in self.tool_calls:
+ tc_id = tc.get("id", "")
+ tc_name = tc.get("name", "")
+ if tc_id and tc_id not in self._result_map and tc_name == name:
+ self._result_map[tc_id] = result
+ return
+ # Fallback: match first unmatched tool call
+ for tc in self.tool_calls:
+ tc_id = tc.get("id", "")
+ if tc_id and tc_id not in self._result_map:
+ self._result_map[tc_id] = result
+ return
+
+ def get_result_for(self, tc: dict) -> dict | None:
+ """Get matched result for a tool call."""
+ tc_id = tc.get("id", "")
+ if tc_id:
+ return self._result_map.get(tc_id)
+ # Fallback: index-based matching
+ try:
+ idx = self.tool_calls.index(tc)
+ if idx < len(self.tool_results):
+ return self.tool_results[idx]
+ except ValueError:
+ pass
+ return None
+
+
+class StreamState:
+ """Accumulates stream state for display updates."""
+
+ def __init__(self):
+ self.thinking_text = ""
+ self.response_text = ""
+ self.tool_calls = []
+ self.tool_results = []
+ self.is_thinking = False
+ self.is_responding = False
+ self.is_processing = False
+ # Sub-agent tracking
+ self.subagents: list[SubAgentState] = []
+ self._subagent_map: dict[str, SubAgentState] = {} # name → state
+
+ def _get_or_create_subagent(self, name: str, description: str = "") -> SubAgentState:
+ if name not in self._subagent_map:
+ # Check if there's a generic "sub-agent" entry that should be merged
+ # This happens when namespace events arrive before the task tool call
+ # registers the proper name
+ if name != "sub-agent" and "sub-agent" in self._subagent_map:
+ old_sa = self._subagent_map.pop("sub-agent")
+ old_sa.name = name
+ if description:
+ old_sa.description = description
+ self._subagent_map[name] = old_sa
+ return old_sa
+ sa = SubAgentState(name, description)
+ self.subagents.append(sa)
+ self._subagent_map[name] = sa
+ elif description and not self._subagent_map[name].description:
+ self._subagent_map[name].description = description
+ return self._subagent_map[name]
+
+ def handle_event(self, event: dict) -> str:
+ """Process a single stream event, update internal state, return event type."""
+ event_type: str = event.get("type", "")
+
+ if event_type == "thinking":
+ self.is_thinking = True
+ self.is_responding = False
+ self.is_processing = False
+ self.thinking_text += event.get("content", "")
+
+ elif event_type == "text":
+ self.is_thinking = False
+ self.is_responding = True
+ self.is_processing = False
+ self.response_text += event.get("content", "")
+
+ elif event_type == "tool_call":
+ self.is_thinking = False
+ self.is_responding = False
+ self.is_processing = False
+
+ tool_id = event.get("id", "")
+ tc_data = {
+ "id": tool_id,
+ "name": event.get("name", "unknown"),
+ "args": event.get("args", {}),
+ }
+
+ if tool_id:
+ updated = False
+ for i, tc in enumerate(self.tool_calls):
+ if tc.get("id") == tool_id:
+ self.tool_calls[i] = tc_data
+ updated = True
+ break
+ if not updated:
+ self.tool_calls.append(tc_data)
+ else:
+ self.tool_calls.append(tc_data)
+
+ elif event_type == "tool_result":
+ self.is_processing = True
+ self.tool_results.append({
+ "name": event.get("name", "unknown"),
+ "content": event.get("content", ""),
+ })
+
+ elif event_type == "subagent_start":
+ name = event.get("name", "sub-agent")
+ desc = event.get("description", "")
+ sa = self._get_or_create_subagent(name, desc)
+ sa.is_active = True
+
+ elif event_type == "subagent_tool_call":
+ sa_name = event.get("subagent", "sub-agent")
+ sa = self._get_or_create_subagent(sa_name)
+ sa.add_tool_call(
+ event.get("name", "unknown"),
+ event.get("args", {}),
+ event.get("id", ""),
+ )
+
+ elif event_type == "subagent_tool_result":
+ sa_name = event.get("subagent", "sub-agent")
+ sa = self._get_or_create_subagent(sa_name)
+ sa.add_tool_result(
+ event.get("name", "unknown"),
+ event.get("content", ""),
+ event.get("success", True),
+ )
+
+ elif event_type == "subagent_end":
+ name = event.get("name", "sub-agent")
+ if name in self._subagent_map:
+ self._subagent_map[name].is_active = False
+
+ elif event_type == "done":
+ self.is_processing = False
+ if not self.response_text:
+ self.response_text = event.get("response", "")
+
+ elif event_type == "error":
+ self.is_processing = False
+ self.is_thinking = False
+ self.is_responding = False
+ error_msg = event.get("message", "Unknown error")
+ self.response_text += f"\n\n[Error] {error_msg}"
+
+ return event_type
+
+ def get_display_args(self) -> dict:
+ """Get kwargs for create_streaming_display()."""
+ return {
+ "thinking_text": self.thinking_text,
+ "response_text": self.response_text,
+ "tool_calls": self.tool_calls,
+ "tool_results": self.tool_results,
+ "is_thinking": self.is_thinking,
+ "is_responding": self.is_responding,
+ "is_processing": self.is_processing,
+ "subagents": self.subagents,
+ }
+
+
+# =============================================================================
+# Display functions
+# =============================================================================
+
+def _parse_todo_items(content: str) -> list[dict] | None:
+ """Parse todo items from write_todos output.
+
+ Attempts to extract a list of dicts with 'status' and 'content' keys
+ from the tool result string. Returns None if parsing fails.
+ """
+ import ast
+ import json
+
+ content = content.strip()
+
+ # Try JSON first
+ try:
+ data = json.loads(content)
+ if isinstance(data, list) and data and isinstance(data[0], dict):
+ return data
+ except (json.JSONDecodeError, ValueError):
+ pass
+
+ # Try Python literal
+ try:
+ data = ast.literal_eval(content)
+ if isinstance(data, list) and data and isinstance(data[0], dict):
+ return data
+ except (ValueError, SyntaxError):
+ pass
+
+ # Try to find a list embedded in the output
+ for line in content.split("\n"):
+ line = line.strip()
+ if line.startswith("[") and line.endswith("]"):
+ try:
+ data = json.loads(line)
+ if isinstance(data, list):
+ return data
+ except (json.JSONDecodeError, ValueError):
+ try:
+ data = ast.literal_eval(line)
+ if isinstance(data, list):
+ return data
+ except (ValueError, SyntaxError):
+ pass
+
+ return None
+
+
+def _build_todo_stats(items: list[dict]) -> str:
+ """Build stats string like '2 active | 1 pending | 3 done'."""
+ counts: dict[str, int] = {}
+ for item in items:
+ status = str(item.get("status", "todo")).lower()
+ # Normalize status names
+ if status in ("done", "completed", "complete"):
+ status = "done"
+ elif status in ("active", "in_progress", "in-progress", "working"):
+ status = "active"
+ else:
+ status = "pending"
+ counts[status] = counts.get(status, 0) + 1
+
+ parts = []
+ for key in ("active", "pending", "done"):
+ if counts.get(key, 0) > 0:
+ parts.append(f"{counts[key]} {key}")
+ return " | ".join(parts) if parts else f"{len(items)} items"
+
+
+def _format_single_todo(item: dict) -> Text:
+ """Format a single todo item with status symbol."""
+ status = str(item.get("status", "todo")).lower()
+ content_text = str(item.get("content", item.get("task", item.get("title", ""))))
+
+ if status in ("done", "completed", "complete"):
+ symbol = "\u2713"
+ label = "done "
+ style = "green dim"
+ elif status in ("active", "in_progress", "in-progress", "working"):
+ symbol = "\u25cf"
+ label = "active"
+ style = "yellow"
+ else:
+ symbol = "\u25cb"
+ label = "todo "
+ style = "dim"
+
+ line = Text()
+ line.append(f" {symbol} ", style=style)
+ line.append(label, style=style)
+ line.append(" ", style="dim")
+ # Truncate long content
+ if len(content_text) > 60:
+ content_text = content_text[:57] + "..."
+ line.append(content_text, style=style)
+ return line
+
+
+def format_tool_result_compact(_name: str, content: str, max_lines: int = 5) -> list:
+ """Format tool result as tree output.
+
+ Special handling for write_todos: shows formatted checklist with status symbols.
+ """
+ elements = []
+
+ if not content.strip():
+ elements.append(Text(" \u2514 (empty)", style="dim"))
+ return elements
+
+ # Special handling for write_todos
+ if _name == "write_todos":
+ items = _parse_todo_items(content)
+ if items:
+ stats = _build_todo_stats(items)
+ stats_line = Text()
+ stats_line.append(" \u2514 ", style="dim")
+ stats_line.append(stats, style="dim")
+ elements.append(stats_line)
+ elements.append(Text("", style="dim")) # blank line
+
+ max_preview = 4
+ for item in items[:max_preview]:
+ elements.append(_format_single_todo(item))
+
+ remaining = len(items) - max_preview
+ if remaining > 0:
+ elements.append(Text(f" ... {remaining} more", style="dim italic"))
+
+ return elements
+
+ lines = content.strip().split("\n")
+ total_lines = len(lines)
+
+ display_lines = lines[:max_lines]
+ for i, line in enumerate(display_lines):
+ prefix = "\u2514" if i == 0 else " "
+ if len(line) > 80:
+ line = line[:77] + "..."
+ style = "dim" if is_success(content) else "red dim"
+ elements.append(Text(f" {prefix} {line}", style=style))
+
+ remaining = total_lines - max_lines
+ if remaining > 0:
+ elements.append(Text(f" ... +{remaining} lines", style="dim italic"))
+
+ return elements
+
+
+def _render_tool_call_line(tc: dict, tr: dict | None) -> Text:
+ """Render a single tool call line with status indicator."""
+ is_task = tc.get('name', '').lower() == 'task'
+
+ if tr is not None:
+ content = tr.get('content', '')
+ if is_success(content):
+ style = "bold green"
+ indicator = "\u2713" if is_task else ToolStatus.SUCCESS.value
+ else:
+ style = "bold red"
+ indicator = "\u2717" if is_task else ToolStatus.ERROR.value
+ else:
+ style = "bold yellow" if not is_task else "bold cyan"
+ indicator = "\u25b6" if is_task else ToolStatus.RUNNING.value
+
+ tool_compact = format_tool_compact(tc['name'], tc.get('args'))
+ tool_text = Text()
+ tool_text.append(f"{indicator} ", style=style)
+ tool_text.append(tool_compact, style=style)
+ return tool_text
+
+
+def _render_subagent_section(sa: 'SubAgentState', compact: bool = False) -> list:
+ """Render a sub-agent's activity as a compact indented section.
+
+ Args:
+ sa: Sub-agent state to render
+ compact: If True, render minimal 1-2 line summary (for final display)
+
+ Completed tools are collapsed into a summary line.
+ Only the currently running tool is shown expanded.
+ """
+ elements = []
+ BORDER = "dim cyan" if sa.is_active else "dim"
+
+ # Filter out tool calls with empty names
+ valid_calls = [tc for tc in sa.tool_calls if tc.get("name")]
+
+ # Split into completed and pending
+ completed = []
+ pending = []
+ for tc in valid_calls:
+ tr = sa.get_result_for(tc)
+ if tr is not None:
+ completed.append((tc, tr))
+ else:
+ pending.append(tc)
+
+ succeeded = sum(1 for _, tr in completed if tr.get("success", True))
+ failed = len(completed) - succeeded
+
+ # --- Compact mode: 1-2 line summary for final display ---
+ if compact:
+ line = Text()
+ if not sa.is_active:
+ line.append(" \u2713 ", style="green")
+ line.append(sa.name, style="bold green")
+ else:
+ line.append(" \u25b6 ", style="cyan")
+ line.append(sa.name, style="bold cyan")
+ if sa.description:
+ desc = sa.description[:50] + "..." if len(sa.description) > 50 else sa.description
+ line.append(f" \u2014 {desc}", style="dim")
+ elements.append(line)
+ # Stats line
+ if valid_calls:
+ stats = Text(" ")
+ stats.append(f"{succeeded} completed", style="dim green")
+ if failed > 0:
+ stats.append(f" \u00b7 {failed} failed", style="dim red")
+ if pending:
+ stats.append(f" \u00b7 {len(pending)} running", style="dim yellow")
+ elements.append(stats)
+ return elements
+
+ # --- Full mode: bordered section for Live streaming ---
+ # Shows every tool call individually with status indicators
+
+ # Header
+ header = Text()
+ header.append(" \u250c ", style=BORDER)
+ if sa.is_active:
+ header.append(sa.name, style="bold cyan")
+ else:
+ header.append(sa.name, style="bold green")
+ header.append(" \u2713", style="green")
+ if sa.description:
+ desc = sa.description[:55] + "..." if len(sa.description) > 55 else sa.description
+ header.append(f" \u2014 {desc}", style="dim")
+ elements.append(header)
+
+ # Show every tool call with its status
+ for tc, tr in completed:
+ tc_line = Text(" \u2502 ", style=BORDER)
+ tc_name = format_tool_compact(tc["name"], tc.get("args"))
+ if tr.get("success", True):
+ tc_line.append(f"\u2713 {tc_name}", style="green")
+ else:
+ tc_line.append(f"\u2717 {tc_name}", style="red")
+ # Show first line of error
+ content = tr.get("content", "")
+ first_line = content.strip().split("\n")[0][:70]
+ if first_line:
+ err_line = Text(" \u2502 ", style=BORDER)
+ err_line.append(f"\u2514 {first_line}", style="red dim")
+ elements.append(tc_line)
+ elements.append(err_line)
+ continue
+ elements.append(tc_line)
+
+ # Pending/running tools
+ for tc in pending:
+ tc_line = Text(" \u2502 ", style=BORDER)
+ tc_name = format_tool_compact(tc["name"], tc.get("args"))
+ tc_line.append(f"\u25cf {tc_name}", style="bold yellow")
+ elements.append(tc_line)
+ spinner_line = Text(" \u2502 ", style=BORDER)
+ spinner_line.append("\u21bb running...", style="yellow dim")
+ elements.append(spinner_line)
+
+ # Footer
+ if not sa.is_active:
+ total = len(valid_calls)
+ footer = Text(f" \u2514 done ({total} tools)", style="dim green")
+ elements.append(footer)
+ elif valid_calls:
+ footer = Text(" \u2514 running...", style="dim cyan")
+ elements.append(footer)
+
+ return elements
+
+
+def create_streaming_display(
+ thinking_text: str = "",
+ response_text: str = "",
+ tool_calls: list | None = None,
+ tool_results: list | None = None,
+ is_thinking: bool = False,
+ is_responding: bool = False,
+ is_waiting: bool = False,
+ is_processing: bool = False,
+ show_thinking: bool = True,
+ subagents: list | None = None,
+) -> Any:
+ """Create Rich display layout for streaming output.
+
+ Returns:
+ Rich Group for Live display
+ """
+ elements = []
+ tool_calls = tool_calls or []
+ tool_results = tool_results or []
+ subagents = subagents or []
+
+ # Initial waiting state
+ if is_waiting and not thinking_text and not response_text and not tool_calls:
+ spinner = Spinner("dots", text=" Thinking...", style="cyan")
+ elements.append(spinner)
+ return Group(*elements)
+
+ # Thinking panel
+ if show_thinking and thinking_text:
+ thinking_title = "Thinking"
+ if is_thinking:
+ thinking_title += " ..."
+ display_thinking = thinking_text
+ if len(display_thinking) > DisplayLimits.THINKING_STREAM:
+ display_thinking = "..." + display_thinking[-DisplayLimits.THINKING_STREAM:]
+ elements.append(Panel(
+ Text(display_thinking, style="dim"),
+ title=thinking_title,
+ border_style="blue",
+ padding=(0, 1),
+ ))
+
+ # Tool calls and results paired display
+ # Collapse older completed tools to prevent overflow in Live mode
+ MAX_VISIBLE_TOOLS = 4
+
+ if tool_calls:
+ # Split into completed and pending/running
+ completed_tools = []
+ recent_tools = [] # last few completed + all pending
+
+ for i, tc in enumerate(tool_calls):
+ has_result = i < len(tool_results)
+ tr = tool_results[i] if has_result else None
+ if has_result:
+ completed_tools.append((tc, tr))
+ else:
+ recent_tools.append((tc, None))
+
+ # Determine how many completed tools to show
+ # Keep the last few completed + all pending within MAX_VISIBLE_TOOLS
+ slots_for_completed = max(0, MAX_VISIBLE_TOOLS - len(recent_tools))
+ hidden_completed = completed_tools[:-slots_for_completed] if slots_for_completed and len(completed_tools) > slots_for_completed else (completed_tools if not slots_for_completed else [])
+ visible_completed = completed_tools[-slots_for_completed:] if slots_for_completed else []
+
+ # Summary line for hidden completed tools
+ if hidden_completed:
+ ok = sum(1 for _, tr in hidden_completed if is_success(tr.get('content', '')))
+ fail = len(hidden_completed) - ok
+ summary = Text()
+ summary.append(f"\u2713 {ok} completed", style="dim green")
+ if fail > 0:
+ summary.append(f" | {fail} failed", style="dim red")
+ elements.append(summary)
+
+ # Render visible completed tools (compact: 1 line each, no result expansion)
+ for tc, tr in visible_completed:
+ elements.append(_render_tool_call_line(tc, tr))
+ # Only expand result for write_todos (useful) or errors
+ content = tr.get('content', '') if tr else ''
+ if tc.get('name') == 'write_todos' or (tr and not is_success(content)):
+ result_elements = format_tool_result_compact(
+ tr['name'],
+ content,
+ max_lines=5,
+ )
+ elements.extend(result_elements)
+
+ # Render pending/running tools (expanded with spinner)
+ for tc, tr in recent_tools:
+ elements.append(_render_tool_call_line(tc, tr))
+ if tc.get('name') != 'task':
+ spinner = Spinner("dots", text=" Running...", style="yellow")
+ elements.append(spinner)
+
+ # Sub-agent activity sections
+ for sa in subagents:
+ if sa.tool_calls or sa.is_active:
+ elements.extend(_render_subagent_section(sa))
+
+ # Processing state after tool execution
+ if is_processing and not is_thinking and not is_responding and not response_text:
+ # Check if any sub-agent is active
+ any_active = any(sa.is_active for sa in subagents)
+ if not any_active:
+ spinner = Spinner("dots", text=" Analyzing results...", style="cyan")
+ elements.append(spinner)
+
+ # Response text display logic
+ has_pending_tools = len(tool_calls) > len(tool_results)
+ any_active_subagent = any(sa.is_active for sa in subagents)
+ has_used_tools = len(tool_calls) > 0
+
+ if response_text and not has_pending_tools and not any_active_subagent:
+ if has_used_tools:
+ # Tools were used — treat all text as intermediate during Live streaming.
+ # Final rendering is handled by display_final_results().
+ preview = response_text
+ if len(preview) > 200:
+ preview = "..." + preview[-197:]
+ for line in preview.strip().split("\n")[-3:]:
+ if line.strip():
+ elements.append(Text(f" {line.strip()}", style="dim italic"))
+ else:
+ # Pure text response (no tools used) — render as Markdown
+ elements.append(Text("")) # blank separator
+ elements.append(Markdown(response_text))
+ elif is_responding and not thinking_text and not has_pending_tools:
+ elements.append(Text("Generating response...", style="dim"))
+
+ return Group(*elements) if elements else Text("Processing...", style="dim")
+
+
+def display_final_results(
+ state: StreamState,
+ thinking_max_length: int = DisplayLimits.THINKING_FINAL,
+ show_thinking: bool = True,
+ show_tools: bool = True,
+) -> None:
+ """Display final results after streaming completes."""
+ if show_thinking and state.thinking_text:
+ display_thinking = state.thinking_text
+ if len(display_thinking) > thinking_max_length:
+ half = thinking_max_length // 2
+ display_thinking = display_thinking[:half] + "\n\n... (truncated) ...\n\n" + display_thinking[-half:]
+ console.print(Panel(
+ Text(display_thinking, style="dim"),
+ title="Thinking",
+ border_style="blue",
+ ))
+
+ if show_tools and state.tool_calls:
+ shown_sa_names: set[str] = set()
+
+ for i, tc in enumerate(state.tool_calls):
+ has_result = i < len(state.tool_results)
+ tr = state.tool_results[i] if has_result else None
+ content = tr.get('content', '') if tr is not None else ''
+ is_task = tc.get('name', '').lower() == 'task'
+
+ # Task tools: show delegation line + compact sub-agent summary
+ if is_task:
+ console.print(_render_tool_call_line(tc, tr))
+ sa_name = tc.get('args', {}).get('subagent_type', '')
+ task_desc = tc.get('args', {}).get('description', '')
+ matched_sa = None
+ for sa in state.subagents:
+ if sa.name == sa_name or (task_desc and task_desc in (sa.description or '')):
+ matched_sa = sa
+ break
+ if matched_sa:
+ shown_sa_names.add(matched_sa.name)
+ for elem in _render_subagent_section(matched_sa, compact=True):
+ console.print(elem)
+ continue
+
+ # Regular tools: show tool call line + result
+ console.print(_render_tool_call_line(tc, tr))
+ if has_result and tr is not None:
+ result_elements = format_tool_result_compact(
+ tr['name'],
+ content,
+ max_lines=10,
+ )
+ for elem in result_elements:
+ console.print(elem)
+
+ # Render any sub-agents not already shown via task tool calls
+ for sa in state.subagents:
+ if sa.name not in shown_sa_names and (sa.tool_calls or sa.is_active):
+ for elem in _render_subagent_section(sa, compact=True):
+ console.print(elem)
+
+ console.print()
+
+ if state.response_text:
+ console.print()
+ console.print(Markdown(state.response_text))
+ console.print()
+
+
+# =============================================================================
+# Async-to-sync bridge
+# =============================================================================
+
+def _run_streaming(
+ agent: Any,
+ message: str,
+ thread_id: str,
+ show_thinking: bool,
+ interactive: bool,
+) -> None:
+ """Run async streaming and render with Rich Live display.
+
+ Bridges the async stream_agent_events() into synchronous Rich Live rendering
+ using asyncio.run().
+
+ Args:
+ agent: Compiled agent graph
+ message: User message
+ thread_id: Thread ID
+ show_thinking: Whether to show thinking panel
+ interactive: If True, use simplified final display (no panel)
+ """
+ state = StreamState()
+
+ async def _consume() -> None:
+ async for event in stream_agent_events(agent, message, thread_id):
+ event_type = state.handle_event(event)
+ live.update(create_streaming_display(
+ **state.get_display_args(),
+ show_thinking=show_thinking,
+ ))
+ if event_type in (
+ "tool_call", "tool_result",
+ "subagent_start", "subagent_tool_call",
+ "subagent_tool_result", "subagent_end",
+ ):
+ live.refresh()
+
+ with Live(console=console, refresh_per_second=10, transient=True) as live:
+ live.update(create_streaming_display(is_waiting=True))
+ asyncio.run(_consume())
+
+ if interactive:
+ display_final_results(
+ state,
+ thinking_max_length=500,
+ show_thinking=False,
+ show_tools=True,
+ )
+ else:
+ console.print()
+ display_final_results(
+ state,
+ show_tools=True,
+ )
+
+
+# =============================================================================
+# CLI commands
+# =============================================================================
+
+EVOSCIENTIST_ASCII_LINES = [
+ r" ███████╗ ██╗ ██╗ ██████╗ ███████╗ ██████╗ ██╗ ███████╗ ███╗ ██╗ ████████╗ ██╗ ███████╗ ████████╗",
+ r" ██╔════╝ ██║ ██║ ██╔═══██╗ ██╔════╝ ██╔════╝ ██║ ██╔════╝ ████╗ ██║ ╚══██╔══╝ ██║ ██╔════╝ ╚══██╔══╝",
+ r" █████╗ ██║ ██║ ██║ ██║ ███████╗ ██║ ██║ █████╗ ██╔██╗ ██║ ██║ ██║ ███████╗ ██║ ",
+ r" ██╔══╝ ╚██╗ ██╔╝ ██║ ██║ ╚════██║ ██║ ██║ ██╔══╝ ██║╚██╗██║ ██║ ██║ ╚════██║ ██║ ",
+ r" ███████╗ ╚████╔╝ ╚██████╔╝ ███████║ ╚██████╗ ██║ ███████╗ ██║ ╚████║ ██║ ██║ ███████║ ██║ ",
+ r" ╚══════╝ ╚═══╝ ╚═════╝ ╚══════╝ ╚═════╝ ╚═╝ ╚══════╝ ╚═╝ ╚═══╝ ╚═╝ ╚═╝ ╚══════╝ ╚═╝ ",
+]
+
+# Blue gradient: deep navy → royal blue → sky blue → cyan
+_GRADIENT_COLORS = ["#1a237e", "#1565c0", "#1e88e5", "#42a5f5", "#64b5f6", "#90caf9"]
+
+
+def print_banner(thread_id: str, workspace_dir: str | None = None):
+ """Print welcome banner with ASCII art logo, thread ID, and workspace path."""
+ for line, color in zip(EVOSCIENTIST_ASCII_LINES, _GRADIENT_COLORS):
+ console.print(Text(line, style=f"{color} bold"))
+ info = Text()
+ info.append(" Thread: ", style="dim")
+ info.append(thread_id, style="yellow")
+ if workspace_dir:
+ info.append("\n Workspace: ", style="dim")
+ info.append(workspace_dir, style="cyan")
+ info.append("\n Commands: ", style="dim")
+ info.append("/exit", style="bold")
+ info.append(", ", style="dim")
+ info.append("/new", style="bold")
+ info.append(" (new session), ", style="dim")
+ info.append("/thread", style="bold")
+ info.append(" (show thread ID)", style="dim")
+ console.print(info)
+ console.print()
+
+
+def cmd_interactive(agent: Any, show_thinking: bool = True, workspace_dir: str | None = None) -> None:
+ """Interactive conversation mode with streaming output.
+
+ Args:
+ agent: Compiled agent graph
+ show_thinking: Whether to display thinking panels
+ workspace_dir: Per-session workspace directory path
+ """
+ thread_id = str(uuid.uuid4())
+ print_banner(thread_id, workspace_dir)
+
+ history_file = str(os.path.expanduser("~/.EvoScientist_history"))
+ session = PromptSession(
+ history=FileHistory(history_file),
+ auto_suggest=AutoSuggestFromHistory(),
+ enable_history_search=True,
+ )
+
+ while True:
+ try:
+ user_input = session.prompt(
+ HTML('You: ')
+ ).strip()
+
+ if not user_input:
+ continue
+
+ # Special commands
+ if user_input.lower() in ("/exit", "/quit", "/q"):
+ console.print("[dim]Goodbye![/dim]")
+ break
+
+ if user_input.lower() == "/new":
+ # New session: new workspace, new agent, new thread
+ workspace_dir = _create_session_workspace()
+ console.print("[dim]Loading new session...[/dim]")
+ agent = _load_agent(workspace_dir=workspace_dir)
+ thread_id = str(uuid.uuid4())
+ console.print(f"[green]New session:[/green] [yellow]{thread_id}[/yellow]")
+ console.print(f"[dim]Workspace:[/dim] [cyan]{workspace_dir}[/cyan]\n")
+ continue
+
+ if user_input.lower() == "/thread":
+ console.print(f"[dim]Thread:[/dim] [yellow]{thread_id}[/yellow]")
+ if workspace_dir:
+ console.print(f"[dim]Workspace:[/dim] [cyan]{workspace_dir}[/cyan]")
+ console.print()
+ continue
+
+ # Stream agent response
+ console.print()
+ _run_streaming(agent, user_input, thread_id, show_thinking, interactive=True)
+
+ except KeyboardInterrupt:
+ console.print("\n[dim]Goodbye![/dim]")
+ break
+ except Exception as e:
+ console.print(f"[red]Error: {e}[/red]")
+
+
+def cmd_run(agent: Any, prompt: str, thread_id: str | None = None, show_thinking: bool = True, workspace_dir: str | None = None) -> None:
+ """Single-shot execution with streaming display.
+
+ Args:
+ agent: Compiled agent graph
+ prompt: User prompt
+ thread_id: Optional thread ID (generates new one if None)
+ show_thinking: Whether to display thinking panels
+ workspace_dir: Per-session workspace directory path
+ """
+ thread_id = thread_id or str(uuid.uuid4())
+
+ console.print(Panel(f"[bold cyan]Query:[/bold cyan]\n{prompt}"))
+ console.print(f"[dim]Thread: {thread_id}[/dim]")
+ if workspace_dir:
+ console.print(f"[dim]Workspace: {workspace_dir}[/dim]")
+ console.print()
+
+ try:
+ _run_streaming(agent, prompt, thread_id, show_thinking, interactive=False)
+ except Exception as e:
+ console.print(f"[red]Error: {e}[/red]")
+ raise
+
+
+# =============================================================================
+# Entry point
+# =============================================================================
+
+def _create_session_workspace() -> str:
+ """Create a per-session workspace directory and return its path."""
+ session_id = datetime.now().strftime("%Y%m%d_%H%M%S")
+ workspace_dir = os.path.join(".", "workspace", session_id)
+ os.makedirs(workspace_dir, exist_ok=True)
+ return workspace_dir
+
+
+def _load_agent(workspace_dir: str | None = None):
+ """Load the CLI agent (with InMemorySaver checkpointer for multi-turn).
+
+ Args:
+ workspace_dir: Optional per-session workspace directory.
+ """
+ from .EvoScientist import create_cli_agent
+ return create_cli_agent(workspace_dir=workspace_dir)
+
+
+def main():
+ """CLI entry point."""
+ parser = argparse.ArgumentParser(
+ description="EvoScientist Agent - AI-powered research & code execution CLI",
+ formatter_class=argparse.RawDescriptionHelpFormatter,
+ epilog="""
+Examples:
+ # Interactive mode (default)
+ python -m EvoScientist --interactive
+
+ # Single-shot query
+ python -m EvoScientist "What is quantum computing?"
+
+ # Resume a conversation thread
+ python -m EvoScientist --thread-id "Follow-up question"
+
+ # Disable thinking display
+ python -m EvoScientist --no-thinking "Your query"
+""",
+ )
+
+ parser.add_argument(
+ "prompt",
+ nargs="?",
+ help="Query to execute (single-shot mode)",
+ )
+ parser.add_argument(
+ "-i", "--interactive",
+ action="store_true",
+ help="Interactive conversation mode",
+ )
+ parser.add_argument(
+ "--thread-id",
+ type=str,
+ default=None,
+ help="Thread ID for conversation persistence (resume session)",
+ )
+ parser.add_argument(
+ "--no-thinking",
+ action="store_true",
+ help="Disable thinking display",
+ )
+
+ args = parser.parse_args()
+ show_thinking = not args.no_thinking
+
+ # Create per-session workspace
+ workspace_dir = _create_session_workspace()
+
+ # Load agent with session workspace
+ console.print("[dim]Loading agent...[/dim]")
+ agent = _load_agent(workspace_dir=workspace_dir)
+
+ if args.interactive:
+ cmd_interactive(agent, show_thinking=show_thinking, workspace_dir=workspace_dir)
+ elif args.prompt:
+ cmd_run(agent, args.prompt, thread_id=args.thread_id, show_thinking=show_thinking, workspace_dir=workspace_dir)
+ else:
+ # Default: interactive mode
+ cmd_interactive(agent, show_thinking=show_thinking, workspace_dir=workspace_dir)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/EvoScientist/middleware.py b/EvoScientist/middleware.py
new file mode 100644
index 0000000..44734b1
--- /dev/null
+++ b/EvoScientist/middleware.py
@@ -0,0 +1,27 @@
+"""Middleware configuration for the EvoScientist agent."""
+
+from deepagents.backends import FilesystemBackend
+from deepagents.middleware.skills import SkillsMiddleware
+
+
+def create_skills_middleware(
+ skills_dir: str = "./skills/",
+ workspace_dir: str = "./workspace/",
+) -> SkillsMiddleware:
+ """Create a SkillsMiddleware that loads skills.
+
+ All skills (system and user-installed) live in ./skills/.
+ The --user flag in install_skill.py also installs to ./skills/.
+
+ Args:
+ skills_dir: Path to the skills directory
+ workspace_dir: Unused, kept for API compatibility
+
+ Returns:
+ Configured SkillsMiddleware instance
+ """
+ skills_backend = FilesystemBackend(root_dir=skills_dir, virtual_mode=True)
+ return SkillsMiddleware(
+ backend=skills_backend,
+ sources=["/"],
+ )
diff --git a/EvoScientist/prompts.py b/EvoScientist/prompts.py
new file mode 100644
index 0000000..10a782f
--- /dev/null
+++ b/EvoScientist/prompts.py
@@ -0,0 +1,277 @@
+"""Prompt templates for the EvoScientist experimental agent."""
+
+# =============================================================================
+# Main agent workflow
+# =============================================================================
+
+EXPERIMENT_WORKFLOW = """# Experiment Workflow
+
+You are the main experimental agent. Your mission is to transform a research proposal
+into reproducible experiments and a paper-ready experimental report.
+
+## Core Principles
+- Baseline first, then iterate (ablation-friendly).
+- Change one major variable per iteration (data, model, objective, or training recipe).
+- Never invent results. If you cannot run something, say so and propose the smallest next step.
+- Delegate aggressively using the `task` tool. Prefer the research sub-agent for web search.
+- Use local skills via `load_skill` when they match the task. Skills provide proven workflows and checklists.
+ All skills are available under `/skills/` (read-only).
+ When calling `load_skill`, use the skill id from the SKILL.md frontmatter (`name:`), not the folder name.
+
+## Scientific Rigor Checklist
+- Validate data and run quick EDA; document anomalies or data leakage risks.
+- Separate exploratory vs confirmatory analyses; define primary metrics up front.
+- Report effect sizes with uncertainty (confidence intervals/error bars) where possible.
+- Apply multiple-testing correction when comparing many conditions.
+- State limitations, negative results, and sensitivity to key parameters.
+- Track reproducibility (seeds, versions, configs, and exact commands).
+
+## Step 1: Intake & Scope
+- Read the proposal and extract goals, datasets, constraints, and evaluation metrics
+- Capture key assumptions and open questions
+- Save the original proposal to `/research_request.md`
+
+## Step 2: Plan (Recommended Structure)
+- Create experiment stages with success signals (flexible, not rigid)
+- Identify resource/data dependencies and baseline requirements
+- Use `write_todos` to track the execution plan and updates
+- If delegating planning to planner-agent, start your message with: `MODE: PLAN`
+- If a stage matches an existing skill, note the skill name in the plan and load it before implementation.
+ Use the skill id from SKILL.md frontmatter (`name:`).
+-- Save the plan to `/todos.md` (recommended). Include per-stage:
+ - objective and success signals
+ - what to run (commands/scripts)
+ - expected artifacts (tables/plots/logs)
+- Optionally save:
+ - `/plan.md` for stages
+ - `/success_criteria.md` for success signals
+
+## Step 3: Execute & Debug
+- Delegate tasks to sub-agents using the `task` tool:
+ - Planning/structuring → planner-agent
+ - Methods/baselines/datasets → research-agent
+ - Implementation → code-agent
+ - Debugging → debug-agent
+ - Analysis/visualization → data-analysis-agent
+ - Report drafting → writing-agent
+- Prefer the research-agent for web search; avoid searching directly
+- Use `execute` for shell commands when running experiments
+- When a task matches an existing skill, `load_skill` it and follow it rather than reinventing the workflow.
+- Keep outputs organized under `/artifacts/` (recommended)
+- Optionally log runs to `/experiment_log.md` (params, seeds, env, outputs)
+
+## Step 4: Evaluate & Iterate
+- Compare results against success signals
+- If results are weak or ambiguous, iterate:
+ - identify gaps
+ - propose new methods/data
+ - re-run and re-evaluate
+- Prefer evidence-driven iteration: error analysis, sanity checks, and minimal ablations
+- Update `/todos.md` to reflect new iterations
+- Stop iterating when evidence is sufficient or diminishing returns appear
+
+### Stage Reflection (Recommended Checkpoint)
+After any meaningful experimental stage (baseline, new dataset, new training recipe, etc.),
+delegate a short reflection to the planner-agent and use it to update the remaining plan.
+
+Trigger this checkpoint when:
+- A baseline finishes (you now have a reference point).
+- You introduce a new dataset/model/training recipe (risk of confounding changes).
+- Two iterations in a row fail to improve the primary metric.
+- Results look suspicious (metric mismatch, unstable training, unexpected regressions).
+
+When calling the planner-agent in reflection mode, provide:
+- Start your message with: `MODE: REFLECTION`
+- Stage name/index and intent
+- Commands run + key parameters (model, dataset, seeds, batch size, lr, epochs, hardware)
+- Key metrics vs baseline (a small table is ideal)
+- Artifact paths (logs, plots, checkpoints)
+- Which success signals were met/unmet
+- If proposing skills, use skill ids from SKILL.md frontmatter (`name:`).
+
+Ask the planner-agent to output a **Plan Update JSON** with this schema:
+```json
+{
+ "completed": ["..."],
+ "unmet_success_signals": ["..."],
+ "skill_suggestions": ["..."],
+ "stage_modifications": [
+ {"stage": "Stage name or index", "change": "What to adjust and why"}
+ ],
+ "new_stages": [
+ {
+ "title": "...",
+ "goal": "...",
+ "success_signals": ["..."],
+ "what_to_run": ["..."],
+ "expected_artifacts": ["..."]
+ }
+ ],
+ "todo_updates": ["..."]
+}
+```
+Empty arrays are valid. If no changes are needed, return the JSON with empty arrays.
+Then revise `/todos.md` accordingly.
+
+## Step 5: Write Report
+- Write the final report to `/final_report.md` (Markdown)
+- Include:
+ - Problem summary
+ - Experiment plan (stages + success signals)
+ - Experimental setup and configurations
+ - Results and visualizations (reference artifacts)
+ - Analysis, limitations, and next steps
+- If web research was used, include a Sources section with real URLs (no fabricated citations)
+- When applicable, include effect sizes, uncertainty, and notes on statistical corrections.
+- Be precise, technical, and concise
+
+## Step 6: Verify
+- Re-read `/research_request.md` to ensure coverage
+- Confirm the report answers the proposal and documents key settings/results
+
+## Experiment Report Template (Recommended)
+1. Summary & goals
+2. Experiment plan (stages + success signals)
+3. Setup (data, model, environment, parameters)
+4. Baselines and comparisons
+5. Results (tables/figures + references to artifacts)
+6. Analysis, limitations, and next steps
+
+## Writing Guidelines
+- Use bullets for configs, stage lists, and key results; use short paragraphs for reasoning
+- Avoid first-person singular ("I ..."). Prefer neutral phrasing ("This experiment...") or "we" style.
+- Professional, objective tone
+
+## Shell Execution Guidelines
+When using the `execute` tool for shell commands:
+
+**Short commands** (< 30 seconds): Run directly
+```bash
+python script.py
+pip install pandas
+```
+
+**Long-running commands** (> 30 seconds): Run in background, then check results
+```bash
+# Step 1: Start in background, redirect output to log
+python long_task.py > /output.log 2>&1 &
+
+# Step 2: Check if still running
+ps aux | grep long_task
+
+# Step 3: Read results when done
+cat /output.log
+```
+
+This prevents blocking the conversation during long operations.
+"""
+
+# =============================================================================
+# Sub-agent delegation strategy
+# =============================================================================
+
+DELEGATION_STRATEGY = """# Sub-Agent Delegation
+
+## Default: Use 1 Sub-Agent
+For most tasks, a single sub-agent is sufficient:
+- "Plan experimental stages" → planner-agent
+- "Reflect and update the plan after a stage" → planner-agent
+- "Find related methods/baselines/datasets" → research-agent
+- "Implement baseline or training loop" → code-agent
+- "Debug runtime failures" → debug-agent
+- "Analyze metrics and plot figures" → data-analysis-agent
+- "Draft report sections" → writing-agent
+
+## Task Granularity
+- One sub-agent task = one topic / one experiment / one artifact bundle
+- Provide concrete file paths, commands, and success signals in each task
+ so the sub-agent can respond precisely
+
+## Parallelize Only When Necessary
+Use multiple sub-agents ONLY for:
+
+**Explicit comparisons** (1 per method/baseline):
+- "Compare A vs B vs C" → 3 parallel sub-agents
+
+**Distinct experiments** with separate datasets or setups:
+- "Run baselines on X and Y" → 2 parallel sub-agents
+
+## Limits
+- Maximum {max_concurrent} parallel sub-agents per round
+- Maximum {max_iterations} delegation rounds total
+- Stop when evidence is sufficient
+
+## Key Principles
+- Bias towards a single sub-agent (token-efficient)
+- Avoid premature decomposition
+- Each sub-agent returns focused, self-contained findings
+"""
+
+# =============================================================================
+# Sub-agent research instructions
+# =============================================================================
+
+RESEARCHER_INSTRUCTIONS = """You are a research assistant. Today's date is {date}.
+
+## Task
+Use tools to gather information on the assigned topic (methods, baselines,
+datasets, or prior results) to support experimental planning or iteration.
+Prefer actionable details: datasets, metrics, code availability, and common pitfalls.
+Do not fabricate citations or URLs.
+Capture evaluation protocols (splits, metrics, calibration) and known failure modes.
+
+## Available Tools
+1. **tavily_search** - Web search for information
+2. **think_tool** - Reflect on findings and plan next steps
+
+**CRITICAL: Use think_tool after each search**
+
+## Research Strategy
+1. Read the question carefully
+2. Start with broad searches
+3. After each search, reflect: Do I have enough? What's missing?
+4. Narrow searches to fill gaps
+5. Stop when you can answer confidently
+
+## Hard Limits
+- Simple queries: 2-3 searches maximum
+- Complex queries: up to 5 searches maximum
+- Stop after 5 searches regardless
+
+## Stop When
+- You can answer comprehensively
+- You have 3+ relevant sources
+- Last 2 searches returned similar information
+
+## Response Format
+Structure findings with clear headings and cite sources inline:
+
+```
+## Key Findings
+
+Finding one with context [1]. Another insight [2].
+
+## Recommended Next Experiments
+- One actionable experiment suggestion with motivation and expected outcome.
+
+### Sources
+[1] Title: URL
+[2] Title: URL
+```
+"""
+
+# =============================================================================
+# Combined exports
+# =============================================================================
+
+def get_system_prompt(max_concurrent: int = 3, max_iterations: int = 3) -> str:
+ """Generate the complete system prompt with configured limits."""
+ delegation = DELEGATION_STRATEGY.format(
+ max_concurrent=max_concurrent,
+ max_iterations=max_iterations,
+ )
+ return EXPERIMENT_WORKFLOW + "\n" + delegation
+
+
+# Default export (backward compatible)
+SYSTEM_PROMPT = get_system_prompt()
diff --git a/EvoScientist/stream/__init__.py b/EvoScientist/stream/__init__.py
new file mode 100644
index 0000000..12e28c5
--- /dev/null
+++ b/EvoScientist/stream/__init__.py
@@ -0,0 +1,53 @@
+"""
+Stream module - streaming event processing for CLI display.
+
+Provides:
+- StreamEventEmitter: Standardized event creation
+- ToolCallTracker: Incremental JSON parsing for tool parameters
+- ToolResultFormatter: Content-aware result formatting with Rich
+- Utility functions and constants
+"""
+
+from .emitter import StreamEventEmitter, StreamEvent
+from .tracker import ToolCallTracker, ToolCallInfo
+from .formatter import ToolResultFormatter, ContentType, FormattedResult
+from .utils import (
+ SUCCESS_PREFIX,
+ FAILURE_PREFIX,
+ ToolStatus,
+ DisplayLimits,
+ has_args,
+ is_success,
+ truncate,
+ format_tool_compact,
+ format_tree_output,
+ count_lines,
+ truncate_with_line_hint,
+ get_status_symbol,
+)
+
+__all__ = [
+ # Emitter
+ "StreamEventEmitter",
+ "StreamEvent",
+ # Tracker
+ "ToolCallTracker",
+ "ToolCallInfo",
+ # Formatter
+ "ToolResultFormatter",
+ "ContentType",
+ "FormattedResult",
+ # Utils
+ "SUCCESS_PREFIX",
+ "FAILURE_PREFIX",
+ "ToolStatus",
+ "DisplayLimits",
+ "has_args",
+ "is_success",
+ "truncate",
+ "format_tool_compact",
+ "format_tree_output",
+ "count_lines",
+ "truncate_with_line_hint",
+ "get_status_symbol",
+]
diff --git a/EvoScientist/stream/emitter.py b/EvoScientist/stream/emitter.py
new file mode 100644
index 0000000..85e9e63
--- /dev/null
+++ b/EvoScientist/stream/emitter.py
@@ -0,0 +1,94 @@
+"""
+StreamEventEmitter - standardized event format.
+
+All events contain a type and associated data dict.
+"""
+
+from dataclasses import dataclass
+from typing import Any, Dict
+
+
+@dataclass
+class StreamEvent:
+ """Unified stream event."""
+ type: str
+ data: Dict[str, Any]
+
+
+class StreamEventEmitter:
+ """Stream event emitter - creates standardized event dicts."""
+
+ @staticmethod
+ def thinking(content: str, thinking_id: int = 0) -> StreamEvent:
+ """Thinking content event."""
+ return StreamEvent("thinking", {"type": "thinking", "content": content, "id": thinking_id})
+
+ @staticmethod
+ def text(content: str) -> StreamEvent:
+ """Text content event."""
+ return StreamEvent("text", {"type": "text", "content": content})
+
+ @staticmethod
+ def tool_call(name: str, args: Dict[str, Any], tool_id: str = "") -> StreamEvent:
+ """Tool call event."""
+ return StreamEvent("tool_call", {"type": "tool_call", "name": name, "args": args, "id": tool_id})
+
+ @staticmethod
+ def tool_result(name: str, content: str, success: bool = True) -> StreamEvent:
+ """Tool result event."""
+ return StreamEvent("tool_result", {
+ "type": "tool_result",
+ "name": name,
+ "content": content,
+ "success": success,
+ })
+
+ @staticmethod
+ def subagent_start(name: str, description: str) -> StreamEvent:
+ """Sub-agent delegation started."""
+ return StreamEvent("subagent_start", {
+ "type": "subagent_start",
+ "name": name,
+ "description": description,
+ })
+
+ @staticmethod
+ def subagent_tool_call(
+ subagent: str, name: str, args: Dict[str, Any], tool_id: str = ""
+ ) -> StreamEvent:
+ """Tool call from inside a sub-agent."""
+ return StreamEvent("subagent_tool_call", {
+ "type": "subagent_tool_call",
+ "subagent": subagent,
+ "name": name,
+ "args": args,
+ "id": tool_id,
+ })
+
+ @staticmethod
+ def subagent_tool_result(
+ subagent: str, name: str, content: str, success: bool = True
+ ) -> StreamEvent:
+ """Tool result from inside a sub-agent."""
+ return StreamEvent("subagent_tool_result", {
+ "type": "subagent_tool_result",
+ "subagent": subagent,
+ "name": name,
+ "content": content,
+ "success": success,
+ })
+
+ @staticmethod
+ def subagent_end(name: str) -> StreamEvent:
+ """Sub-agent delegation completed."""
+ return StreamEvent("subagent_end", {"type": "subagent_end", "name": name})
+
+ @staticmethod
+ def done(response: str = "") -> StreamEvent:
+ """Done event."""
+ return StreamEvent("done", {"type": "done", "response": response})
+
+ @staticmethod
+ def error(message: str) -> StreamEvent:
+ """Error event."""
+ return StreamEvent("error", {"type": "error", "message": message})
diff --git a/EvoScientist/stream/formatter.py b/EvoScientist/stream/formatter.py
new file mode 100644
index 0000000..f0fdf5e
--- /dev/null
+++ b/EvoScientist/stream/formatter.py
@@ -0,0 +1,168 @@
+"""
+ToolResultFormatter - content-aware tool result formatting with Rich.
+
+Detects content type (success/error/json/markdown/text) and formats accordingly.
+"""
+
+import json
+from dataclasses import dataclass
+from enum import Enum
+from typing import Any, List
+
+from rich.panel import Panel
+from rich.syntax import Syntax
+from rich.text import Text
+from rich.markdown import Markdown
+
+from .utils import SUCCESS_PREFIX, FAILURE_PREFIX, is_success as _is_success, truncate
+
+
+class ContentType(Enum):
+ """Content type categories."""
+ SUCCESS = "success"
+ ERROR = "error"
+ JSON = "json"
+ MARKDOWN = "markdown"
+ TEXT = "text"
+
+
+@dataclass
+class FormattedResult:
+ """Formatted result container."""
+ content_type: ContentType
+ elements: List[Any] # Rich renderable elements
+ success: bool = True
+
+
+class ToolResultFormatter:
+ """Tool result formatter with content type detection.
+
+ Usage:
+ formatter = ToolResultFormatter()
+ result = formatter.format("execute", output, max_length=800)
+ for elem in result.elements:
+ console.print(elem)
+ """
+
+ def detect_type(self, content: str) -> ContentType:
+ """Detect content type."""
+ content = content.strip()
+
+ if content.startswith(SUCCESS_PREFIX):
+ body = self._extract_body(content)
+ if self._is_json(body):
+ return ContentType.JSON
+ return ContentType.SUCCESS
+
+ if content.startswith(FAILURE_PREFIX):
+ return ContentType.ERROR
+
+ if self._is_json(content):
+ return ContentType.JSON
+
+ if self._is_error(content):
+ return ContentType.ERROR
+
+ if self._is_markdown(content):
+ return ContentType.MARKDOWN
+
+ return ContentType.TEXT
+
+ def is_success(self, content: str) -> bool:
+ """Check if content indicates successful execution."""
+ return _is_success(content)
+
+ def format(self, name: str, content: str, max_length: int = 800) -> FormattedResult:
+ """Format tool result based on detected content type."""
+ content_type = self.detect_type(content)
+ success = self.is_success(content)
+
+ formatter_map = {
+ ContentType.SUCCESS: self._format_success,
+ ContentType.ERROR: self._format_error,
+ ContentType.JSON: self._format_json,
+ ContentType.MARKDOWN: self._format_markdown,
+ ContentType.TEXT: self._format_text,
+ }
+
+ formatter = formatter_map.get(content_type, self._format_text)
+ elements = formatter(name, content, max_length)
+
+ return FormattedResult(content_type=content_type, elements=elements, success=success)
+
+ def _extract_body(self, content: str) -> str:
+ """Extract body after status prefix."""
+ lines = content.split("\n", 2)
+ return lines[2].strip() if len(lines) > 2 else ""
+
+ def _is_json(self, content: str) -> bool:
+ content = content.strip()
+ if not content:
+ return False
+ if (content.startswith('{') and content.endswith('}')) or \
+ (content.startswith('[') and content.endswith(']')):
+ try:
+ json.loads(content)
+ return True
+ except (json.JSONDecodeError, ValueError):
+ pass
+ return False
+
+ def _is_error(self, content: str) -> bool:
+ error_patterns = [
+ 'Traceback (most recent call last)',
+ 'Exception:',
+ 'Error:',
+ ]
+ return any(pattern in content for pattern in error_patterns)
+
+ def _is_markdown(self, content: str) -> bool:
+ md_patterns = ['```', '**', '##', '- **']
+ return content.startswith('#') or any(p in content for p in md_patterns)
+
+ def _format_success(self, name: str, content: str, max_length: int) -> List[Any]:
+ display = truncate(content, max_length)
+ return [Panel(
+ Text(display, style="green"),
+ title=f"{name} OK",
+ border_style="green",
+ )]
+
+ def _format_error(self, name: str, content: str, max_length: int) -> List[Any]:
+ display = truncate(content, max_length)
+ return [Panel(
+ Text(display, style="red"),
+ title=f"{name} FAILED",
+ border_style="red",
+ )]
+
+ def _format_json(self, name: str, content: str, max_length: int) -> List[Any]:
+ json_content = content
+ if content.startswith(SUCCESS_PREFIX):
+ json_content = self._extract_body(content)
+
+ try:
+ data = json.loads(json_content)
+ formatted = json.dumps(data, indent=2, ensure_ascii=False)
+ formatted = truncate(formatted, max_length)
+ return [
+ Text(f"{name} OK", style="cyan bold"),
+ Syntax(formatted, "json", theme="monokai", line_numbers=False),
+ ]
+ except (json.JSONDecodeError, ValueError):
+ return self._format_text(name, content, max_length)
+
+ def _format_markdown(self, name: str, content: str, max_length: int) -> List[Any]:
+ display = truncate(content, max_length)
+ return [Panel(
+ Markdown(display),
+ title=f"{name}",
+ border_style="cyan dim",
+ )]
+
+ def _format_text(self, name: str, content: str, max_length: int) -> List[Any]:
+ display = truncate(content, max_length)
+ return [
+ Text(f"{name}:", style="cyan bold"),
+ Text(f" {display}", style="dim"),
+ ]
diff --git a/EvoScientist/stream/tracker.py b/EvoScientist/stream/tracker.py
new file mode 100644
index 0000000..7971997
--- /dev/null
+++ b/EvoScientist/stream/tracker.py
@@ -0,0 +1,115 @@
+"""
+ToolCallTracker - manages incremental JSON parsing for tool parameters.
+
+Handles tool_use blocks where arguments arrive in fragments via input_json_delta.
+"""
+
+import json
+from dataclasses import dataclass, field
+from typing import Dict, Optional
+
+
+@dataclass
+class ToolCallInfo:
+ """Tool call information."""
+ id: str
+ name: str
+ args: Dict = field(default_factory=dict)
+ emitted: bool = False
+ args_complete: bool = False
+ _json_buffer: str = ""
+
+
+class ToolCallTracker:
+ """Tool call tracker for incremental argument parsing.
+
+ Usage:
+ tracker = ToolCallTracker()
+ tracker.update(tool_id, name="execute")
+ tracker.append_json_delta('{"command')
+ tracker.append_json_delta('": "ls"}')
+ tracker.finalize_all()
+ info = tracker.get(tool_id)
+ yield emitter.tool_call(info.name, info.args)
+ """
+
+ def __init__(self):
+ self._calls: Dict[str, ToolCallInfo] = {}
+ self._last_tool_id: Optional[str] = None
+
+ def update(
+ self,
+ tool_id: str,
+ name: Optional[str] = None,
+ args: Optional[Dict] = None,
+ args_complete: bool = False,
+ ) -> None:
+ """Update tool call info (accumulative)."""
+ if tool_id not in self._calls:
+ self._calls[tool_id] = ToolCallInfo(
+ id=tool_id,
+ name=name or "",
+ args=args or {},
+ args_complete=args_complete,
+ )
+ self._last_tool_id = tool_id
+ else:
+ info = self._calls[tool_id]
+ if name:
+ info.name = name
+ if args:
+ info.args = args
+ if args_complete:
+ info.args_complete = True
+
+ def append_json_delta(self, partial_json: str, index: int = 0) -> None:
+ """Accumulate input_json_delta fragment."""
+ tool_id = self._last_tool_id
+ if tool_id and tool_id in self._calls:
+ self._calls[tool_id]._json_buffer += partial_json
+
+ def finalize_all(self) -> None:
+ """Finalize all tool calls: parse accumulated JSON and mark complete."""
+ for info in self._calls.values():
+ if info._json_buffer:
+ try:
+ info.args = json.loads(info._json_buffer)
+ except json.JSONDecodeError:
+ pass
+ info._json_buffer = ""
+ info.args_complete = True
+
+ def is_ready(self, tool_id: str) -> bool:
+ """Check if a tool call is ready to emit (has name and not yet emitted)."""
+ if tool_id not in self._calls:
+ return False
+ info = self._calls[tool_id]
+ return bool(info.name) and not info.emitted
+
+ def get_all(self) -> list[ToolCallInfo]:
+ """Get all tool calls."""
+ return list(self._calls.values())
+
+ def mark_emitted(self, tool_id: str) -> None:
+ """Mark a tool call as emitted."""
+ if tool_id in self._calls:
+ self._calls[tool_id].emitted = True
+
+ def get(self, tool_id: str) -> Optional[ToolCallInfo]:
+ """Get tool call info by ID."""
+ return self._calls.get(tool_id)
+
+ def get_pending(self) -> list[ToolCallInfo]:
+ """Get all unemitted tool calls."""
+ return [info for info in self._calls.values() if not info.emitted]
+
+ def emit_all_pending(self) -> list[ToolCallInfo]:
+ """Emit all pending tool calls and mark them."""
+ pending = self.get_pending()
+ for info in pending:
+ info.emitted = True
+ return pending
+
+ def clear(self) -> None:
+ """Clear all tracked tool calls."""
+ self._calls.clear()
diff --git a/EvoScientist/stream/utils.py b/EvoScientist/stream/utils.py
new file mode 100644
index 0000000..f1bbc84
--- /dev/null
+++ b/EvoScientist/stream/utils.py
@@ -0,0 +1,255 @@
+"""
+Stream utility functions and constants.
+
+Provides tool status indicators, display limits, and formatting helpers
+adapted for deepagents tool names.
+"""
+
+import sys
+from pathlib import PurePath
+from enum import Enum
+
+
+# === Status marker constants ===
+SUCCESS_PREFIX = "[OK]"
+FAILURE_PREFIX = "[FAILED]"
+
+
+# === Tool status indicators ===
+class ToolStatus(str, Enum):
+ """Tool execution status indicators."""
+ RUNNING = "\u25cf" # Running - yellow
+ SUCCESS = "\u25cf" # Success - green
+ ERROR = "\u25cf" # Failed - red
+ PENDING = "\u25cb" # Pending - gray
+
+
+def get_status_symbol(status: ToolStatus) -> str:
+ """Get status symbol with ASCII fallback for terminals without Unicode."""
+ try:
+ supports_unicode = (
+ sys.stdout.encoding
+ and 'utf' in sys.stdout.encoding.lower()
+ )
+ except Exception:
+ supports_unicode = False
+
+ if supports_unicode:
+ return status.value
+
+ fallback = {
+ ToolStatus.RUNNING: "*",
+ ToolStatus.SUCCESS: "+",
+ ToolStatus.ERROR: "x",
+ ToolStatus.PENDING: "-",
+ }
+ return fallback.get(status, "?")
+
+
+# === Display limit constants ===
+class DisplayLimits:
+ """Display length limits."""
+ THINKING_STREAM = 1000
+ THINKING_FINAL = 2000
+ ARGS_INLINE = 100
+ ARGS_FORMATTED = 300
+ TOOL_RESULT_STREAM = 500
+ TOOL_RESULT_FINAL = 800
+ TOOL_RESULT_MAX = 2000
+
+
+def has_args(args) -> bool:
+ """Check if args has content (handles empty dict falsy issue)."""
+ return args is not None and args != {}
+
+
+def is_success(content: str) -> bool:
+ """Determine if tool output indicates successful execution."""
+ content = content.strip()
+ if content.startswith(SUCCESS_PREFIX):
+ return True
+ if content.startswith(FAILURE_PREFIX):
+ return False
+ error_patterns = [
+ 'Traceback (most recent call last)',
+ 'Exception:',
+ 'Error:',
+ ]
+ return not any(pattern in content for pattern in error_patterns)
+
+
+def truncate(content: str, max_length: int, suffix: str = "\n... (truncated)") -> str:
+ """Truncate content to specified length."""
+ if len(content) > max_length:
+ return content[:max_length] + suffix
+ return content
+
+
+# === Compact formatting for deepagents tools ===
+
+def _shorten_path(path: str, max_len: int = 40) -> str:
+ """Shorten a file path for display."""
+ if len(path) <= max_len:
+ return path
+ path_obj = PurePath(path)
+ parts = path_obj.parts
+ if len(parts) > 2:
+ return ".../" + "/".join(parts[-2:])
+ return path
+
+
+def format_tool_compact(name: str, args: dict | None) -> str:
+ """Format as compact tool call string: ToolName(key_arg).
+
+ Adapted for deepagents tool names: execute, read_file, write_file,
+ edit_file, grep, glob, ls, write_todos, read_todos, task, load_skill,
+ tavily_search, think_tool.
+ """
+ if not args:
+ return f"{name}()"
+
+ name_lower = name.lower()
+
+ # Shell execution
+ if name_lower == "execute":
+ cmd = args.get("command", "")
+ if len(cmd) > 50:
+ cmd = cmd[:47] + "..."
+ return f"execute({cmd})"
+
+ # File operations
+ if name_lower == "read_file":
+ path = _shorten_path(args.get("path", ""))
+ return f"read_file({path})"
+
+ if name_lower == "write_file":
+ path = _shorten_path(args.get("path", ""))
+ return f"write_file({path})"
+
+ if name_lower == "edit_file":
+ path = _shorten_path(args.get("path", ""))
+ return f"edit_file({path})"
+
+ # Search operations
+ if name_lower == "glob":
+ pattern = args.get("pattern", "")
+ if len(pattern) > 40:
+ pattern = pattern[:37] + "..."
+ return f"glob({pattern})"
+
+ if name_lower == "grep":
+ pattern = args.get("pattern", "")
+ path = args.get("path", ".")
+ if len(pattern) > 30:
+ pattern = pattern[:27] + "..."
+ return f"grep({pattern}, {path})"
+
+ # Directory listing
+ if name_lower == "ls":
+ path = args.get("path", ".")
+ return f"ls({path})"
+
+ # Todo management
+ if name_lower == "write_todos":
+ todos = args.get("todos", [])
+ if isinstance(todos, list):
+ return f"write_todos({len(todos)} items)"
+ return "write_todos(...)"
+
+ if name_lower == "read_todos":
+ return "read_todos()"
+
+ # Sub-agent delegation — display as "Cooking with {agent}" instead of "task()"
+ if name_lower == "task":
+ sa_type = args.get("subagent_type", "").strip()
+ task_desc = args.get("description", args.get("task", "")).strip()
+ if sa_type:
+ if task_desc:
+ if len(task_desc) > 50:
+ task_desc = task_desc[:47] + "..."
+ return f"Cooking with {sa_type} — {task_desc}"
+ return f"Cooking with {sa_type}"
+ # Fallback if no subagent_type
+ if task_desc:
+ if len(task_desc) > 50:
+ task_desc = task_desc[:47] + "..."
+ return f"Cooking with sub-agent — {task_desc}"
+ return "Cooking with sub-agent"
+
+ # Skills
+ if name_lower == "load_skill":
+ skill_name = args.get("skill_name", args.get("name", ""))
+ return f"load_skill({skill_name})"
+
+ # Web search
+ if name_lower in ("tavily_search", "internet_search"):
+ query = args.get("query", "")
+ if len(query) > 40:
+ query = query[:37] + "..."
+ return f"{name}({query})"
+
+ # Think/reflection
+ if name_lower == "think_tool":
+ reflection = args.get("reflection", "")
+ if len(reflection) > 40:
+ reflection = reflection[:37] + "..."
+ return f"think_tool({reflection})"
+
+ # Default: show first few params
+ params = []
+ for k, v in list(args.items())[:2]:
+ v_str = str(v)
+ if len(v_str) > 20:
+ v_str = v_str[:17] + "..."
+ params.append(f"{k}={v_str}")
+
+ params_str = ", ".join(params)
+ if len(params_str) > 50:
+ params_str = params_str[:47] + "..."
+
+ return f"{name}({params_str})"
+
+
+def format_tree_output(lines: list[str], max_lines: int = 5, indent: str = " ") -> str:
+ """Format output as tree structure.
+
+ Example:
+ └ On branch main
+ Your branch is up to date
+ ... +16 lines
+ """
+ if not lines:
+ return ""
+
+ result = []
+ display_lines = lines[:max_lines]
+
+ for i, line in enumerate(display_lines):
+ prefix = "\u2514" if i == 0 else " "
+ result.append(f"{indent}{prefix} {line}")
+
+ remaining = len(lines) - max_lines
+ if remaining > 0:
+ result.append(f"{indent} ... +{remaining} lines")
+
+ return "\n".join(result)
+
+
+def count_lines(content: str) -> int:
+ """Count number of lines in content."""
+ if not content:
+ return 0
+ return len(content.strip().split("\n"))
+
+
+def truncate_with_line_hint(content: str, max_lines: int = 5) -> tuple[str, int]:
+ """Truncate by line count, returning remaining line count."""
+ lines = content.strip().split("\n")
+ total = len(lines)
+
+ if total <= max_lines:
+ return content.strip(), 0
+
+ truncated = "\n".join(lines[:max_lines])
+ remaining = total - max_lines
+ return truncated, remaining
diff --git a/EvoScientist/subagent.yaml b/EvoScientist/subagent.yaml
new file mode 100644
index 0000000..774a69f
--- /dev/null
+++ b/EvoScientist/subagent.yaml
@@ -0,0 +1,147 @@
+planner-agent:
+ description: "Plan experiments: stages, success signals, and dependencies (no web search, no implementation)."
+ tools: [think_tool]
+ system_prompt: |
+ You are the planner-agent. You do NOT implement code. You create and update experimental plans
+ that are practical to run locally.
+
+ You may be invoked in two modes:
+ 1) PLAN MODE: produce an initial experimental plan.
+ 2) REFLECTION MODE: update the plan based on stage results.
+
+ The caller should start the task with either:
+ - MODE: PLAN
+ - MODE: REFLECTION
+ If MODE is not specified, assume PLAN.
+
+ PLAN MODE output (Markdown):
+ 1) Assumptions & scope
+ 2) Stages (numbered). For each stage include:
+ - goal
+ - success signals (metrics/thresholds or qualitative checks)
+ - what to run (scripts/commands at a high level)
+ - expected artifacts (tables/plots/logs)
+ 3) Dependencies (data, compute, environment)
+ 4) Iteration triggers (when to change dataset/model/objective)
+ 5) Evaluation protocol (splits, primary metrics, baselines) and data quality checks
+ 6) Environment preflight (GPU/CUDA/VRAM/disk) and required dependencies (pip packages)
+
+ REFLECTION MODE output (JSON only, no extra text):
+ {
+ "completed": ["..."],
+ "unmet_success_signals": ["..."],
+ "skill_suggestions": ["..."],
+ "stage_modifications": [
+ {"stage": "Stage name or index", "change": "What to adjust and why"}
+ ],
+ "new_stages": [
+ {
+ "title": "...",
+ "goal": "...",
+ "success_signals": ["..."],
+ "what_to_run": ["..."],
+ "expected_artifacts": ["..."]
+ }
+ ],
+ "todo_updates": ["..."]
+ }
+
+ Empty arrays are valid. If no changes are needed, return the JSON with empty arrays.
+ "skill_suggestions" must contain skill ids from SKILL.md frontmatter ("name:").
+
+ Keep the structure flexible (not rigid templates). If model size is unspecified, default to
+ <=7B-class models and lightweight baselines.
+
+research-agent:
+ description: "Web research for methods/baselines/datasets (one topic at a time, return actionable notes + sources)."
+ tools: [tavily_search, think_tool]
+ system_prompt_ref: RESEARCHER_INSTRUCTIONS
+
+code-agent:
+ description: "Implement experiment code and runnable scripts; keep changes minimal and reproducible."
+ tools: [think_tool]
+ system_prompt: |
+ You are the code-agent. Implement experiment code in the workspace and keep changes minimal,
+ reproducible, and easy to run.
+
+ Guidelines:
+ - Prefer small scripts and clear entry points.
+ - Record exact commands to run and where outputs are written.
+ - Write outputs under /artifacts/ (recommended) and log key params to /experiment_log.md (optional).
+ - Do not modify /skills/.
+ - If a relevant local skill exists, load it (load_skill) and follow it instead of reinventing.
+ - Before heavy runs, confirm GPU/CUDA/VRAM availability and required packages.
+ - Suggested preflight commands:
+ - nvidia-smi
+ - python -c "import torch; print(torch.cuda.is_available(), torch.version.cuda, torch.cuda.get_device_name(0))"
+
+ When responding, include:
+ - Files changed
+ - Commands to run
+ - Output paths
+ - Any remaining issues/next steps
+
+debug-agent:
+ description: "Debug runtime failures and fix bugs with minimal, verifiable patches."
+ tools: [think_tool]
+ system_prompt: |
+ You are the debug-agent. Reproduce failures, identify root causes, apply minimal fixes, and provide
+ concise diagnostics.
+
+ Guidelines:
+ - Prefer small, safe changes.
+ - Explain the root cause in one paragraph.
+ - Provide how to reproduce and how to verify the fix.
+ - Do not modify /skills/.
+ - If a relevant local skill exists, load it (load_skill) and use it as a checklist.
+
+ When responding, include:
+ - Root cause
+ - Fix summary (files/changes)
+ - Repro steps
+ - Verification steps
+
+data-analysis-agent:
+ description: "Analyze experiment outputs: compute metrics, make plots, summarize insights."
+ tools: [think_tool]
+ system_prompt: |
+ You are the data-analysis-agent. Analyze experiment outputs, compute metrics, and create
+ publication-friendly plots.
+
+ Guidelines:
+ - Do not invent numbers; compute from files or state what is missing.
+ - Save figures/tables under /artifacts/ (recommended) and reference paths.
+ - Summarize insights and provide 1-3 recommended next experiments.
+ - If a relevant local skill exists (evaluation, logging, plotting), load it (load_skill) and follow it.
+ - Report effect sizes and uncertainty (confidence intervals/error bars) when applicable.
+ - Apply multiple-testing corrections when comparing many conditions.
+ - Distinguish exploratory vs confirmatory findings.
+
+ When responding, include:
+ - Metrics computed (with definitions)
+ - Figures/tables produced (paths)
+ - Interpretation and next steps
+
+writing-agent:
+ description: "Draft a paper-ready Markdown experiment report (no fabricated results/citations)."
+ tools: [think_tool]
+ system_prompt: |
+ You are the writing-agent. Draft a clear Markdown experimental report suitable for later paper writing.
+
+ Guidelines:
+ - Use the experiment plan, logs, and artifacts. Reference file paths for figures/tables.
+ - Do not fabricate results or citations.
+ - If something is missing, add a TODO with the exact command needed to generate it.
+ - If a relevant local skill exists (e.g., evaluation/reporting conventions), load it (load_skill) and apply it.
+ - Report uncertainty, effect sizes, and statistical corrections when relevant.
+ - Include negative results and clear limitations.
+ - Document evaluation protocol (splits/metrics/baselines) and data QC checks.
+
+ Preferred sections:
+ 1) Summary & goals
+ 2) Experiment plan (stages + success signals)
+ 3) Setup (data, model, environment, parameters)
+ 4) Baselines and comparisons
+ 5) Results (with artifact paths)
+ 6) Analysis, limitations, and next steps
+ 7) Sources (only if web research was used)
diff --git a/EvoScientist/tools.py b/EvoScientist/tools.py
new file mode 100644
index 0000000..cc50333
--- /dev/null
+++ b/EvoScientist/tools.py
@@ -0,0 +1,135 @@
+"""Tools.
+
+This module provides search and reflection tools for the research agent,
+using Tavily for URL discovery and fetching full webpage content.
+"""
+
+import asyncio
+from typing import Literal
+
+import httpx
+from dotenv import load_dotenv
+from langchain_core.tools import InjectedToolArg, tool
+from markdownify import markdownify
+from tavily import TavilyClient
+from typing_extensions import Annotated
+
+load_dotenv(override=True)
+
+tavily_client = TavilyClient()
+
+
+async def fetch_webpage_content(url: str, timeout: float = 10.0) -> str:
+ """Fetch and convert webpage content to markdown.
+
+ Args:
+ url: URL to fetch
+ timeout: Request timeout in seconds
+
+ Returns:
+ Webpage content as markdown
+ """
+ headers = {
+ "User-Agent": (
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
+ "AppleWebKit/537.36 (KHTML, like Gecko) "
+ "Chrome/91.0.4472.124 Safari/537.36"
+ )
+ }
+
+ try:
+ async with httpx.AsyncClient() as client:
+ response = await client.get(url, headers=headers, timeout=timeout)
+ response.raise_for_status()
+ return markdownify(response.text)
+ except Exception as e:
+ return f"Error fetching content from {url}: {str(e)}"
+
+
+@tool(parse_docstring=True)
+async def tavily_search(
+ query: str,
+ max_results: Annotated[int, InjectedToolArg] = 3,
+ topic: Annotated[
+ Literal["general", "news", "finance"], InjectedToolArg
+ ] = "general",
+) -> str:
+ """Search the web for information on a given query.
+
+ Uses Tavily to discover relevant URLs, then fetches and returns
+ full webpage content as markdown for comprehensive research.
+
+ Args:
+ query: Search query to execute
+
+ Returns:
+ Formatted search results with full webpage content in markdown
+ """
+
+ def _sync_search() -> dict:
+ return tavily_client.search(
+ query,
+ max_results=max_results,
+ topic=topic,
+ )
+
+ try:
+ # Run Tavily search asynchronously
+ search_results = await asyncio.to_thread(_sync_search)
+
+ # Fetch full content for each URL concurrently
+ results = search_results.get("results", [])
+ if not results:
+ return f"No results found for '{query}'"
+
+ # Fetch all webpages concurrently
+ fetch_tasks = [fetch_webpage_content(r["url"]) for r in results]
+ contents = await asyncio.gather(*fetch_tasks)
+
+ # Format results
+ result_texts = []
+ for result, content in zip(results, contents):
+ result_text = f"""## {result["title"]}
+**URL:** {result["url"]}
+
+{content}
+
+---
+"""
+ result_texts.append(result_text)
+
+ return f"""Found {len(result_texts)} result(s) for '{query}':
+
+{"".join(result_texts)}"""
+
+ except Exception as e:
+ return f"Search failed: {str(e)}"
+
+
+@tool(parse_docstring=True)
+def think_tool(reflection: str) -> str:
+ """Tool for strategic reflection on research progress and decision-making.
+
+ Use this tool after each search to analyze results and plan next steps systematically.
+ This creates a deliberate pause in the research workflow for quality decision-making.
+
+ When to use:
+ - After receiving search results: What key information did I find?
+ - Before deciding next steps: Do I have enough to answer comprehensively?
+ - When assessing research gaps: What specific information am I still missing?
+ - Before concluding research: Can I provide a complete answer now?
+
+ Reflection should address:
+ 1. Analysis of current findings - What concrete information have I gathered?
+ 2. Gap assessment - What crucial information is still missing?
+ 3. Quality evaluation - Do I have sufficient evidence/examples for a good answer?
+ 4. Strategic decision - Should I continue searching or provide my answer?
+ 5. Skill leverage - Is there a relevant local skill to load that can accelerate this work?
+
+ Args:
+ reflection: Your detailed reflection on research progress, findings, gaps, and next steps
+
+ Returns:
+ Confirmation that reflection was recorded for decision-making
+ """
+ return f"Reflection recorded: {reflection}"
diff --git a/EvoScientist/utils.py b/EvoScientist/utils.py
new file mode 100644
index 0000000..55319b4
--- /dev/null
+++ b/EvoScientist/utils.py
@@ -0,0 +1,207 @@
+"""Utility functions for EvoScientist.
+
+This module primarily contains helpers for displaying messages and prompts in
+notebooks, and lightweight configuration loaders used by the agent runtime.
+"""
+
+import json
+from pathlib import Path
+from typing import Any
+
+import yaml
+from rich.console import Console
+from rich.panel import Panel
+from rich.text import Text
+
+console = Console()
+
+
+def format_message_content(message):
+ """Convert message content to displayable string."""
+ parts = []
+ tool_calls_processed = False
+
+ # Handle main content
+ if isinstance(message.content, str):
+ parts.append(message.content)
+ elif isinstance(message.content, list):
+ # Handle complex content like tool calls (Anthropic format)
+ for item in message.content:
+ if item.get("type") == "text":
+ parts.append(item["text"])
+ elif item.get("type") == "tool_use":
+ parts.append(f"\n🔧 Tool Call: {item['name']}")
+ parts.append(f" Args: {json.dumps(item['input'], indent=2)}")
+ parts.append(f" ID: {item.get('id', 'N/A')}")
+ tool_calls_processed = True
+ else:
+ parts.append(str(message.content))
+
+ # Handle tool calls attached to the message (OpenAI format) - only if not already processed
+ if (
+ not tool_calls_processed
+ and hasattr(message, "tool_calls")
+ and message.tool_calls
+ ):
+ for tool_call in message.tool_calls:
+ parts.append(f"\n🔧 Tool Call: {tool_call['name']}")
+ parts.append(f" Args: {json.dumps(tool_call['args'], indent=2)}")
+ parts.append(f" ID: {tool_call['id']}")
+
+ return "\n".join(parts)
+
+
+def format_messages(messages):
+ """Format and display a list of messages with Rich formatting."""
+ for m in messages:
+ msg_type = m.__class__.__name__.replace("Message", "")
+ content = format_message_content(m)
+
+ if msg_type == "Human":
+ console.print(Panel(content, title="🧑 Human", border_style="blue"))
+ elif msg_type == "Ai":
+ console.print(Panel(content, title="🤖 Assistant", border_style="green"))
+ elif msg_type == "Tool":
+ console.print(Panel(content, title="🔧 Tool Output", border_style="yellow"))
+ else:
+ console.print(Panel(content, title=f"📝 {msg_type}", border_style="white"))
+
+
+def format_message(messages):
+ """Alias for format_messages for backward compatibility."""
+ return format_messages(messages)
+
+
+def show_prompt(prompt_text: str, title: str = "Prompt", border_style: str = "blue"):
+ """Display a prompt with rich formatting and XML tag highlighting.
+
+ Args:
+ prompt_text: The prompt string to display
+ title: Title for the panel (default: "Prompt")
+ border_style: Border color style (default: "blue")
+ """
+ # Create a formatted display of the prompt
+ formatted_text = Text(prompt_text)
+ formatted_text.highlight_regex(r"<[^>]+>", style="bold blue") # Highlight XML tags
+ formatted_text.highlight_regex(
+ r"##[^#\n]+", style="bold magenta"
+ ) # Highlight headers
+ formatted_text.highlight_regex(
+ r"###[^#\n]+", style="bold cyan"
+ ) # Highlight sub-headers
+
+ # Display in a panel for better presentation
+ console.print(
+ Panel(
+ formatted_text,
+ title=f"[bold green]{title}[/bold green]",
+ border_style=border_style,
+ padding=(1, 2),
+ )
+ )
+
+
+def load_subagents(
+ config_path: Path,
+ *,
+ tool_registry: dict[str, Any],
+ prompt_refs: dict[str, str] | None = None,
+) -> list[dict[str, Any]]:
+ """Load subagent definitions from YAML and wire up tools.
+
+ NOTE: This is a custom utility. deepagents does not natively load subagents
+ from files - they're normally defined inline in the create_deep_agent() call.
+ We externalize to YAML here to keep configuration separate from code.
+
+ Supported YAML schemas:
+
+ 1) Mapping style (recommended):
+ planner-agent:
+ description: "..."
+ tools: [think_tool]
+ system_prompt: |
+ ...
+ research-agent:
+ description: "..."
+ tools: [tavily_search, think_tool]
+ system_prompt_ref: RESEARCHER_INSTRUCTIONS
+
+ 2) List style (legacy):
+ subagents:
+ - name: planner-agent
+ description: "..."
+ tools: [think_tool]
+ system_prompt: |
+ ...
+ """
+ prompt_refs = prompt_refs or {}
+
+ with config_path.open(encoding="utf-8") as f:
+ config = yaml.safe_load(f) or {}
+
+ if not isinstance(config, dict) or not config:
+ raise ValueError("subagent.yaml must be a mapping or contain 'subagents:'")
+
+ subagents: list[dict[str, Any]] = []
+
+ def _build_one(name: str, spec: dict[str, Any]) -> dict[str, Any]:
+ subagent: dict[str, Any] = {
+ "name": name,
+ "description": spec.get("description", ""),
+ }
+
+ if "system_prompt_ref" in spec:
+ ref = spec["system_prompt_ref"]
+ if ref not in prompt_refs:
+ raise ValueError(f"Unknown system_prompt_ref '{ref}' for subagent '{name}'")
+ subagent["system_prompt"] = prompt_refs[ref]
+ else:
+ subagent["system_prompt"] = spec.get("system_prompt", "")
+
+ if "model" in spec:
+ subagent["model"] = spec["model"]
+
+ if "tools" in spec:
+ subagent["tools"] = [tool_registry[t] for t in spec["tools"]]
+
+ return subagent
+
+ # Legacy list style
+ if "subagents" in config:
+ items = config.get("subagents")
+ if not isinstance(items, list) or not items:
+ raise ValueError("subagent.yaml must contain a non-empty 'subagents:' list")
+ for item in items:
+ if not isinstance(item, dict):
+ continue
+ name = item.get("name")
+ if not name:
+ raise ValueError("Each subagent entry must have a 'name'")
+ subagents.append(_build_one(name, item))
+ return subagents
+
+ # Mapping style: {: }
+ for name, spec in config.items():
+ if not isinstance(spec, dict):
+ continue
+ subagents.append(_build_one(name, spec))
+
+ return subagents
+
+
+def load_subagent(
+ config_path: Path,
+ name: str,
+ *,
+ tool_registry: dict[str, Any],
+ prompt_refs: dict[str, str] | None = None,
+) -> dict[str, Any]:
+ """Load a single sub-agent by name from YAML."""
+ for agent in load_subagents(
+ config_path,
+ tool_registry=tool_registry,
+ prompt_refs=prompt_refs,
+ ):
+ if agent.get("name") == name:
+ return agent
+ raise KeyError(f"Sub-agent not found: {name}")
diff --git a/README.md b/README.md
index 06d8b76..35f55b3 100644
--- a/README.md
+++ b/README.md
@@ -42,6 +42,18 @@
> [!TIP]
> Use [`uv`](https://pypi.org/project/uv) for installation — it's faster and more reliable than `pip`.
+### For Development
+
+```Shell
+# Create and activate a conda environment
+conda create -n EvoSci python=3.11 -y
+conda activate EvoSci
+
+# Install in development (editable) mode
+pip install EvoScientist
+# or
+pip install -e .
+```
### Option 1:
Install the latest version directly from GitHub for quick setup:
@@ -66,14 +78,35 @@ uv pip install -e .
### CLI Inference
You can perform inference directly from the command line using our CLI tool:
-> TODO
+
+```Shell
+python -m EvoScientist
+```
**Optional arguments:**
> TODO
### Script Inference
-> TODO
+```python
+from EvoScientist import agent
+from langchain_core.messages import HumanMessage
+from EvoScientist.utils import format_messages
+thread = {"configurable": {"thread_id": "1"}}
+question = "Hi?"
+last_len = 0
+
+for state in agent.stream(
+ {"messages": [HumanMessage(content=question)]},
+ config=thread,
+ stream_mode="values",
+):
+ msgs = state["messages"]
+ if len(msgs) > last_len:
+ format_messages(msgs[last_len:])
+ last_len = len(msgs)
+```
+
### Web Interface
diff --git a/assets/EvoScientist_cli.png b/assets/EvoScientist_cli.png
new file mode 100644
index 0000000..3f08f5c
Binary files /dev/null and b/assets/EvoScientist_cli.png differ
diff --git a/langgraph.json b/langgraph.json
new file mode 100644
index 0000000..76a5a92
--- /dev/null
+++ b/langgraph.json
@@ -0,0 +1,10 @@
+{
+ "dependencies": ["."],
+ "graphs": {
+ "EvoScientist": "./langgraph_dev.py:EvoScientist_agent"
+ },
+ "env": ".env",
+ "config": {
+ "recursion_limit": 500
+ }
+}
\ No newline at end of file
diff --git a/langgraph_dev.py b/langgraph_dev.py
new file mode 100644
index 0000000..d07fb3c
--- /dev/null
+++ b/langgraph_dev.py
@@ -0,0 +1,7 @@
+"""Entry point for ``langgraph dev``.
+
+langgraph.json references ``./langgraph_dev.py:EvoScientist_agent``.
+Actual construction lives in ``EvoScientist/EvoScientist.py``.
+"""
+
+from EvoScientist.EvoScientist import EvoScientist_agent, create_cli_agent # noqa: F401
diff --git a/pyproject.toml b/pyproject.toml
new file mode 100644
index 0000000..a1857f1
--- /dev/null
+++ b/pyproject.toml
@@ -0,0 +1,52 @@
+[project]
+name = "EvoScientist"
+version = "0.1.0"
+description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery"
+readme = "README.md"
+requires-python = ">=3.11"
+license = "MIT"
+authors = [
+ { name = "Xi Zhang" },
+]
+maintainers = [
+ { name = "Xi Zhang" },
+]
+keywords = ["ai-scientific", "scientific-discovery", "self-evolving", "ai-scientists", "end-to-end"]
+classifiers = [
+ "Programming Language :: Python :: 3",
+]
+dependencies = [
+ "deepagents>=0.3.6",
+ "langchain>=1.2",
+ "langchain-anthropic>=1.3",
+ "anthropic>=0.76",
+ "tavily-python>=0.7",
+ "pyyaml>=6.0",
+ "rich>=14.0",
+ "prompt-toolkit>=3.0",
+ "python-dotenv>=1.0",
+ "langgraph-cli[inmem]>=0.4",
+ "httpx>=0.27",
+ "markdownify>=0.14",
+]
+
+[project.optional-dependencies]
+dev = ["pytest>=8.0", "pytest-cov>=5.0"]
+
+[project.urls]
+"Homepage" = "https://github.com/EvoScientist/EvoScientist"
+"Bug Tracker" = "https://github.com/EvoScientist/EvoScientist/issues"
+"Documentation" = "https://github.com/EvoScientist/EvoScientist#readme"
+
+[project.scripts]
+evoscientist = "EvoScientist.cli:main"
+
+[build-system]
+requires = ["setuptools>=68.0"]
+build-backend = "setuptools.build_meta"
+
+[tool.setuptools.packages.find]
+include = ["EvoScientist*"]
+
+[tool.setuptools.package-data]
+EvoScientist = ["subagent.yaml"]
diff --git a/skills/accelerate/SKILL.md b/skills/accelerate/SKILL.md
new file mode 100644
index 0000000..27e4952
--- /dev/null
+++ b/skills/accelerate/SKILL.md
@@ -0,0 +1,332 @@
+---
+name: accelerate
+description: Simplest distributed training API. 4 lines to add distributed support to any PyTorch script. Unified API for DeepSpeed/FSDP/Megatron/DDP. Automatic device placement, mixed precision (FP16/BF16/FP8). Interactive config, single launch command. HuggingFace ecosystem standard.
+version: 1.0.0
+author: Orchestra Research
+license: MIT
+tags: [Distributed Training, HuggingFace, Accelerate, DeepSpeed, FSDP, Mixed Precision, PyTorch, DDP, Unified API, Simple]
+dependencies: [accelerate, torch, transformers]
+---
+
+# HuggingFace Accelerate - Unified Distributed Training
+
+## Quick start
+
+Accelerate simplifies distributed training to 4 lines of code.
+
+**Installation**:
+```bash
+pip install accelerate
+```
+
+**Convert PyTorch script** (4 lines):
+```python
+import torch
++ from accelerate import Accelerator
+
++ accelerator = Accelerator()
+
+ model = torch.nn.Transformer()
+ optimizer = torch.optim.Adam(model.parameters())
+ dataloader = torch.utils.data.DataLoader(dataset)
+
++ model, optimizer, dataloader = accelerator.prepare(model, optimizer, dataloader)
+
+ for batch in dataloader:
+ optimizer.zero_grad()
+ loss = model(batch)
+- loss.backward()
++ accelerator.backward(loss)
+ optimizer.step()
+```
+
+**Run** (single command):
+```bash
+accelerate launch train.py
+```
+
+## Common workflows
+
+### Workflow 1: From single GPU to multi-GPU
+
+**Original script**:
+```python
+# train.py
+import torch
+
+model = torch.nn.Linear(10, 2).to('cuda')
+optimizer = torch.optim.Adam(model.parameters())
+dataloader = torch.utils.data.DataLoader(dataset, batch_size=32)
+
+for epoch in range(10):
+ for batch in dataloader:
+ batch = batch.to('cuda')
+ optimizer.zero_grad()
+ loss = model(batch).mean()
+ loss.backward()
+ optimizer.step()
+```
+
+**With Accelerate** (4 lines added):
+```python
+# train.py
+import torch
+from accelerate import Accelerator # +1
+
+accelerator = Accelerator() # +2
+
+model = torch.nn.Linear(10, 2)
+optimizer = torch.optim.Adam(model.parameters())
+dataloader = torch.utils.data.DataLoader(dataset, batch_size=32)
+
+model, optimizer, dataloader = accelerator.prepare(model, optimizer, dataloader) # +3
+
+for epoch in range(10):
+ for batch in dataloader:
+ # No .to('cuda') needed - automatic!
+ optimizer.zero_grad()
+ loss = model(batch).mean()
+ accelerator.backward(loss) # +4
+ optimizer.step()
+```
+
+**Configure** (interactive):
+```bash
+accelerate config
+```
+
+**Questions**:
+- Which machine? (single/multi GPU/TPU/CPU)
+- How many machines? (1)
+- Mixed precision? (no/fp16/bf16/fp8)
+- DeepSpeed? (no/yes)
+
+**Launch** (works on any setup):
+```bash
+# Single GPU
+accelerate launch train.py
+
+# Multi-GPU (8 GPUs)
+accelerate launch --multi_gpu --num_processes 8 train.py
+
+# Multi-node
+accelerate launch --multi_gpu --num_processes 16 \
+ --num_machines 2 --machine_rank 0 \
+ --main_process_ip $MASTER_ADDR \
+ train.py
+```
+
+### Workflow 2: Mixed precision training
+
+**Enable FP16/BF16**:
+```python
+from accelerate import Accelerator
+
+# FP16 (with gradient scaling)
+accelerator = Accelerator(mixed_precision='fp16')
+
+# BF16 (no scaling, more stable)
+accelerator = Accelerator(mixed_precision='bf16')
+
+# FP8 (H100+)
+accelerator = Accelerator(mixed_precision='fp8')
+
+model, optimizer, dataloader = accelerator.prepare(model, optimizer, dataloader)
+
+# Everything else is automatic!
+for batch in dataloader:
+ with accelerator.autocast(): # Optional, done automatically
+ loss = model(batch)
+ accelerator.backward(loss)
+```
+
+### Workflow 3: DeepSpeed ZeRO integration
+
+**Enable DeepSpeed ZeRO-2**:
+```python
+from accelerate import Accelerator
+
+accelerator = Accelerator(
+ mixed_precision='bf16',
+ deepspeed_plugin={
+ "zero_stage": 2, # ZeRO-2
+ "offload_optimizer": False,
+ "gradient_accumulation_steps": 4
+ }
+)
+
+# Same code as before!
+model, optimizer, dataloader = accelerator.prepare(model, optimizer, dataloader)
+```
+
+**Or via config**:
+```bash
+accelerate config
+# Select: DeepSpeed → ZeRO-2
+```
+
+**deepspeed_config.json**:
+```json
+{
+ "fp16": {"enabled": false},
+ "bf16": {"enabled": true},
+ "zero_optimization": {
+ "stage": 2,
+ "offload_optimizer": {"device": "cpu"},
+ "allgather_bucket_size": 5e8,
+ "reduce_bucket_size": 5e8
+ }
+}
+```
+
+**Launch**:
+```bash
+accelerate launch --config_file deepspeed_config.json train.py
+```
+
+### Workflow 4: FSDP (Fully Sharded Data Parallel)
+
+**Enable FSDP**:
+```python
+from accelerate import Accelerator, FullyShardedDataParallelPlugin
+
+fsdp_plugin = FullyShardedDataParallelPlugin(
+ sharding_strategy="FULL_SHARD", # ZeRO-3 equivalent
+ auto_wrap_policy="TRANSFORMER_AUTO_WRAP",
+ cpu_offload=False
+)
+
+accelerator = Accelerator(
+ mixed_precision='bf16',
+ fsdp_plugin=fsdp_plugin
+)
+
+model, optimizer, dataloader = accelerator.prepare(model, optimizer, dataloader)
+```
+
+**Or via config**:
+```bash
+accelerate config
+# Select: FSDP → Full Shard → No CPU Offload
+```
+
+### Workflow 5: Gradient accumulation
+
+**Accumulate gradients**:
+```python
+from accelerate import Accelerator
+
+accelerator = Accelerator(gradient_accumulation_steps=4)
+
+model, optimizer, dataloader = accelerator.prepare(model, optimizer, dataloader)
+
+for batch in dataloader:
+ with accelerator.accumulate(model): # Handles accumulation
+ optimizer.zero_grad()
+ loss = model(batch)
+ accelerator.backward(loss)
+ optimizer.step()
+```
+
+**Effective batch size**: `batch_size * num_gpus * gradient_accumulation_steps`
+
+## When to use vs alternatives
+
+**Use Accelerate when**:
+- Want simplest distributed training
+- Need single script for any hardware
+- Use HuggingFace ecosystem
+- Want flexibility (DDP/DeepSpeed/FSDP/Megatron)
+- Need quick prototyping
+
+**Key advantages**:
+- **4 lines**: Minimal code changes
+- **Unified API**: Same code for DDP, DeepSpeed, FSDP, Megatron
+- **Automatic**: Device placement, mixed precision, sharding
+- **Interactive config**: No manual launcher setup
+- **Single launch**: Works everywhere
+
+**Use alternatives instead**:
+- **PyTorch Lightning**: Need callbacks, high-level abstractions
+- **Ray Train**: Multi-node orchestration, hyperparameter tuning
+- **DeepSpeed**: Direct API control, advanced features
+- **Raw DDP**: Maximum control, minimal abstraction
+
+## Common issues
+
+**Issue: Wrong device placement**
+
+Don't manually move to device:
+```python
+# WRONG
+batch = batch.to('cuda')
+
+# CORRECT
+# Accelerate handles it automatically after prepare()
+```
+
+**Issue: Gradient accumulation not working**
+
+Use context manager:
+```python
+# CORRECT
+with accelerator.accumulate(model):
+ optimizer.zero_grad()
+ accelerator.backward(loss)
+ optimizer.step()
+```
+
+**Issue: Checkpointing in distributed**
+
+Use accelerator methods:
+```python
+# Save only on main process
+if accelerator.is_main_process:
+ accelerator.save_state('checkpoint/')
+
+# Load on all processes
+accelerator.load_state('checkpoint/')
+```
+
+**Issue: Different results with FSDP**
+
+Ensure same random seed:
+```python
+from accelerate.utils import set_seed
+set_seed(42)
+```
+
+## Advanced topics
+
+**Megatron integration**: See [references/megatron-integration.md](references/megatron-integration.md) for tensor parallelism, pipeline parallelism, and sequence parallelism setup.
+
+**Custom plugins**: See [references/custom-plugins.md](references/custom-plugins.md) for creating custom distributed plugins and advanced configuration.
+
+**Performance tuning**: See [references/performance.md](references/performance.md) for profiling, memory optimization, and best practices.
+
+## Hardware requirements
+
+- **CPU**: Works (slow)
+- **Single GPU**: Works
+- **Multi-GPU**: DDP (default), DeepSpeed, or FSDP
+- **Multi-node**: DDP, DeepSpeed, FSDP, Megatron
+- **TPU**: Supported
+- **Apple MPS**: Supported
+
+**Launcher requirements**:
+- **DDP**: `torch.distributed.run` (built-in)
+- **DeepSpeed**: `deepspeed` (pip install deepspeed)
+- **FSDP**: PyTorch 1.12+ (built-in)
+- **Megatron**: Custom setup
+
+## Resources
+
+- Docs: https://huggingface.co/docs/accelerate
+- GitHub: https://github.com/huggingface/accelerate
+- Version: 1.11.0+
+- Tutorial: "Accelerate your scripts"
+- Examples: https://github.com/huggingface/accelerate/tree/main/examples
+- Used by: HuggingFace Transformers, TRL, PEFT, all HF libraries
+
+
+
diff --git a/skills/accelerate/references/custom-plugins.md b/skills/accelerate/references/custom-plugins.md
new file mode 100644
index 0000000..d8207ee
--- /dev/null
+++ b/skills/accelerate/references/custom-plugins.md
@@ -0,0 +1,453 @@
+# Custom Plugins for Accelerate
+
+## Overview
+
+Accelerate allows creating **custom plugins** to extend distributed training strategies beyond built-in options (DDP, FSDP, DeepSpeed).
+
+## Plugin Architecture
+
+### Base Plugin Structure
+
+```python
+from accelerate.utils import DistributedDataParallelKwargs
+from dataclasses import dataclass
+
+@dataclass
+class CustomPlugin:
+ """Custom training plugin."""
+
+ # Plugin configuration
+ param1: int = 1
+ param2: str = "default"
+
+ def __post_init__(self):
+ # Validation logic
+ if self.param1 < 1:
+ raise ValueError("param1 must be >= 1")
+```
+
+### Using Custom Plugin
+
+```python
+from accelerate import Accelerator
+
+# Create plugin
+custom_plugin = CustomPlugin(param1=4, param2="value")
+
+# Pass to Accelerator
+accelerator = Accelerator(
+ custom_plugin=custom_plugin # Not a real parameter, example only
+)
+```
+
+## Built-In Plugin Examples
+
+### 1. GradScalerKwargs (FP16 Configuration)
+
+```python
+from accelerate.utils import GradScalerKwargs
+
+# Configure gradient scaler for FP16
+scaler_kwargs = GradScalerKwargs(
+ init_scale=2.**16, # Initial loss scale
+ growth_factor=2.0, # Scale growth rate
+ backoff_factor=0.5, # Scale backoff rate
+ growth_interval=2000, # Steps between scale increases
+ enabled=True # Enable scaler
+)
+
+accelerator = Accelerator(
+ mixed_precision='fp16',
+ kwargs_handlers=[scaler_kwargs] # Pass as kwargs handler
+)
+```
+
+**Use case**: Fine-tune FP16 gradient scaling behavior
+
+### 2. DistributedDataParallelKwargs
+
+```python
+from accelerate.utils import DistributedDataParallelKwargs
+
+# Configure DDP behavior
+ddp_kwargs = DistributedDataParallelKwargs(
+ bucket_cap_mb=25, # Gradient bucketing size
+ find_unused_parameters=False, # Find unused params (slower)
+ check_reduction=False, # Check gradient reduction
+ gradient_as_bucket_view=True, # Memory optimization
+ static_graph=False # Static computation graph
+)
+
+accelerator = Accelerator(
+ kwargs_handlers=[ddp_kwargs]
+)
+```
+
+**Use case**: Optimize DDP performance for specific models
+
+### 3. FP8RecipeKwargs (H100 FP8)
+
+```python
+from accelerate.utils import FP8RecipeKwargs
+
+# Configure FP8 training (H100)
+fp8_recipe = FP8RecipeKwargs(
+ backend="te", # TransformerEngine backend
+ margin=0, # Scaling margin
+ interval=1, # Scaling interval
+ fp8_format="HYBRID", # E4M3 + E5M2 hybrid
+ amax_history_len=1024, # AMAX history length
+ amax_compute_algo="max" # AMAX computation algorithm
+)
+
+accelerator = Accelerator(
+ mixed_precision='fp8',
+ kwargs_handlers=[fp8_recipe]
+)
+```
+
+**Use case**: Ultra-fast training on H100 GPUs
+
+## Custom DeepSpeed Configuration
+
+### ZeRO-3 with CPU Offload
+
+```python
+from accelerate import Accelerator
+from accelerate.utils import DeepSpeedPlugin
+
+# Custom DeepSpeed config
+ds_plugin = DeepSpeedPlugin(
+ zero_stage=3, # ZeRO-3
+ offload_optimizer_device="cpu", # CPU offload optimizer
+ offload_param_device="cpu", # CPU offload parameters
+ zero3_init_flag=True, # ZeRO-3 initialization
+ zero3_save_16bit_model=True, # Save FP16 weights
+)
+
+accelerator = Accelerator(
+ deepspeed_plugin=ds_plugin,
+ mixed_precision='bf16'
+)
+```
+
+### ZeRO-2 with NVMe Offload
+
+```python
+ds_plugin = DeepSpeedPlugin(
+ zero_stage=2,
+ offload_optimizer_device="nvme", # NVMe offload
+ offload_param_device="nvme",
+ nvme_path="/local_nvme", # NVMe mount path
+)
+```
+
+### Custom JSON Config
+
+```python
+import json
+
+# Load custom DeepSpeed config
+with open('deepspeed_config.json', 'r') as f:
+ ds_config = json.load(f)
+
+ds_plugin = DeepSpeedPlugin(hf_ds_config=ds_config)
+
+accelerator = Accelerator(deepspeed_plugin=ds_plugin)
+```
+
+**Example config** (`deepspeed_config.json`):
+```json
+{
+ "train_batch_size": "auto",
+ "train_micro_batch_size_per_gpu": "auto",
+ "gradient_accumulation_steps": "auto",
+ "gradient_clipping": 1.0,
+ "zero_optimization": {
+ "stage": 3,
+ "offload_optimizer": {
+ "device": "cpu",
+ "pin_memory": true
+ },
+ "offload_param": {
+ "device": "cpu",
+ "pin_memory": true
+ },
+ "overlap_comm": true,
+ "contiguous_gradients": true,
+ "sub_group_size": 1e9,
+ "reduce_bucket_size": 5e8,
+ "stage3_prefetch_bucket_size": 5e8,
+ "stage3_param_persistence_threshold": 1e6,
+ "stage3_max_live_parameters": 1e9,
+ "stage3_max_reuse_distance": 1e9,
+ "stage3_gather_16bit_weights_on_model_save": true
+ },
+ "bf16": {
+ "enabled": true
+ },
+ "steps_per_print": 100,
+ "wall_clock_breakdown": false
+}
+```
+
+## Custom FSDP Configuration
+
+### FSDP with Custom Auto-Wrap Policy
+
+```python
+from accelerate.utils import FullyShardedDataParallelPlugin
+from torch.distributed.fsdp import BackwardPrefetch, ShardingStrategy
+from torch.distributed.fsdp.wrap import size_based_auto_wrap_policy
+import functools
+
+# Custom wrap policy (size-based)
+wrap_policy = functools.partial(
+ size_based_auto_wrap_policy,
+ min_num_params=1e6 # Wrap layers with 1M+ params
+)
+
+fsdp_plugin = FullyShardedDataParallelPlugin(
+ sharding_strategy=ShardingStrategy.FULL_SHARD, # ZeRO-3 equivalent
+ backward_prefetch=BackwardPrefetch.BACKWARD_PRE, # Prefetch strategy
+ mixed_precision_policy=None, # Use Accelerator's mixed precision
+ auto_wrap_policy=wrap_policy, # Custom wrapping
+ cpu_offload=False,
+ ignored_modules=None, # Modules to not wrap
+ state_dict_type="FULL_STATE_DICT", # Save format
+ optim_state_dict_config=None,
+ limit_all_gathers=False,
+ use_orig_params=True, # Use original param shapes
+)
+
+accelerator = Accelerator(
+ fsdp_plugin=fsdp_plugin,
+ mixed_precision='bf16'
+)
+```
+
+### FSDP with Transformer Auto-Wrap
+
+```python
+from torch.distributed.fsdp.wrap import transformer_auto_wrap_policy
+from transformers.models.gpt2.modeling_gpt2 import GPT2Block
+
+# Wrap at transformer block level
+wrap_policy = functools.partial(
+ transformer_auto_wrap_policy,
+ transformer_layer_cls={GPT2Block} # Wrap GPT2Block layers
+)
+
+fsdp_plugin = FullyShardedDataParallelPlugin(
+ auto_wrap_policy=wrap_policy
+)
+```
+
+## Creating Custom Training Strategy
+
+### Example: Custom Gradient Accumulation
+
+```python
+from accelerate import Accelerator
+
+class CustomGradientAccumulation:
+ def __init__(self, steps=4, adaptive=False):
+ self.steps = steps
+ self.adaptive = adaptive
+ self.current_step = 0
+
+ def should_sync(self, loss):
+ """Decide whether to sync gradients."""
+ self.current_step += 1
+
+ # Adaptive: sync on high loss
+ if self.adaptive and loss > threshold:
+ self.current_step = 0
+ return True
+
+ # Regular: sync every N steps
+ if self.current_step >= self.steps:
+ self.current_step = 0
+ return True
+
+ return False
+
+# Usage
+custom_accum = CustomGradientAccumulation(steps=8, adaptive=True)
+accelerator = Accelerator()
+
+for batch in dataloader:
+ outputs = model(**batch)
+ loss = outputs.loss
+
+ # Scale loss
+ loss = loss / custom_accum.steps
+ accelerator.backward(loss)
+
+ # Conditional sync
+ if custom_accum.should_sync(loss.item()):
+ optimizer.step()
+ optimizer.zero_grad()
+```
+
+### Example: Custom Mixed Precision
+
+```python
+import torch
+
+class CustomMixedPrecision:
+ """Custom mixed precision with dynamic loss scaling."""
+
+ def __init__(self, init_scale=2**16, scale_window=2000):
+ self.scaler = torch.cuda.amp.GradScaler(
+ init_scale=init_scale,
+ growth_interval=scale_window
+ )
+ self.scale_history = []
+
+ def scale_loss(self, loss):
+ """Scale loss for backward."""
+ return self.scaler.scale(loss)
+
+ def unscale_and_clip(self, optimizer, max_norm=1.0):
+ """Unscale gradients and clip."""
+ self.scaler.unscale_(optimizer)
+ torch.nn.utils.clip_grad_norm_(
+ optimizer.param_groups[0]['params'],
+ max_norm
+ )
+
+ def step(self, optimizer):
+ """Optimizer step with scaler update."""
+ scale_before = self.scaler.get_scale()
+ self.scaler.step(optimizer)
+ self.scaler.update()
+ scale_after = self.scaler.get_scale()
+
+ # Track scale changes
+ if scale_before != scale_after:
+ self.scale_history.append(scale_after)
+
+# Usage
+custom_mp = CustomMixedPrecision()
+
+for batch in dataloader:
+ with torch.cuda.amp.autocast(dtype=torch.float16):
+ loss = model(**batch).loss
+
+ scaled_loss = custom_mp.scale_loss(loss)
+ scaled_loss.backward()
+
+ custom_mp.unscale_and_clip(optimizer, max_norm=1.0)
+ custom_mp.step(optimizer)
+ optimizer.zero_grad()
+```
+
+## Advanced: Custom Distributed Backend
+
+### Custom AllReduce Strategy
+
+```python
+import torch.distributed as dist
+
+class CustomAllReduce:
+ """Custom all-reduce with compression."""
+
+ def __init__(self, compression_ratio=0.1):
+ self.compression_ratio = compression_ratio
+
+ def compress_gradients(self, tensor):
+ """Top-k gradient compression."""
+ k = int(tensor.numel() * self.compression_ratio)
+ values, indices = torch.topk(tensor.abs().view(-1), k)
+ return values, indices
+
+ def all_reduce_compressed(self, tensor):
+ """All-reduce with gradient compression."""
+ # Compress
+ values, indices = self.compress_gradients(tensor)
+
+ # All-reduce compressed gradients
+ dist.all_reduce(values, op=dist.ReduceOp.SUM)
+
+ # Decompress
+ tensor_compressed = torch.zeros_like(tensor).view(-1)
+ tensor_compressed[indices] = values / dist.get_world_size()
+
+ return tensor_compressed.view_as(tensor)
+
+# Usage in training loop
+custom_ar = CustomAllReduce(compression_ratio=0.1)
+
+for batch in dataloader:
+ loss = model(**batch).loss
+ loss.backward()
+
+ # Custom all-reduce
+ for param in model.parameters():
+ if param.grad is not None:
+ param.grad.data = custom_ar.all_reduce_compressed(param.grad.data)
+
+ optimizer.step()
+ optimizer.zero_grad()
+```
+
+## Plugin Best Practices
+
+### 1. Validation in `__post_init__`
+
+```python
+@dataclass
+class CustomPlugin:
+ learning_rate: float = 1e-3
+ warmup_steps: int = 1000
+
+ def __post_init__(self):
+ # Validate parameters
+ if self.learning_rate <= 0:
+ raise ValueError("learning_rate must be positive")
+ if self.warmup_steps < 0:
+ raise ValueError("warmup_steps must be non-negative")
+
+ # Compute derived values
+ self.min_lr = self.learning_rate * 0.1
+```
+
+### 2. Compatibility Checks
+
+```python
+@dataclass
+class CustomPlugin:
+ feature_enabled: bool = True
+
+ def is_compatible(self, accelerator):
+ """Check if plugin is compatible with accelerator config."""
+ if self.feature_enabled and accelerator.mixed_precision == 'fp8':
+ raise ValueError("Custom plugin not compatible with FP8")
+ return True
+```
+
+### 3. State Management
+
+```python
+@dataclass
+class CustomPlugin:
+ counter: int = 0
+ history: list = None
+
+ def __post_init__(self):
+ if self.history is None:
+ self.history = []
+
+ def update_state(self, value):
+ """Update plugin state during training."""
+ self.counter += 1
+ self.history.append(value)
+```
+
+## Resources
+
+- Accelerate Plugins: https://huggingface.co/docs/accelerate/package_reference/kwargs
+- DeepSpeed Config: https://www.deepspeed.ai/docs/config-json/
+- FSDP Guide: https://pytorch.org/docs/stable/fsdp.html
+- Custom Training Loops: https://huggingface.co/docs/accelerate/usage_guides/training_tpu
diff --git a/skills/accelerate/references/megatron-integration.md b/skills/accelerate/references/megatron-integration.md
new file mode 100644
index 0000000..61b025b
--- /dev/null
+++ b/skills/accelerate/references/megatron-integration.md
@@ -0,0 +1,489 @@
+# Megatron Integration with Accelerate
+
+## Overview
+
+Accelerate supports Megatron-LM for massive model training with tensor parallelism and pipeline parallelism.
+
+**Megatron capabilities**:
+- **Tensor Parallelism (TP)**: Split layers across GPUs
+- **Pipeline Parallelism (PP)**: Split model depth across GPUs
+- **Data Parallelism (DP)**: Replicate model across GPU groups
+- **Sequence Parallelism**: Split sequences for long contexts
+
+## Setup
+
+### Install Megatron-LM
+
+```bash
+# Clone Megatron-LM repository
+git clone https://github.com/NVIDIA/Megatron-LM.git
+cd Megatron-LM
+pip install -e .
+
+# Install Apex (NVIDIA optimizations)
+git clone https://github.com/NVIDIA/apex
+cd apex
+pip install -v --disable-pip-version-check --no-cache-dir --no-build-isolation \
+ --config-settings "--build-option=--cpp_ext" --config-settings "--build-option=--cuda_ext" ./
+```
+
+### Accelerate Configuration
+
+```bash
+accelerate config
+```
+
+**Questions**:
+```
+In which compute environment are you running?
+> This machine
+
+Which type of machine are you using?
+> Multi-GPU
+
+How many different machines will you use?
+> 1
+
+Do you want to use DeepSpeed/FSDP?
+> No
+
+Do you want to use Megatron-LM?
+> Yes
+
+What is the Tensor Parallelism degree? [1-8]
+> 2
+
+Do you want to enable Sequence Parallelism?
+> No
+
+What is the Pipeline Parallelism degree? [1-8]
+> 2
+
+What is the Data Parallelism degree? [1-8]
+> 2
+
+Where to perform activation checkpointing? ['SELECTIVE', 'FULL', 'NONE']
+> SELECTIVE
+
+Where to perform activation partitioning? ['SEQUENTIAL', 'UNIFORM']
+> SEQUENTIAL
+```
+
+**Generated config** (`~/.cache/huggingface/accelerate/default_config.yaml`):
+```yaml
+compute_environment: LOCAL_MACHINE
+distributed_type: MEGATRON_LM
+downcast_bf16: 'no'
+machine_rank: 0
+main_training_function: main
+megatron_lm_config:
+ megatron_lm_gradient_clipping: 1.0
+ megatron_lm_learning_rate_decay_iters: 320000
+ megatron_lm_num_micro_batches: 1
+ megatron_lm_pp_degree: 2
+ megatron_lm_recompute_activations: true
+ megatron_lm_sequence_parallelism: false
+ megatron_lm_tp_degree: 2
+mixed_precision: bf16
+num_machines: 1
+num_processes: 8
+rdzv_backend: static
+same_network: true
+tpu_env: []
+tpu_use_cluster: false
+tpu_use_sudo: false
+use_cpu: false
+```
+
+## Parallelism Strategies
+
+### Tensor Parallelism (TP)
+
+**Splits each transformer layer across GPUs**:
+
+```python
+# Layer split across 2 GPUs
+# GPU 0: First half of attention heads
+# GPU 1: Second half of attention heads
+
+# Each GPU computes partial outputs
+# All-reduce combines results
+```
+
+**TP degree recommendations**:
+- **TP=1**: No tensor parallelism (single GPU per layer)
+- **TP=2**: 2 GPUs per layer (good for 7-13B models)
+- **TP=4**: 4 GPUs per layer (good for 20-40B models)
+- **TP=8**: 8 GPUs per layer (good for 70B+ models)
+
+**Benefits**:
+- Reduces memory per GPU
+- All-reduce communication (fast)
+
+**Drawbacks**:
+- Requires fast inter-GPU bandwidth (NVLink)
+- Communication overhead per layer
+
+### Pipeline Parallelism (PP)
+
+**Splits model depth across GPUs**:
+
+```python
+# 12-layer model, PP=4
+# GPU 0: Layers 0-2
+# GPU 1: Layers 3-5
+# GPU 2: Layers 6-8
+# GPU 3: Layers 9-11
+```
+
+**PP degree recommendations**:
+- **PP=1**: No pipeline parallelism
+- **PP=2**: 2 pipeline stages (good for 20-40B models)
+- **PP=4**: 4 pipeline stages (good for 70B+ models)
+- **PP=8**: 8 pipeline stages (good for 175B+ models)
+
+**Benefits**:
+- Linear memory reduction (4× PP = 4× less memory)
+- Works across nodes (slower interconnect OK)
+
+**Drawbacks**:
+- Pipeline bubbles (idle time)
+- Requires micro-batching
+
+### Data Parallelism (DP)
+
+**Replicates model across GPU groups**:
+
+```python
+# 8 GPUs, TP=2, PP=2, DP=2
+# Group 0 (GPUs 0-3): Full model replica
+# Group 1 (GPUs 4-7): Full model replica
+```
+
+**DP degree**:
+- `DP = total_gpus / (TP × PP)`
+- Example: 8 GPUs, TP=2, PP=2 → DP=2
+
+**Benefits**:
+- Increases throughput
+- Scales batch size
+
+### Sequence Parallelism
+
+**Splits long sequences across GPUs** (extends TP):
+
+```python
+# 8K sequence, TP=2, Sequence Parallel=True
+# GPU 0: Tokens 0-4095
+# GPU 1: Tokens 4096-8191
+```
+
+**Benefits**:
+- Enables very long sequences (100K+ tokens)
+- Reduces activation memory
+
+**Requirements**:
+- Must use with TP > 1
+- RoPE/ALiBi position encodings work best
+
+## Accelerate Code Example
+
+### Basic Setup
+
+```python
+from accelerate import Accelerator
+from accelerate.utils import MegatronLMPlugin
+
+# Configure Megatron
+megatron_plugin = MegatronLMPlugin(
+ tp_degree=2, # Tensor parallelism degree
+ pp_degree=2, # Pipeline parallelism degree
+ num_micro_batches=4, # Micro-batches for pipeline
+ gradient_clipping=1.0, # Gradient clipping value
+ sequence_parallelism=False, # Enable sequence parallelism
+ recompute_activations=True, # Activation checkpointing
+ use_distributed_optimizer=True, # Distributed optimizer
+ custom_prepare_model_function=None, # Custom model prep
+)
+
+# Initialize accelerator
+accelerator = Accelerator(
+ mixed_precision='bf16',
+ megatron_lm_plugin=megatron_plugin
+)
+
+# Prepare model and optimizer
+model, optimizer, train_dataloader = accelerator.prepare(
+ model, optimizer, train_dataloader
+)
+
+# Training loop (same as DDP!)
+for batch in train_dataloader:
+ optimizer.zero_grad()
+ outputs = model(**batch)
+ loss = outputs.loss
+ accelerator.backward(loss)
+ optimizer.step()
+```
+
+### Full Training Script
+
+```python
+import torch
+from accelerate import Accelerator
+from accelerate.utils import MegatronLMPlugin
+from transformers import GPT2Config, GPT2LMHeadModel
+
+def main():
+ # Megatron configuration
+ megatron_plugin = MegatronLMPlugin(
+ tp_degree=2,
+ pp_degree=2,
+ num_micro_batches=4,
+ gradient_clipping=1.0,
+ )
+
+ accelerator = Accelerator(
+ mixed_precision='bf16',
+ gradient_accumulation_steps=8,
+ megatron_lm_plugin=megatron_plugin
+ )
+
+ # Model
+ config = GPT2Config(
+ n_layer=24,
+ n_head=16,
+ n_embd=1024,
+ )
+ model = GPT2LMHeadModel(config)
+
+ # Optimizer
+ optimizer = torch.optim.AdamW(model.parameters(), lr=6e-4)
+
+ # Prepare
+ model, optimizer, train_loader = accelerator.prepare(
+ model, optimizer, train_loader
+ )
+
+ # Training loop
+ for epoch in range(num_epochs):
+ for batch in train_loader:
+ with accelerator.accumulate(model):
+ outputs = model(**batch)
+ loss = outputs.loss
+ accelerator.backward(loss)
+ optimizer.step()
+ optimizer.zero_grad()
+
+ # Save checkpoint
+ accelerator.wait_for_everyone()
+ accelerator.save_state(f'checkpoint-epoch-{epoch}')
+
+if __name__ == '__main__':
+ main()
+```
+
+### Launch Command
+
+```bash
+# 8 GPUs, TP=2, PP=2, DP=2
+accelerate launch --multi_gpu --num_processes 8 train.py
+
+# Multi-node (2 nodes, 8 GPUs each)
+# Node 0
+accelerate launch --multi_gpu --num_processes 16 \
+ --num_machines 2 --machine_rank 0 \
+ --main_process_ip $MASTER_ADDR \
+ --main_process_port 29500 \
+ train.py
+
+# Node 1
+accelerate launch --multi_gpu --num_processes 16 \
+ --num_machines 2 --machine_rank 1 \
+ --main_process_ip $MASTER_ADDR \
+ --main_process_port 29500 \
+ train.py
+```
+
+## Activation Checkpointing
+
+**Reduces memory by recomputing activations**:
+
+```python
+megatron_plugin = MegatronLMPlugin(
+ recompute_activations=True, # Enable checkpointing
+ checkpoint_num_layers=1, # Checkpoint every N layers
+ distribute_checkpointed_activations=True, # Distribute across TP
+ partition_activations=True, # Partition in PP
+ check_for_nan_in_loss_and_grad=True, # Stability check
+)
+```
+
+**Strategies**:
+- `SELECTIVE`: Checkpoint transformer blocks only
+- `FULL`: Checkpoint all layers
+- `NONE`: No checkpointing
+
+**Memory savings**: 30-50% with 10-15% slowdown
+
+## Distributed Optimizer
+
+**Shards optimizer state across DP ranks**:
+
+```python
+megatron_plugin = MegatronLMPlugin(
+ use_distributed_optimizer=True, # Enable sharded optimizer
+)
+```
+
+**Benefits**:
+- Reduces optimizer memory by DP degree
+- Example: DP=4 → 4× less optimizer memory per GPU
+
+**Compatible with**:
+- AdamW, Adam, SGD
+- Mixed precision training
+
+## Performance Tuning
+
+### Micro-Batch Size
+
+```python
+# Pipeline parallelism requires micro-batching
+megatron_plugin = MegatronLMPlugin(
+ pp_degree=4,
+ num_micro_batches=16, # 16 micro-batches per pipeline
+)
+
+# Effective batch = num_micro_batches × micro_batch_size × DP
+# Example: 16 × 2 × 4 = 128
+```
+
+**Recommendations**:
+- More micro-batches → less pipeline bubble
+- Typical: 4-16 micro-batches
+
+### Sequence Length
+
+```python
+# For long sequences, enable sequence parallelism
+megatron_plugin = MegatronLMPlugin(
+ tp_degree=4,
+ sequence_parallelism=True, # Required: TP > 1
+)
+
+# Enables sequences up to TP × normal limit
+# Example: TP=4, 8K normal → 32K with sequence parallel
+```
+
+### GPU Topology
+
+**NVLink required for TP**:
+```bash
+# Check NVLink topology
+nvidia-smi topo -m
+
+# Good topology (NVLink between all GPUs)
+# GPU0 - GPU1: NV12 (fast)
+# GPU0 - GPU2: NV12 (fast)
+
+# Bad topology (PCIe only)
+# GPU0 - GPU4: PHB (slow, avoid TP across these)
+```
+
+**Recommendations**:
+- **TP**: Within same node (NVLink)
+- **PP**: Across nodes (slower interconnect OK)
+- **DP**: Any topology
+
+## Model Size Guidelines
+
+| Model Size | GPUs | TP | PP | DP | Micro-Batches |
+|------------|------|----|----|----|--------------|
+| 7B | 8 | 1 | 1 | 8 | 1 |
+| 13B | 8 | 2 | 1 | 4 | 1 |
+| 20B | 16 | 4 | 1 | 4 | 1 |
+| 40B | 32 | 4 | 2 | 4 | 4 |
+| 70B | 64 | 8 | 2 | 4 | 8 |
+| 175B | 128 | 8 | 4 | 4 | 16 |
+
+**Assumptions**: BF16, 2K sequence length, A100 80GB
+
+## Checkpointing
+
+### Save Checkpoint
+
+```python
+# Save full model state
+accelerator.save_state('checkpoint-1000')
+
+# Megatron saves separate files per rank
+# checkpoint-1000/
+# pytorch_model_tp_0_pp_0.bin
+# pytorch_model_tp_0_pp_1.bin
+# pytorch_model_tp_1_pp_0.bin
+# pytorch_model_tp_1_pp_1.bin
+# optimizer_tp_0_pp_0.bin
+# ...
+```
+
+### Load Checkpoint
+
+```python
+# Resume training
+accelerator.load_state('checkpoint-1000')
+
+# Automatically loads correct shard per rank
+```
+
+### Convert to Standard PyTorch
+
+```bash
+# Merge Megatron checkpoint to single file
+python merge_megatron_checkpoint.py \
+ --checkpoint-dir checkpoint-1000 \
+ --output pytorch_model.bin
+```
+
+## Common Issues
+
+### Issue: OOM with Pipeline Parallelism
+
+**Solution**: Increase micro-batches
+```python
+megatron_plugin = MegatronLMPlugin(
+ pp_degree=4,
+ num_micro_batches=16, # Increase from 4
+)
+```
+
+### Issue: Slow Training
+
+**Check 1**: Pipeline bubbles (PP too high)
+```python
+# Reduce PP, increase TP
+tp_degree=4 # Increase
+pp_degree=2 # Decrease
+```
+
+**Check 2**: Micro-batch size too small
+```python
+num_micro_batches=8 # Increase
+```
+
+### Issue: NVLink Not Detected
+
+```bash
+# Verify NVLink
+nvidia-smi nvlink -s
+
+# If no NVLink, avoid TP > 1
+# Use PP or DP instead
+```
+
+## Resources
+
+- Megatron-LM: https://github.com/NVIDIA/Megatron-LM
+- Accelerate Megatron docs: https://huggingface.co/docs/accelerate/usage_guides/megatron_lm
+- Paper: "Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism"
+- NVIDIA Apex: https://github.com/NVIDIA/apex
diff --git a/skills/accelerate/references/performance.md b/skills/accelerate/references/performance.md
new file mode 100644
index 0000000..62560d2
--- /dev/null
+++ b/skills/accelerate/references/performance.md
@@ -0,0 +1,525 @@
+# Accelerate Performance Tuning
+
+## Profiling
+
+### Basic Profiling
+
+```python
+from accelerate import Accelerator
+import time
+
+accelerator = Accelerator()
+
+# Warmup
+for _ in range(10):
+ batch = next(iter(dataloader))
+ outputs = model(**batch)
+ loss = outputs.loss
+ accelerator.backward(loss)
+ optimizer.step()
+ optimizer.zero_grad()
+
+# Profile training loop
+start = time.time()
+total_batches = 100
+
+for i, batch in enumerate(dataloader):
+ if i >= total_batches:
+ break
+
+ outputs = model(**batch)
+ loss = outputs.loss
+ accelerator.backward(loss)
+ optimizer.step()
+ optimizer.zero_grad()
+
+accelerator.wait_for_everyone() # Sync all processes
+elapsed = time.time() - start
+
+# Metrics
+batches_per_sec = total_batches / elapsed
+samples_per_sec = (total_batches * batch_size * accelerator.num_processes) / elapsed
+
+print(f"Throughput: {samples_per_sec:.2f} samples/sec")
+print(f"Batches/sec: {batches_per_sec:.2f}")
+```
+
+### PyTorch Profiler Integration
+
+```python
+from torch.profiler import profile, ProfilerActivity
+
+with profile(
+ activities=[ProfilerActivity.CPU, ProfilerActivity.CUDA],
+ record_shapes=True,
+ profile_memory=True,
+ with_stack=True
+) as prof:
+ for i, batch in enumerate(dataloader):
+ if i >= 10: # Profile first 10 batches
+ break
+
+ outputs = model(**batch)
+ loss = outputs.loss
+ accelerator.backward(loss)
+ optimizer.step()
+ optimizer.zero_grad()
+
+# Print profiling results
+print(prof.key_averages().table(
+ sort_by="cuda_time_total", row_limit=20
+))
+
+# Export to Chrome tracing
+prof.export_chrome_trace("trace.json")
+# View at chrome://tracing
+```
+
+## Memory Optimization
+
+### 1. Gradient Accumulation
+
+**Problem**: Large batch size causes OOM
+
+**Solution**: Accumulate gradients across micro-batches
+
+```python
+accelerator = Accelerator(gradient_accumulation_steps=8)
+
+# Effective batch = batch_size × accumulation_steps × num_gpus
+# Example: 4 × 8 × 8 = 256
+
+for batch in dataloader:
+ with accelerator.accumulate(model): # Handles accumulation logic
+ outputs = model(**batch)
+ loss = outputs.loss
+ accelerator.backward(loss)
+ optimizer.step()
+ optimizer.zero_grad()
+```
+
+**Memory savings**: 8× less activation memory (with 8 accumulation steps)
+
+### 2. Gradient Checkpointing
+
+**Enable in model**:
+
+```python
+from transformers import AutoModelForCausalLM
+
+model = AutoModelForCausalLM.from_pretrained(
+ "gpt2",
+ use_cache=False # Required for gradient checkpointing
+)
+
+# Enable checkpointing
+model.gradient_checkpointing_enable()
+
+# Prepare with Accelerate
+model = accelerator.prepare(model)
+```
+
+**Memory savings**: 30-50% with 10-15% slowdown
+
+### 3. Mixed Precision
+
+**BF16 (A100/H100)**:
+```python
+accelerator = Accelerator(mixed_precision='bf16')
+
+# Automatic mixed precision
+for batch in dataloader:
+ outputs = model(**batch) # Forward in BF16
+ loss = outputs.loss
+ accelerator.backward(loss) # Backward in FP32
+ optimizer.step()
+```
+
+**FP16 (V100, older GPUs)**:
+```python
+from accelerate.utils import GradScalerKwargs
+
+scaler_kwargs = GradScalerKwargs(
+ init_scale=2.**16,
+ growth_interval=2000
+)
+
+accelerator = Accelerator(
+ mixed_precision='fp16',
+ kwargs_handlers=[scaler_kwargs]
+)
+```
+
+**Memory savings**: 50% compared to FP32
+
+### 4. CPU Offloading (DeepSpeed)
+
+```python
+from accelerate.utils import DeepSpeedPlugin
+
+ds_plugin = DeepSpeedPlugin(
+ zero_stage=3,
+ offload_optimizer_device="cpu", # Offload optimizer to CPU
+ offload_param_device="cpu", # Offload parameters to CPU
+)
+
+accelerator = Accelerator(
+ deepspeed_plugin=ds_plugin,
+ mixed_precision='bf16'
+)
+```
+
+**Memory savings**: 10-20× for optimizer state, 5-10× for parameters
+
+**Trade-off**: 20-30% slower due to CPU-GPU transfers
+
+### 5. Flash Attention
+
+```python
+# Install flash-attn
+# pip install flash-attn
+
+from transformers import AutoModelForCausalLM
+
+model = AutoModelForCausalLM.from_pretrained(
+ "gpt2",
+ attn_implementation="flash_attention_2" # Enable Flash Attention 2
+)
+
+model = accelerator.prepare(model)
+```
+
+**Memory savings**: 50% for attention, 2× faster
+
+**Requirements**: A100/H100, sequence length must be multiple of 128
+
+## Communication Optimization
+
+### 1. Gradient Bucketing (DDP)
+
+```python
+from accelerate.utils import DistributedDataParallelKwargs
+
+ddp_kwargs = DistributedDataParallelKwargs(
+ bucket_cap_mb=25, # Bucket size for gradient reduction
+ gradient_as_bucket_view=True, # Reduce memory copies
+ static_graph=False # Set True if model doesn't change
+)
+
+accelerator = Accelerator(kwargs_handlers=[ddp_kwargs])
+```
+
+**Recommended bucket sizes**:
+- Small models (<1B): 25 MB
+- Medium models (1-10B): 50-100 MB
+- Large models (>10B): 100-200 MB
+
+### 2. Find Unused Parameters
+
+```python
+# Only enable if model has unused parameters (slower!)
+ddp_kwargs = DistributedDataParallelKwargs(
+ find_unused_parameters=True
+)
+```
+
+**Use case**: Models with conditional branches (e.g., mixture of experts)
+
+**Cost**: 10-20% slower
+
+### 3. NCCL Tuning
+
+```bash
+# Set environment variables before launch
+export NCCL_DEBUG=INFO # Debug info
+export NCCL_IB_DISABLE=0 # Enable InfiniBand
+export NCCL_SOCKET_IFNAME=eth0 # Network interface
+export NCCL_P2P_LEVEL=NVL # Use NVLink
+
+accelerate launch train.py
+```
+
+**NCCL_P2P_LEVEL options**:
+- `NVL`: NVLink (fastest, within node)
+- `PIX`: PCIe (fast, within node)
+- `PHB`: PCIe host bridge (slow, cross-node)
+
+## Data Loading Optimization
+
+### 1. DataLoader Workers
+
+```python
+from torch.utils.data import DataLoader
+
+train_loader = DataLoader(
+ dataset,
+ batch_size=32,
+ num_workers=4, # Parallel data loading
+ pin_memory=True, # Pin memory for faster GPU transfer
+ prefetch_factor=2, # Prefetch batches per worker
+ persistent_workers=True # Keep workers alive between epochs
+)
+
+train_loader = accelerator.prepare(train_loader)
+```
+
+**Recommendations**:
+- `num_workers`: 2-4 per GPU (8 GPUs → 16-32 workers)
+- `pin_memory`: Always True for GPU training
+- `prefetch_factor`: 2-4 (higher for slow data loading)
+
+### 2. Data Preprocessing
+
+```python
+from datasets import load_dataset
+
+# Bad: Preprocess during training (slow)
+dataset = load_dataset("openwebtext")
+
+for batch in dataset:
+ tokens = tokenizer(batch['text']) # Slow!
+ ...
+
+# Good: Preprocess once, save
+dataset = load_dataset("openwebtext")
+tokenized = dataset.map(
+ lambda x: tokenizer(x['text']),
+ batched=True,
+ num_proc=8, # Parallel preprocessing
+ remove_columns=['text']
+)
+tokenized.save_to_disk("preprocessed_data")
+
+# Load preprocessed
+dataset = load_from_disk("preprocessed_data")
+```
+
+### 3. Faster Tokenization
+
+```python
+import os
+
+# Enable Rust-based tokenizers (10× faster)
+os.environ["TOKENIZERS_PARALLELISM"] = "true"
+
+from transformers import AutoTokenizer
+
+tokenizer = AutoTokenizer.from_pretrained(
+ "gpt2",
+ use_fast=True # Use fast Rust tokenizer
+)
+```
+
+## Compilation (PyTorch 2.0+)
+
+### Compile Model
+
+```python
+import torch
+
+# Compile model for faster execution
+model = torch.compile(
+ model,
+ mode="reduce-overhead", # Options: default, reduce-overhead, max-autotune
+ fullgraph=False, # Compile entire graph (stricter)
+ dynamic=True # Support dynamic shapes
+)
+
+model = accelerator.prepare(model)
+```
+
+**Speedup**: 10-50% depending on model
+
+**Compilation modes**:
+- `default`: Balanced (best for most cases)
+- `reduce-overhead`: Min overhead (best for small batches)
+- `max-autotune`: Max performance (slow compile, best for production)
+
+### Compilation Best Practices
+
+```python
+# Bad: Compile after prepare (won't work)
+model = accelerator.prepare(model)
+model = torch.compile(model) # Error!
+
+# Good: Compile before prepare
+model = torch.compile(model)
+model = accelerator.prepare(model)
+
+# Training loop
+for batch in dataloader:
+ # First iteration: slow (compilation)
+ # Subsequent iterations: fast (compiled)
+ outputs = model(**batch)
+ ...
+```
+
+## Benchmarking Different Strategies
+
+### Script Template
+
+```python
+import time
+import torch
+from accelerate import Accelerator
+
+def benchmark_strategy(strategy_name, accelerator_kwargs):
+ """Benchmark a specific training strategy."""
+ accelerator = Accelerator(**accelerator_kwargs)
+
+ # Setup
+ model = create_model()
+ optimizer = torch.optim.AdamW(model.parameters(), lr=1e-4)
+ dataloader = create_dataloader()
+
+ model, optimizer, dataloader = accelerator.prepare(
+ model, optimizer, dataloader
+ )
+
+ # Warmup
+ for i, batch in enumerate(dataloader):
+ if i >= 10:
+ break
+ outputs = model(**batch)
+ loss = outputs.loss
+ accelerator.backward(loss)
+ optimizer.step()
+ optimizer.zero_grad()
+
+ # Benchmark
+ accelerator.wait_for_everyone()
+ torch.cuda.synchronize()
+ start = time.time()
+
+ num_batches = 100
+ for i, batch in enumerate(dataloader):
+ if i >= num_batches:
+ break
+
+ outputs = model(**batch)
+ loss = outputs.loss
+ accelerator.backward(loss)
+ optimizer.step()
+ optimizer.zero_grad()
+
+ accelerator.wait_for_everyone()
+ torch.cuda.synchronize()
+ elapsed = time.time() - start
+
+ # Metrics
+ throughput = (num_batches * batch_size * accelerator.num_processes) / elapsed
+ memory_used = torch.cuda.max_memory_allocated() / 1e9 # GB
+
+ if accelerator.is_main_process:
+ print(f"\n{strategy_name}:")
+ print(f" Throughput: {throughput:.2f} samples/sec")
+ print(f" Memory: {memory_used:.2f} GB")
+ print(f" Time: {elapsed:.2f} sec")
+
+ torch.cuda.reset_peak_memory_stats()
+
+# Benchmark different strategies
+strategies = [
+ ("DDP + FP32", {}),
+ ("DDP + BF16", {"mixed_precision": "bf16"}),
+ ("DDP + BF16 + GradAccum", {"mixed_precision": "bf16", "gradient_accumulation_steps": 4}),
+ ("FSDP", {"fsdp_plugin": fsdp_plugin}),
+ ("DeepSpeed ZeRO-2", {"deepspeed_plugin": ds_plugin_stage2}),
+ ("DeepSpeed ZeRO-3", {"deepspeed_plugin": ds_plugin_stage3}),
+]
+
+for name, kwargs in strategies:
+ benchmark_strategy(name, kwargs)
+```
+
+## Performance Checklist
+
+**Before training**:
+- [ ] Use BF16/FP16 mixed precision
+- [ ] Enable gradient checkpointing (if OOM)
+- [ ] Set appropriate `num_workers` (2-4 per GPU)
+- [ ] Enable `pin_memory=True`
+- [ ] Preprocess data once, not during training
+- [ ] Compile model with `torch.compile` (PyTorch 2.0+)
+
+**For large models**:
+- [ ] Use FSDP or DeepSpeed ZeRO-3
+- [ ] Enable CPU offloading (if still OOM)
+- [ ] Use Flash Attention
+- [ ] Increase gradient accumulation
+
+**For multi-node**:
+- [ ] Check network topology (InfiniBand > Ethernet)
+- [ ] Tune NCCL settings
+- [ ] Use larger bucket sizes for DDP
+- [ ] Verify NVLink for tensor parallelism
+
+**Profiling**:
+- [ ] Profile first 10-100 batches
+- [ ] Check GPU utilization (`nvidia-smi dmon`)
+- [ ] Check data loading time (should be <5% of iteration)
+- [ ] Identify communication bottlenecks
+
+## Common Performance Issues
+
+### Issue: Low GPU Utilization (<80%)
+
+**Cause 1**: Data loading bottleneck
+```python
+# Solution: Increase workers and prefetch
+num_workers=8
+prefetch_factor=4
+```
+
+**Cause 2**: Small batch size
+```python
+# Solution: Increase batch size or use gradient accumulation
+batch_size=32 # Increase
+gradient_accumulation_steps=4 # Or accumulate
+```
+
+### Issue: High Memory Usage
+
+**Solution 1**: Gradient checkpointing
+```python
+model.gradient_checkpointing_enable()
+```
+
+**Solution 2**: Reduce batch size, increase accumulation
+```python
+batch_size=8 # Reduce from 32
+gradient_accumulation_steps=16 # Maintain effective batch
+```
+
+**Solution 3**: Use FSDP or DeepSpeed ZeRO-3
+```python
+accelerator = Accelerator(fsdp_plugin=fsdp_plugin)
+```
+
+### Issue: Slow Multi-GPU Training
+
+**Cause**: Communication bottleneck
+
+**Check 1**: Gradient bucket size
+```python
+ddp_kwargs = DistributedDataParallelKwargs(bucket_cap_mb=100)
+```
+
+**Check 2**: NCCL settings
+```bash
+export NCCL_DEBUG=INFO
+# Check for "Using NVLS" (good) vs "Using PHB" (bad)
+```
+
+**Check 3**: Network bandwidth
+```bash
+# Test inter-GPU bandwidth
+nvidia-smi nvlink -s
+```
+
+## Resources
+
+- Accelerate Performance: https://huggingface.co/docs/accelerate/usage_guides/performance
+- PyTorch Profiler: https://pytorch.org/tutorials/recipes/recipes/profiler_recipe.html
+- NCCL Tuning: https://docs.nvidia.com/deeplearning/nccl/user-guide/docs/env.html
+- Flash Attention: https://github.com/Dao-AILab/flash-attention
diff --git a/skills/bitsandbytes/SKILL.md b/skills/bitsandbytes/SKILL.md
new file mode 100644
index 0000000..9f6880a
--- /dev/null
+++ b/skills/bitsandbytes/SKILL.md
@@ -0,0 +1,411 @@
+---
+name: bitsandbytes
+description: Quantizes LLMs to 8-bit or 4-bit for 50-75% memory reduction with minimal accuracy loss. Use when GPU memory is limited, need to fit larger models, or want faster inference. Supports INT8, NF4, FP4 formats, QLoRA training, and 8-bit optimizers. Works with HuggingFace Transformers.
+version: 1.0.0
+author: Orchestra Research
+license: MIT
+tags: [Optimization, Bitsandbytes, Quantization, 8-Bit, 4-Bit, Memory Optimization, QLoRA, NF4, INT8, HuggingFace, Efficient Inference]
+dependencies: [bitsandbytes, transformers, accelerate, torch]
+---
+
+# bitsandbytes - LLM Quantization
+
+## Quick start
+
+bitsandbytes reduces LLM memory by 50% (8-bit) or 75% (4-bit) with <1% accuracy loss.
+
+**Installation**:
+```bash
+pip install bitsandbytes transformers accelerate
+```
+
+**8-bit quantization** (50% memory reduction):
+```python
+from transformers import AutoModelForCausalLM, BitsAndBytesConfig
+
+config = BitsAndBytesConfig(load_in_8bit=True)
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-7b-hf",
+ quantization_config=config,
+ device_map="auto"
+)
+
+# Memory: 14GB → 7GB
+```
+
+**4-bit quantization** (75% memory reduction):
+```python
+config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.float16
+)
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-7b-hf",
+ quantization_config=config,
+ device_map="auto"
+)
+
+# Memory: 14GB → 3.5GB
+```
+
+## Common workflows
+
+### Workflow 1: Load large model in limited GPU memory
+
+Copy this checklist:
+
+```
+Quantization Loading:
+- [ ] Step 1: Calculate memory requirements
+- [ ] Step 2: Choose quantization level (4-bit or 8-bit)
+- [ ] Step 3: Configure quantization
+- [ ] Step 4: Load and verify model
+```
+
+**Step 1: Calculate memory requirements**
+
+Estimate model memory:
+```
+FP16 memory (GB) = Parameters × 2 bytes / 1e9
+INT8 memory (GB) = Parameters × 1 byte / 1e9
+INT4 memory (GB) = Parameters × 0.5 bytes / 1e9
+
+Example (Llama 2 7B):
+FP16: 7B × 2 / 1e9 = 14 GB
+INT8: 7B × 1 / 1e9 = 7 GB
+INT4: 7B × 0.5 / 1e9 = 3.5 GB
+```
+
+**Step 2: Choose quantization level**
+
+| GPU VRAM | Model Size | Recommended |
+|----------|------------|-------------|
+| 8 GB | 3B | 4-bit |
+| 12 GB | 7B | 4-bit |
+| 16 GB | 7B | 8-bit or 4-bit |
+| 24 GB | 13B | 8-bit or 70B 4-bit |
+| 40+ GB | 70B | 8-bit |
+
+**Step 3: Configure quantization**
+
+For 8-bit (better accuracy):
+```python
+from transformers import BitsAndBytesConfig
+import torch
+
+config = BitsAndBytesConfig(
+ load_in_8bit=True,
+ llm_int8_threshold=6.0, # Outlier threshold
+ llm_int8_has_fp16_weight=False
+)
+```
+
+For 4-bit (maximum memory savings):
+```python
+config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.float16, # Compute in FP16
+ bnb_4bit_quant_type="nf4", # NormalFloat4 (recommended)
+ bnb_4bit_use_double_quant=True # Nested quantization
+)
+```
+
+**Step 4: Load and verify model**
+
+```python
+from transformers import AutoModelForCausalLM, AutoTokenizer
+
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-13b-hf",
+ quantization_config=config,
+ device_map="auto", # Automatic device placement
+ torch_dtype=torch.float16
+)
+
+tokenizer = AutoTokenizer.from_pretrained("meta-llama/Llama-2-13b-hf")
+
+# Test inference
+inputs = tokenizer("Hello, how are you?", return_tensors="pt").to("cuda")
+outputs = model.generate(**inputs, max_length=50)
+print(tokenizer.decode(outputs[0]))
+
+# Check memory
+import torch
+print(f"Memory allocated: {torch.cuda.memory_allocated()/1e9:.2f}GB")
+```
+
+### Workflow 2: Fine-tune with QLoRA (4-bit training)
+
+QLoRA enables fine-tuning large models on consumer GPUs.
+
+Copy this checklist:
+
+```
+QLoRA Fine-tuning:
+- [ ] Step 1: Install dependencies
+- [ ] Step 2: Configure 4-bit base model
+- [ ] Step 3: Add LoRA adapters
+- [ ] Step 4: Train with standard Trainer
+```
+
+**Step 1: Install dependencies**
+
+```bash
+pip install bitsandbytes transformers peft accelerate datasets
+```
+
+**Step 2: Configure 4-bit base model**
+
+```python
+from transformers import AutoModelForCausalLM, BitsAndBytesConfig
+import torch
+
+bnb_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.float16,
+ bnb_4bit_quant_type="nf4",
+ bnb_4bit_use_double_quant=True
+)
+
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-7b-hf",
+ quantization_config=bnb_config,
+ device_map="auto"
+)
+```
+
+**Step 3: Add LoRA adapters**
+
+```python
+from peft import LoraConfig, get_peft_model, prepare_model_for_kbit_training
+
+# Prepare model for training
+model = prepare_model_for_kbit_training(model)
+
+# Configure LoRA
+lora_config = LoraConfig(
+ r=16, # LoRA rank
+ lora_alpha=32, # LoRA alpha
+ target_modules=["q_proj", "k_proj", "v_proj", "o_proj"],
+ lora_dropout=0.05,
+ bias="none",
+ task_type="CAUSAL_LM"
+)
+
+# Add LoRA adapters
+model = get_peft_model(model, lora_config)
+model.print_trainable_parameters()
+# Output: trainable params: 4.2M || all params: 6.7B || trainable%: 0.06%
+```
+
+**Step 4: Train with standard Trainer**
+
+```python
+from transformers import Trainer, TrainingArguments
+
+training_args = TrainingArguments(
+ output_dir="./qlora-output",
+ per_device_train_batch_size=4,
+ gradient_accumulation_steps=4,
+ num_train_epochs=3,
+ learning_rate=2e-4,
+ fp16=True,
+ logging_steps=10,
+ save_strategy="epoch"
+)
+
+trainer = Trainer(
+ model=model,
+ args=training_args,
+ train_dataset=train_dataset,
+ tokenizer=tokenizer
+)
+
+trainer.train()
+
+# Save LoRA adapters (only ~20MB)
+model.save_pretrained("./qlora-adapters")
+```
+
+### Workflow 3: 8-bit optimizer for memory-efficient training
+
+Use 8-bit Adam/AdamW to reduce optimizer memory by 75%.
+
+```
+8-bit Optimizer Setup:
+- [ ] Step 1: Replace standard optimizer
+- [ ] Step 2: Configure training
+- [ ] Step 3: Monitor memory savings
+```
+
+**Step 1: Replace standard optimizer**
+
+```python
+import bitsandbytes as bnb
+from transformers import Trainer, TrainingArguments
+
+# Instead of torch.optim.AdamW
+model = AutoModelForCausalLM.from_pretrained("model-name")
+
+training_args = TrainingArguments(
+ output_dir="./output",
+ per_device_train_batch_size=8,
+ optim="paged_adamw_8bit", # 8-bit optimizer
+ learning_rate=5e-5
+)
+
+trainer = Trainer(
+ model=model,
+ args=training_args,
+ train_dataset=train_dataset
+)
+
+trainer.train()
+```
+
+**Manual optimizer usage**:
+```python
+import bitsandbytes as bnb
+
+optimizer = bnb.optim.AdamW8bit(
+ model.parameters(),
+ lr=1e-4,
+ betas=(0.9, 0.999),
+ eps=1e-8
+)
+
+# Training loop
+for batch in dataloader:
+ loss = model(**batch).loss
+ loss.backward()
+ optimizer.step()
+ optimizer.zero_grad()
+```
+
+**Step 2: Configure training**
+
+Compare memory:
+```
+Standard AdamW optimizer memory = model_params × 8 bytes (states)
+8-bit AdamW memory = model_params × 2 bytes
+Savings = 75% optimizer memory
+
+Example (Llama 2 7B):
+Standard: 7B × 8 = 56 GB
+8-bit: 7B × 2 = 14 GB
+Savings: 42 GB
+```
+
+**Step 3: Monitor memory savings**
+
+```python
+import torch
+
+before = torch.cuda.memory_allocated()
+
+# Training step
+optimizer.step()
+
+after = torch.cuda.memory_allocated()
+print(f"Memory used: {(after-before)/1e9:.2f}GB")
+```
+
+## When to use vs alternatives
+
+**Use bitsandbytes when:**
+- GPU memory limited (need to fit larger model)
+- Training with QLoRA (fine-tune 70B on single GPU)
+- Inference only (50-75% memory reduction)
+- Using HuggingFace Transformers
+- Acceptable 0-2% accuracy degradation
+
+**Use alternatives instead:**
+- **GPTQ/AWQ**: Production serving (faster inference than bitsandbytes)
+- **GGUF**: CPU inference (llama.cpp)
+- **FP8**: H100 GPUs (hardware FP8 faster)
+- **Full precision**: Accuracy critical, memory not constrained
+
+## Common issues
+
+**Issue: CUDA error during loading**
+
+Install matching CUDA version:
+```bash
+# Check CUDA version
+nvcc --version
+
+# Install matching bitsandbytes
+pip install bitsandbytes --no-cache-dir
+```
+
+**Issue: Model loading slow**
+
+Use CPU offload for large models:
+```python
+model = AutoModelForCausalLM.from_pretrained(
+ "model-name",
+ quantization_config=config,
+ device_map="auto",
+ max_memory={0: "20GB", "cpu": "30GB"} # Offload to CPU
+)
+```
+
+**Issue: Lower accuracy than expected**
+
+Try 8-bit instead of 4-bit:
+```python
+config = BitsAndBytesConfig(load_in_8bit=True)
+# 8-bit has <0.5% accuracy loss vs 1-2% for 4-bit
+```
+
+Or use NF4 with double quantization:
+```python
+config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_quant_type="nf4", # Better than fp4
+ bnb_4bit_use_double_quant=True # Extra accuracy
+)
+```
+
+**Issue: OOM even with 4-bit**
+
+Enable CPU offload:
+```python
+model = AutoModelForCausalLM.from_pretrained(
+ "model-name",
+ quantization_config=config,
+ device_map="auto",
+ offload_folder="offload", # Disk offload
+ offload_state_dict=True
+)
+```
+
+## Advanced topics
+
+**QLoRA training guide**: See [references/qlora-training.md](references/qlora-training.md) for complete fine-tuning workflows, hyperparameter tuning, and multi-GPU training.
+
+**Quantization formats**: See [references/quantization-formats.md](references/quantization-formats.md) for INT8, NF4, FP4 comparison, double quantization, and custom quantization configs.
+
+**Memory optimization**: See [references/memory-optimization.md](references/memory-optimization.md) for CPU offloading strategies, gradient checkpointing, and memory profiling.
+
+## Hardware requirements
+
+- **GPU**: NVIDIA with compute capability 7.0+ (Turing, Ampere, Hopper)
+- **VRAM**: Depends on model and quantization
+ - 4-bit Llama 2 7B: 4GB
+ - 4-bit Llama 2 13B: 8GB
+ - 4-bit Llama 2 70B: 24GB
+- **CUDA**: 11.1+ (12.0+ recommended)
+- **PyTorch**: 2.0+
+
+**Supported platforms**: NVIDIA GPUs (primary), AMD ROCm, Intel GPUs (experimental)
+
+## Resources
+
+- GitHub: https://github.com/bitsandbytes-foundation/bitsandbytes
+- HuggingFace docs: https://huggingface.co/docs/transformers/quantization/bitsandbytes
+- QLoRA paper: "QLoRA: Efficient Finetuning of Quantized LLMs" (2023)
+- LLM.int8() paper: "LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale" (2022)
+
+
+
diff --git a/skills/bitsandbytes/references/memory-optimization.md b/skills/bitsandbytes/references/memory-optimization.md
new file mode 100644
index 0000000..ed20e88
--- /dev/null
+++ b/skills/bitsandbytes/references/memory-optimization.md
@@ -0,0 +1,521 @@
+# Memory Optimization
+
+Complete guide to CPU offloading, gradient checkpointing, memory profiling, and advanced memory-saving strategies with bitsandbytes.
+
+## Overview
+
+Memory optimization techniques for fitting large models:
+- **Quantization**: 50-75% reduction (covered in other docs)
+- **CPU offloading**: Move weights to CPU/disk
+- **Gradient checkpointing**: Trade compute for memory
+- **Optimizer strategies**: 8-bit, paged optimizers
+- **Mixed precision**: FP16/BF16 training
+
+## CPU Offloading
+
+### Basic CPU Offloading
+
+Move parts of the model to CPU RAM when not in use.
+
+```python
+from transformers import AutoModelForCausalLM, BitsAndBytesConfig
+import torch
+
+config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16
+)
+
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-70b-hf",
+ quantization_config=config,
+ device_map="auto", # Automatic device placement
+ max_memory={0: "40GB", "cpu": "100GB"} # 40GB GPU, 100GB CPU
+)
+```
+
+**How it works**:
+- Weights stored on CPU
+- Moved to GPU only when needed for computation
+- Automatically managed by `accelerate`
+
+**Trade-off**: ~5-10× slower but enables larger models
+
+### Multi-GPU Offloading
+
+Distribute across multiple GPUs + CPU:
+
+```python
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-405b-hf",
+ quantization_config=config,
+ device_map="auto",
+ max_memory={
+ 0: "70GB", # GPU 0
+ 1: "70GB", # GPU 1
+ 2: "70GB", # GPU 2
+ 3: "70GB", # GPU 3
+ "cpu": "200GB" # CPU RAM
+ }
+)
+```
+
+**Result**: 405B model (4-bit = ~200GB) fits on 4×80GB GPUs + CPU
+
+### Disk Offloading
+
+For models too large even for CPU RAM:
+
+```python
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-405b-hf",
+ quantization_config=config,
+ device_map="auto",
+ offload_folder="./offload", # Disk offload directory
+ offload_state_dict=True,
+ max_memory={0: "40GB", "cpu": "50GB"}
+)
+```
+
+**Trade-off**: Extremely slow (~100× slower) but works
+
+### Manual Device Mapping
+
+For precise control:
+
+```python
+device_map = {
+ "model.embed_tokens": 0, # GPU 0
+ "model.layers.0": 0,
+ "model.layers.1": 0,
+ # ...
+ "model.layers.40": 1, # GPU 1
+ "model.layers.41": 1,
+ # ...
+ "model.layers.79": "cpu", # CPU
+ "model.norm": "cpu",
+ "lm_head": "cpu"
+}
+
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-70b-hf",
+ quantization_config=config,
+ device_map=device_map
+)
+```
+
+## Gradient Checkpointing
+
+Recompute activations during backward pass instead of storing them.
+
+### Enable for HuggingFace Models
+
+```python
+from transformers import AutoModelForCausalLM
+
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-13b-hf",
+ quantization_config=config
+)
+
+# Enable gradient checkpointing
+model.gradient_checkpointing_enable()
+```
+
+**Memory savings**: ~30-50% activation memory
+**Cost**: ~20% slower training
+
+### With QLoRA
+
+```python
+from peft import prepare_model_for_kbit_training
+
+# Enable gradient checkpointing before preparing for training
+model.gradient_checkpointing_enable()
+model = prepare_model_for_kbit_training(
+ model,
+ use_gradient_checkpointing=True
+)
+```
+
+### Configure Checkpointing Frequency
+
+```python
+# Checkpoint every layer (maximum memory savings)
+model.gradient_checkpointing_enable(gradient_checkpointing_kwargs={"use_reentrant": False})
+```
+
+### Memory Breakdown
+
+Example: Llama 2 13B forward pass
+
+| Component | Without Checkpointing | With Checkpointing |
+|-----------|----------------------|-------------------|
+| Model weights | 26 GB | 26 GB |
+| Activations | 12 GB | **3 GB** |
+| Gradients | 26 GB | 26 GB |
+| Optimizer | 52 GB | 52 GB |
+| **Total** | 116 GB | **107 GB** |
+
+**Savings**: ~9GB for 13B model
+
+## 8-Bit Optimizers
+
+Use 8-bit optimizer states instead of 32-bit.
+
+### Standard AdamW Memory
+
+```
+Optimizer memory = 2 × model_params × 4 bytes (FP32)
+ = 8 × model_params
+
+Example (Llama 2 70B):
+= 8 × 70B = 560 GB
+```
+
+### 8-Bit AdamW Memory
+
+```
+Optimizer memory = 2 × model_params × 1 byte (INT8)
+ = 2 × model_params
+
+Example (Llama 2 70B):
+= 2 × 70B = 140 GB
+
+Savings: 420 GB (75% reduction!)
+```
+
+### Enable in Transformers
+
+```python
+from transformers import TrainingArguments
+
+training_args = TrainingArguments(
+ output_dir="./output",
+ per_device_train_batch_size=4,
+ optim="paged_adamw_8bit", # 8-bit optimizer
+ learning_rate=2e-4
+)
+```
+
+### Available 8-Bit Optimizers
+
+| Optimizer | Name | Use Case |
+|-----------|------|----------|
+| AdamW 8-bit | `adamw_8bit` | General training |
+| Paged AdamW 8-bit | `paged_adamw_8bit` | **Recommended** (prevents OOM) |
+| Paged AdamW 32-bit | `paged_adamw_32bit` | High accuracy needed |
+
+**Recommendation**: Always use `paged_adamw_8bit`
+
+### Manual Usage
+
+```python
+import bitsandbytes as bnb
+
+optimizer = bnb.optim.PagedAdamW8bit(
+ model.parameters(),
+ lr=1e-4,
+ betas=(0.9, 0.999),
+ eps=1e-8
+)
+```
+
+## Paged Optimizers
+
+Paged optimizers use unified memory (GPU + CPU) to prevent OOM.
+
+### How It Works
+
+- Optimizer states stored in paged memory
+- Pages swap between GPU and CPU as needed
+- Prevents hard OOM crashes
+
+### Configuration
+
+```python
+from transformers import TrainingArguments
+
+training_args = TrainingArguments(
+ optim="paged_adamw_8bit", # Enables paging
+ # Paging happens automatically
+)
+```
+
+### Benefits
+
+✅ No hard OOM (graceful degradation)
+✅ Enables larger batch sizes
+✅ Combines with 8-bit for maximum savings
+
+### Performance
+
+**Speed**: ~5-10% slower than standard optimizer
+**Memory**: Effectively unlimited (uses CPU + swap)
+
+## Mixed Precision Training
+
+Use lower precision for faster training and less memory.
+
+### BF16 Training (Recommended)
+
+```python
+training_args = TrainingArguments(
+ bf16=True, # BFloat16 training
+ bf16_full_eval=True
+)
+```
+
+**Requirements**: Ampere+ GPUs (A100, H100, RTX 3090+)
+
+**Benefits**:
+- 2× faster training
+- 50% less activation memory
+- Better stability than FP16
+
+### FP16 Training
+
+```python
+training_args = TrainingArguments(
+ fp16=True, # Float16 training
+ fp16_full_eval=True
+)
+```
+
+**Requirements**: Volta+ GPUs (V100, A100, RTX 2080+)
+
+**Benefits**:
+- 2× faster training
+- 50% less activation memory
+- Slightly less stable than BF16
+
+### Precision Comparison
+
+| Precision | Speed | Memory | Stability | Use Case |
+|-----------|-------|--------|-----------|----------|
+| FP32 | 1× | 100% | Best | Debugging |
+| BF16 | 2× | 50% | Good | **Recommended** |
+| FP16 | 2× | 50% | Fair | V100 only |
+
+## Complete Memory Optimization Stack
+
+### Maximum Optimization (Llama 2 70B on Single A100 80GB)
+
+```python
+from transformers import AutoModelForCausalLM, BitsAndBytesConfig, TrainingArguments
+from peft import LoraConfig, get_peft_model, prepare_model_for_kbit_training
+import torch
+
+# Step 1: 4-bit quantization
+bnb_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16,
+ bnb_4bit_quant_type="nf4",
+ bnb_4bit_use_double_quant=True
+)
+
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-70b-hf",
+ quantization_config=bnb_config,
+ device_map="auto",
+ max_memory={0: "70GB", "cpu": "100GB"} # CPU offload if needed
+)
+
+# Step 2: Gradient checkpointing
+model.gradient_checkpointing_enable()
+
+# Step 3: Prepare for training
+model = prepare_model_for_kbit_training(model, use_gradient_checkpointing=True)
+
+# Step 4: LoRA adapters
+lora_config = LoraConfig(
+ r=16, # Lower rank for memory
+ lora_alpha=32,
+ target_modules="all-linear",
+ lora_dropout=0.05,
+ bias="none",
+ task_type="CAUSAL_LM"
+)
+
+model = get_peft_model(model, lora_config)
+
+# Step 5: Training arguments
+training_args = TrainingArguments(
+ output_dir="./output",
+ per_device_train_batch_size=1, # Small batch
+ gradient_accumulation_steps=16, # Effective batch = 16
+ bf16=True, # Mixed precision
+ optim="paged_adamw_8bit", # 8-bit optimizer
+ max_grad_norm=0.3,
+ learning_rate=2e-4
+)
+
+# Memory usage: ~75GB (fits on A100 80GB!)
+```
+
+### Memory Breakdown
+
+| Component | Memory |
+|-----------|--------|
+| Model (4-bit) | 35 GB |
+| LoRA adapters | 0.5 GB |
+| Activations (with checkpointing) | 8 GB |
+| Gradients | 0.5 GB |
+| Optimizer (8-bit paged) | 1 GB |
+| Batch buffer | 10 GB |
+| CUDA overhead | 5 GB |
+| **Total** | **~75 GB** |
+
+## Memory Profiling
+
+### PyTorch Memory Profiler
+
+```python
+import torch
+
+# Start profiling
+torch.cuda.empty_cache()
+torch.cuda.reset_peak_memory_stats()
+
+# Your code here
+model = AutoModelForCausalLM.from_pretrained(...)
+model.generate(...)
+
+# Check memory
+print(f"Allocated: {torch.cuda.memory_allocated()/1e9:.2f} GB")
+print(f"Peak: {torch.cuda.max_memory_allocated()/1e9:.2f} GB")
+print(f"Cached: {torch.cuda.memory_reserved()/1e9:.2f} GB")
+```
+
+### Detailed Memory Summary
+
+```python
+print(torch.cuda.memory_summary())
+```
+
+Output:
+```
+|===========================================================================|
+| PyTorch CUDA memory summary |
+|---------------------------------------------------------------------------|
+| Metric | Cur Usage | Peak Usage | Tot Alloc | Tot Freed |
+|---------------------------------------------------------------------------|
+| Allocated memory | 45.2 GB | 52.3 GB | 156.8 GB | 111.6 GB |
+| Active memory | 45.2 GB | 52.3 GB | 156.8 GB | 111.6 GB |
+| GPU reserved | 46.0 GB | 54.0 GB | 54.0 GB | 8.0 GB |
+|===========================================================================|
+```
+
+### Track Memory During Training
+
+```python
+from transformers import TrainerCallback
+
+class MemoryCallback(TrainerCallback):
+ def on_step_end(self, args, state, control, **kwargs):
+ if state.global_step % 10 == 0:
+ allocated = torch.cuda.memory_allocated() / 1e9
+ reserved = torch.cuda.memory_reserved() / 1e9
+ print(f"Step {state.global_step}: {allocated:.2f}GB allocated, {reserved:.2f}GB reserved")
+
+trainer = Trainer(
+ model=model,
+ args=training_args,
+ callbacks=[MemoryCallback()]
+)
+```
+
+## Troubleshooting OOM
+
+### Diagnostic Steps
+
+1. **Check current memory**:
+ ```python
+ print(torch.cuda.memory_summary())
+ ```
+
+2. **Try smaller batch**:
+ ```python
+ per_device_train_batch_size=1
+ ```
+
+3. **Enable gradient checkpointing**:
+ ```python
+ model.gradient_checkpointing_enable()
+ ```
+
+4. **Use 8-bit optimizer**:
+ ```python
+ optim="paged_adamw_8bit"
+ ```
+
+5. **Add CPU offloading**:
+ ```python
+ max_memory={0: "70GB", "cpu": "100GB"}
+ ```
+
+6. **Reduce LoRA rank**:
+ ```python
+ r=8 # Instead of 16
+ ```
+
+### Emergency: Last Resort
+
+```python
+# Absolute minimum memory config
+model = AutoModelForCausalLM.from_pretrained(
+ "model-name",
+ quantization_config=BitsAndBytesConfig(load_in_4bit=True),
+ device_map="auto",
+ max_memory={0: "20GB", "cpu": "200GB"},
+ offload_folder="./offload"
+)
+
+model.gradient_checkpointing_enable()
+
+training_args = TrainingArguments(
+ per_device_train_batch_size=1,
+ gradient_accumulation_steps=64,
+ bf16=True,
+ optim="paged_adamw_8bit"
+)
+```
+
+**Result**: Extremely slow but will probably work
+
+## Best Practices
+
+1. **Start with quantization**: 4-bit gives 75% savings
+2. **Add gradient checkpointing**: 30-50% activation savings
+3. **Use 8-bit optimizer**: 75% optimizer savings
+4. **Enable mixed precision**: 50% activation savings
+5. **CPU offload only if needed**: Slow but enables larger models
+6. **Profile regularly**: Identify memory bottlenecks
+7. **Test with small batches**: Prevent OOM during development
+
+## Memory Estimation Formula
+
+```
+Total Memory = Model + Activations + Gradients + Optimizer + Buffer
+
+Model = Parameters × Bytes per param
+Activations = Batch × Seq × Hidden × Layers × Bytes per activation
+Gradients = Parameters × Bytes per gradient
+Optimizer = Parameters × Optimizer factor × Bytes
+Buffer = 2-5 GB (CUDA overhead)
+```
+
+**With all optimizations**:
+```
+Model = Parameters × 0.5 (4-bit)
+Activations = Activations × 0.3 (checkpointing + BF16)
+Gradients = Parameters × 0.5 (LoRA only)
+Optimizer = Parameters × 2 (8-bit)
+```
+
+## References
+
+- PyTorch memory management: https://pytorch.org/docs/stable/notes/cuda.html
+- Accelerate device_map: https://huggingface.co/docs/accelerate/usage_guides/big_modeling
+- Gradient checkpointing: https://pytorch.org/docs/stable/checkpoint.html
+- bitsandbytes optimizers: https://github.com/bitsandbytes-foundation/bitsandbytes#optimizer
diff --git a/skills/bitsandbytes/references/qlora-training.md b/skills/bitsandbytes/references/qlora-training.md
new file mode 100644
index 0000000..9455c0d
--- /dev/null
+++ b/skills/bitsandbytes/references/qlora-training.md
@@ -0,0 +1,521 @@
+# QLoRA Training
+
+Complete guide to fine-tuning large language models using 4-bit quantization with QLoRA (Quantized Low-Rank Adaptation).
+
+## Overview
+
+QLoRA enables fine-tuning 70B+ parameter models on consumer GPUs by:
+- Loading base model in 4-bit (75% memory reduction)
+- Training only small LoRA adapters (~20MB)
+- Maintaining near-full-precision quality
+
+**Memory savings**:
+- Llama 2 70B: 140GB → 35GB (4-bit) + 20MB (LoRA) = **35GB total**
+- Fits on single A100 80GB!
+
+**Accuracy**: <1% degradation vs full fine-tuning
+
+## Quick Start
+
+### Basic QLoRA Fine-tuning
+
+```python
+from transformers import AutoModelForCausalLM, BitsAndBytesConfig, TrainingArguments
+from peft import LoraConfig, get_peft_model, prepare_model_for_kbit_training
+import torch
+
+# Step 1: Load model in 4-bit
+bnb_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16,
+ bnb_4bit_quant_type="nf4",
+ bnb_4bit_use_double_quant=True
+)
+
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-70b-hf",
+ quantization_config=bnb_config,
+ device_map="auto",
+ torch_dtype=torch.bfloat16
+)
+
+# Step 2: Prepare for k-bit training
+model = prepare_model_for_kbit_training(model)
+
+# Step 3: Add LoRA adapters
+lora_config = LoraConfig(
+ r=64,
+ lora_alpha=16,
+ target_modules="all-linear",
+ lora_dropout=0.1,
+ bias="none",
+ task_type="CAUSAL_LM"
+)
+
+model = get_peft_model(model, lora_config)
+model.print_trainable_parameters()
+# trainable params: 335M || all params: 70B || trainable%: 0.48%
+
+# Step 4: Train
+from trl import SFTTrainer
+
+training_args = TrainingArguments(
+ output_dir="./qlora-70b",
+ per_device_train_batch_size=4,
+ gradient_accumulation_steps=4,
+ num_train_epochs=3,
+ learning_rate=2e-4,
+ bf16=True,
+ optim="paged_adamw_8bit",
+ logging_steps=10,
+ save_strategy="epoch"
+)
+
+trainer = SFTTrainer(
+ model=model,
+ args=training_args,
+ train_dataset=dataset,
+ tokenizer=tokenizer
+)
+
+trainer.train()
+```
+
+## Complete Training Workflows
+
+### Workflow 1: Single GPU Training (Consumer GPU)
+
+Train Llama 2 13B on RTX 4090 (24GB).
+
+**Step 1: Prepare dataset**
+
+```python
+from datasets import load_dataset
+
+# Load instruction dataset
+dataset = load_dataset("timdettmers/openassistant-guanaco")
+
+# Format for instruction tuning
+def format_instruction(example):
+ return {
+ "text": f"### Human: {example['text']}\n### Assistant: {example['output']}"
+ }
+
+dataset = dataset.map(format_instruction)
+```
+
+**Step 2: Configure quantization**
+
+```python
+bnb_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16, # BF16 for stability
+ bnb_4bit_quant_type="nf4", # NormalFloat4 (recommended)
+ bnb_4bit_use_double_quant=True # Nested quantization
+)
+```
+
+**Step 3: Load and prepare model**
+
+```python
+from transformers import AutoModelForCausalLM, AutoTokenizer
+
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-13b-hf",
+ quantization_config=bnb_config,
+ device_map="auto"
+)
+
+tokenizer = AutoTokenizer.from_pretrained("meta-llama/Llama-2-13b-hf")
+tokenizer.pad_token = tokenizer.eos_token
+
+# Enable gradient checkpointing (further memory savings)
+model.gradient_checkpointing_enable()
+model = prepare_model_for_kbit_training(model, use_gradient_checkpointing=True)
+```
+
+**Step 4: Configure LoRA**
+
+```python
+from peft import LoraConfig
+
+lora_config = LoraConfig(
+ r=16, # LoRA rank (lower = less memory)
+ lora_alpha=32, # Scaling factor
+ target_modules="all-linear", # Apply to all linear layers
+ lora_dropout=0.05,
+ bias="none",
+ task_type="CAUSAL_LM"
+)
+
+model = get_peft_model(model, lora_config)
+```
+
+**Step 5: Train**
+
+```python
+training_args = TrainingArguments(
+ output_dir="./qlora-13b-results",
+ per_device_train_batch_size=4,
+ gradient_accumulation_steps=4, # Effective batch = 16
+ warmup_steps=100,
+ num_train_epochs=1,
+ learning_rate=2e-4,
+ bf16=True,
+ logging_steps=10,
+ save_strategy="steps",
+ save_steps=100,
+ eval_strategy="steps",
+ eval_steps=100,
+ optim="paged_adamw_8bit", # 8-bit optimizer
+ max_grad_norm=0.3,
+ max_steps=1000
+)
+
+trainer = SFTTrainer(
+ model=model,
+ args=training_args,
+ train_dataset=dataset["train"],
+ eval_dataset=dataset["test"],
+ tokenizer=tokenizer,
+ max_seq_length=512
+)
+
+trainer.train()
+```
+
+**Memory usage**: ~18GB on RTX 4090 (24GB)
+
+### Workflow 2: Multi-GPU Training (FSDP + QLoRA)
+
+Train Llama 2 70B on 8×A100 (80GB each).
+
+**Step 1: Configure FSDP-compatible quantization**
+
+```python
+bnb_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16,
+ bnb_4bit_quant_type="nf4",
+ bnb_4bit_use_double_quant=True,
+ bnb_4bit_quant_storage=torch.bfloat16 # CRITICAL for FSDP!
+)
+```
+
+**Important**: `bnb_4bit_quant_storage=torch.bfloat16` ensures 4-bit layers are wrapped identically to regular layers for FSDP sharding.
+
+**Step 2: Launch with accelerate**
+
+Create `fsdp_config.yaml`:
+```yaml
+compute_environment: LOCAL_MACHINE
+distributed_type: FSDP
+fsdp_config:
+ fsdp_auto_wrap_policy: TRANSFORMER_BASED_WRAP
+ fsdp_backward_prefetch_policy: BACKWARD_PRE
+ fsdp_forward_prefetch: true
+ fsdp_sharding_strategy: 1 # FULL_SHARD
+ fsdp_state_dict_type: SHARDED_STATE_DICT
+ fsdp_transformer_layer_cls_to_wrap: LlamaDecoderLayer
+mixed_precision: bf16
+num_processes: 8
+```
+
+**Launch training**:
+```bash
+accelerate launch --config_file fsdp_config.yaml train_qlora.py
+```
+
+**train_qlora.py**:
+```python
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-70b-hf",
+ quantization_config=bnb_config,
+ torch_dtype=torch.bfloat16
+)
+
+# Rest same as single-GPU workflow
+model = prepare_model_for_kbit_training(model)
+model = get_peft_model(model, lora_config)
+
+trainer = SFTTrainer(...)
+trainer.train()
+```
+
+**Memory per GPU**: ~40GB (70B model sharded across 8 GPUs)
+
+### Workflow 3: Extremely Large Models (405B)
+
+Train Llama 3.1 405B on 8×H100 (80GB each).
+
+**Requirements**:
+- 8×H100 80GB GPUs
+- 256GB+ system RAM
+- FSDP + QLoRA
+
+**Configuration**:
+```python
+bnb_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16,
+ bnb_4bit_quant_type="nf4",
+ bnb_4bit_use_double_quant=True,
+ bnb_4bit_quant_storage=torch.bfloat16
+)
+
+lora_config = LoraConfig(
+ r=32, # Higher rank for 405B
+ lora_alpha=64,
+ target_modules="all-linear",
+ lora_dropout=0.1,
+ bias="none",
+ task_type="CAUSAL_LM"
+)
+
+training_args = TrainingArguments(
+ per_device_train_batch_size=1, # Small batch
+ gradient_accumulation_steps=32, # Effective batch = 256
+ learning_rate=1e-4, # Lower LR for large model
+ bf16=True,
+ optim="paged_adamw_8bit",
+ gradient_checkpointing=True
+)
+```
+
+**Memory per GPU**: ~70GB (405B in 4-bit / 8 GPUs)
+
+## Hyperparameter Tuning
+
+### LoRA Rank (r)
+
+Controls adapter capacity:
+
+| Model Size | Recommended r | Trainable Params | Use Case |
+|------------|---------------|------------------|----------|
+| 7B | 8-16 | ~4M | Simple tasks |
+| 13B | 16-32 | ~8M | General fine-tuning |
+| 70B | 32-64 | ~80M | Complex tasks |
+| 405B | 64-128 | ~300M | Maximum capacity |
+
+**Trade-off**: Higher r = more capacity but more memory and slower training
+
+### LoRA Alpha
+
+Scaling factor for LoRA updates:
+
+```python
+effective_learning_rate = learning_rate * (lora_alpha / r)
+```
+
+**Recommended**: `lora_alpha = 2 × r`
+- r=16 → alpha=32
+- r=64 → alpha=128
+
+### Target Modules
+
+**Options**:
+- `"all-linear"`: All linear layers (recommended for QLoRA)
+- `["q_proj", "v_proj"]`: Only attention (minimal)
+- `["q_proj", "k_proj", "v_proj", "o_proj"]`: All attention
+- `["q_proj", "k_proj", "v_proj", "o_proj", "gate_proj", "up_proj", "down_proj"]`: Attention + FFN
+
+**Trade-off**: More modules = better performance but more memory
+
+### Learning Rate
+
+| Model Size | Recommended LR |
+|------------|----------------|
+| 7-13B | 2e-4 to 3e-4 |
+| 70B | 1e-4 to 2e-4 |
+| 405B | 5e-5 to 1e-4 |
+
+**Rule**: Larger models need lower learning rates
+
+### Batch Size
+
+```python
+effective_batch_size = per_device_batch_size × gradient_accumulation_steps × num_gpus
+```
+
+**Recommended effective batch sizes**:
+- Instruction tuning: 64-128
+- Continued pretraining: 256-512
+
+### Quantization Dtype
+
+| Dtype | Speed | Accuracy | Use Case |
+|-------|-------|----------|----------|
+| `torch.float32` | Slow | Best | Debugging |
+| `torch.bfloat16` | Fast | Good | **Recommended** |
+| `torch.float16` | Fastest | Risky | May have precision issues |
+
+## Advanced Techniques
+
+### Gradient Checkpointing
+
+Save memory by recomputing activations:
+
+```python
+model.gradient_checkpointing_enable()
+model = prepare_model_for_kbit_training(model, use_gradient_checkpointing=True)
+```
+
+**Memory savings**: ~30-40% activation memory
+**Cost**: ~20% slower training
+
+### Nested Quantization
+
+Quantize the quantization constants:
+
+```python
+bnb_config = BitsAndBytesConfig(
+ bnb_4bit_use_double_quant=True # Enable nested quantization
+)
+```
+
+**Memory savings**: Additional ~2-3% reduction
+**Accuracy**: Minimal impact
+
+### CPU Offloading
+
+For models that still don't fit:
+
+```python
+model = AutoModelForCausalLM.from_pretrained(
+ "model-name",
+ quantization_config=bnb_config,
+ device_map="auto",
+ max_memory={0: "40GB", "cpu": "100GB"}
+)
+```
+
+**Trade-off**: Much slower but enables larger models
+
+### Paged Optimizers
+
+Use paged memory for optimizer states:
+
+```python
+training_args = TrainingArguments(
+ optim="paged_adamw_8bit" # Or paged_adamw_32bit
+)
+```
+
+**Benefit**: Prevents OOM from optimizer states
+
+## Deployment
+
+### Save LoRA Adapters
+
+```python
+# Save only adapters (~20MB)
+model.save_pretrained("./qlora-adapters")
+tokenizer.save_pretrained("./qlora-adapters")
+```
+
+### Load for Inference
+
+```python
+from peft import PeftModel
+
+# Load base model in 4-bit
+base_model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-70b-hf",
+ quantization_config=bnb_config,
+ device_map="auto"
+)
+
+# Load adapters
+model = PeftModel.from_pretrained(base_model, "./qlora-adapters")
+
+# Inference
+inputs = tokenizer("Question here", return_tensors="pt").to("cuda")
+outputs = model.generate(**inputs, max_length=200)
+```
+
+### Merge Adapters (Optional)
+
+```python
+# Merge LoRA into base weights
+model = model.merge_and_unload()
+
+# Save merged model
+model.save_pretrained("./merged-model")
+```
+
+**Note**: Merged model loses 4-bit quantization (back to FP16/BF16)
+
+## Troubleshooting
+
+### OOM During Training
+
+1. Reduce batch size:
+ ```python
+ per_device_train_batch_size=1
+ ```
+
+2. Increase gradient accumulation:
+ ```python
+ gradient_accumulation_steps=16
+ ```
+
+3. Lower LoRA rank:
+ ```python
+ r=8 # Instead of 16
+ ```
+
+4. Enable gradient checkpointing
+
+5. Use CPU offloading
+
+### Low Quality Results
+
+1. Increase LoRA rank:
+ ```python
+ r=64 # Instead of 16
+ ```
+
+2. Train longer:
+ ```python
+ num_train_epochs=3 # Instead of 1
+ ```
+
+3. Use more target modules:
+ ```python
+ target_modules="all-linear"
+ ```
+
+4. Check learning rate (try 1e-4 to 3e-4)
+
+### Slow Training
+
+1. Disable gradient checkpointing (if memory allows)
+
+2. Increase batch size
+
+3. Use BF16:
+ ```python
+ bf16=True
+ ```
+
+4. Use paged optimizer
+
+## Best Practices
+
+1. **Start small**: Test on 7B before 70B
+2. **Monitor loss**: Should decrease steadily
+3. **Use validation**: Track eval loss to detect overfitting
+4. **Save checkpoints**: Every 100-500 steps
+5. **Log hyperparameters**: For reproducibility
+6. **Test inference**: Verify quality before full training
+
+## Example: Complete Training Script
+
+See full working example at `examples/qlora_training.py` in the repository.
+
+## References
+
+- QLoRA paper: "QLoRA: Efficient Finetuning of Quantized LLMs" (Dettmers et al., 2023)
+- bitsandbytes GitHub: https://github.com/bitsandbytes-foundation/bitsandbytes
+- PEFT documentation: https://huggingface.co/docs/peft
+- FSDP+QLoRA guide: https://huggingface.co/blog/fsdp-qlora
diff --git a/skills/bitsandbytes/references/quantization-formats.md b/skills/bitsandbytes/references/quantization-formats.md
new file mode 100644
index 0000000..9abe8d7
--- /dev/null
+++ b/skills/bitsandbytes/references/quantization-formats.md
@@ -0,0 +1,447 @@
+# Quantization Formats
+
+Complete guide to INT8, NF4, FP4 quantization formats, double quantization, and custom configurations in bitsandbytes.
+
+## Overview
+
+bitsandbytes supports multiple quantization formats:
+- **INT8**: 8-bit integer quantization (LLM.int8())
+- **NF4**: 4-bit NormalFloat (for normally distributed weights)
+- **FP4**: 4-bit FloatPoint (for uniformly distributed weights)
+- **Double Quantization**: Quantize the quantization constants
+
+## INT8 Quantization
+
+### LLM.int8() Algorithm
+
+LLM.int8() uses mixed 8-bit/16-bit matrix multiplication:
+- Most features (>99.9%) computed in INT8
+- Outlier features (>threshold) computed in FP16
+- Results combined for final output
+
+**Memory**: 50% reduction (2 bytes → 1 byte per parameter)
+**Accuracy**: <0.5% degradation
+
+### Configuration
+
+```python
+from transformers import BitsAndBytesConfig
+
+config = BitsAndBytesConfig(
+ load_in_8bit=True,
+ llm_int8_threshold=6.0, # Outlier threshold
+ llm_int8_has_fp16_weight=False, # Use INT8 storage
+ llm_int8_skip_modules=["lm_head"] # Skip certain layers
+)
+```
+
+### Parameters Explained
+
+**`llm_int8_threshold`** (default: 6.0):
+- Activations with magnitude > threshold are kept in FP16
+- Lower = more FP16 (slower but more accurate)
+- Higher = more INT8 (faster but less accurate)
+
+```python
+# Conservative (more accurate)
+llm_int8_threshold=5.0
+
+# Aggressive (faster)
+llm_int8_threshold=8.0
+```
+
+**`llm_int8_has_fp16_weight`** (default: False):
+- `False`: Store weights in INT8 (50% memory savings)
+- `True`: Store in FP16, quantize only during computation (no memory savings)
+
+**`llm_int8_skip_modules`**:
+```python
+# Skip specific layers (keep in FP16)
+llm_int8_skip_modules=["lm_head", "embed_tokens"]
+```
+
+### Example
+
+```python
+from transformers import AutoModelForCausalLM
+
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-13b-hf",
+ quantization_config=config,
+ device_map="auto"
+)
+
+# Memory: 26GB (FP16) → 13GB (INT8)
+```
+
+### When to Use INT8
+
+✅ **Use INT8 when**:
+- Need high accuracy (<0.5% loss)
+- Model fits with 50% reduction
+- Have Turing+ GPU (tensor cores)
+
+❌ **Don't use when**:
+- Need maximum memory savings (use 4-bit)
+- Inference speed critical (use GPTQ/AWQ)
+
+## 4-Bit Quantization
+
+### NormalFloat4 (NF4)
+
+Optimized for normally distributed weights (most neural networks).
+
+**How it works**:
+- Bins chosen to minimize quantization error for normal distribution
+- Asymmetric quantization bins
+- Better for transformer weights
+
+**Configuration**:
+```python
+config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16,
+ bnb_4bit_quant_type="nf4" # NormalFloat4
+)
+```
+
+**Memory**: 75% reduction (2 bytes → 0.5 bytes per parameter)
+
+### FloatPoint4 (FP4)
+
+Standard 4-bit floating point for uniform distributions.
+
+**How it works**:
+- Symmetric quantization bins
+- Better for weights with broader dynamic range
+- Less common for transformers
+
+**Configuration**:
+```python
+config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16,
+ bnb_4bit_quant_type="fp4" # FloatPoint4
+)
+```
+
+### NF4 vs FP4 Comparison
+
+| Aspect | NF4 | FP4 |
+|--------|-----|-----|
+| Distribution | Normal | Uniform |
+| Typical use | **Transformers** | CNNs, unusual architectures |
+| Accuracy | **Better for LLMs** | Worse for LLMs |
+| Speed | Same | Same |
+| Recommendation | ✅ Default | Use only if NF4 fails |
+
+**Rule of thumb**: Always use NF4 for transformers.
+
+### Example Comparison
+
+```python
+# NF4 (recommended)
+nf4_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_quant_type="nf4"
+)
+
+# FP4 (alternative)
+fp4_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_quant_type="fp4"
+)
+
+# Load and compare
+model_nf4 = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-7b-hf",
+ quantization_config=nf4_config
+)
+
+model_fp4 = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-7b-hf",
+ quantization_config=fp4_config
+)
+
+# Typical results on MMLU:
+# NF4: 45.2%
+# FP4: 43.8%
+# FP16: 45.9%
+```
+
+## Compute Dtype
+
+The `bnb_4bit_compute_dtype` controls the precision used for actual computation.
+
+### Options
+
+**torch.bfloat16** (recommended):
+```python
+bnb_4bit_compute_dtype=torch.bfloat16
+```
+- Good balance of speed and accuracy
+- Recommended for A100/H100
+- Prevents numerical instability
+
+**torch.float16**:
+```python
+bnb_4bit_compute_dtype=torch.float16
+```
+- Slightly faster than BF16
+- Risk of overflow/underflow
+- Use only if BF16 unavailable
+
+**torch.float32**:
+```python
+bnb_4bit_compute_dtype=torch.float32
+```
+- Most accurate
+- Slowest (no tensor core acceleration)
+- Debugging only
+
+### Performance Comparison
+
+| Dtype | Speed | Accuracy | Memory |
+|-------|-------|----------|--------|
+| FP32 | 1× (baseline) | 100% | 4 bytes |
+| FP16 | 3-4× | 99.5% | 2 bytes |
+| BF16 | 3-4× | **99.8%** | 2 bytes |
+
+**Recommendation**: Always use `torch.bfloat16` if supported.
+
+## Double Quantization
+
+Quantize the quantization constants for additional memory savings.
+
+### How It Works
+
+Standard 4-bit quantization stores:
+- 4-bit quantized weights
+- FP32 scaling factors (4 bytes per block)
+
+Double quantization:
+- 4-bit quantized weights
+- **INT8 quantized scaling factors** (1 byte per block)
+
+**Additional savings**: ~2-3% memory reduction
+
+### Configuration
+
+```python
+config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_quant_type="nf4",
+ bnb_4bit_use_double_quant=True # Enable double quantization
+)
+```
+
+### Example
+
+```python
+# Without double quant
+model_single = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-70b-hf",
+ quantization_config=BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_use_double_quant=False
+ )
+)
+# Memory: ~36GB
+
+# With double quant
+model_double = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-70b-hf",
+ quantization_config=BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_use_double_quant=True
+ )
+)
+# Memory: ~35GB (saves ~1GB)
+```
+
+**Accuracy impact**: Negligible (<0.1%)
+
+**Recommendation**: Always enable for maximum memory savings.
+
+## Quantization Storage
+
+Controls storage dtype for quantized weights (important for FSDP).
+
+### Configuration
+
+```python
+config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_quant_storage=torch.bfloat16 # Storage dtype
+)
+```
+
+### When to Use
+
+**Default (uint8)**:
+- Single GPU training/inference
+- No special requirements
+
+**torch.bfloat16** (for FSDP):
+```python
+bnb_4bit_quant_storage=torch.bfloat16
+```
+- **Required for FSDP+QLoRA**
+- Ensures 4-bit layers wrapped like regular layers
+- Enables proper model sharding
+
+### Example: FSDP Configuration
+
+```python
+# CRITICAL: Set quant_storage for FSDP
+fsdp_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16,
+ bnb_4bit_quant_type="nf4",
+ bnb_4bit_use_double_quant=True,
+ bnb_4bit_quant_storage=torch.bfloat16 # Must match torch_dtype!
+)
+
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-70b-hf",
+ quantization_config=fsdp_config,
+ torch_dtype=torch.bfloat16 # Must match quant_storage!
+)
+```
+
+## Recommended Configurations
+
+### Production Inference (Best Accuracy)
+
+```python
+BitsAndBytesConfig(
+ load_in_8bit=True,
+ llm_int8_threshold=6.0
+)
+```
+
+**Use case**: Maximum accuracy with 50% memory savings
+
+### Production Inference (Maximum Memory Savings)
+
+```python
+BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16,
+ bnb_4bit_quant_type="nf4",
+ bnb_4bit_use_double_quant=True
+)
+```
+
+**Use case**: 75% memory reduction with <1% accuracy loss
+
+### QLoRA Training (Single GPU)
+
+```python
+BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16,
+ bnb_4bit_quant_type="nf4",
+ bnb_4bit_use_double_quant=True
+)
+```
+
+**Use case**: Fine-tune 70B on RTX 3090
+
+### FSDP + QLoRA (Multi-GPU)
+
+```python
+BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16,
+ bnb_4bit_quant_type="nf4",
+ bnb_4bit_use_double_quant=True,
+ bnb_4bit_quant_storage=torch.bfloat16 # CRITICAL!
+)
+```
+
+**Use case**: Fine-tune 405B on 8×H100
+
+## Advanced: Block-wise Quantization
+
+bitsandbytes uses block-wise quantization:
+- Weights divided into blocks (typically 64 or 128 elements)
+- Each block has own scaling factor
+- Better accuracy than tensor-wise quantization
+
+**Block size** (automatically determined):
+```python
+# Typical block sizes
+# 4-bit: 64 elements per block
+# 8-bit: 64 elements per block
+```
+
+**Cannot be configured** (internal implementation detail).
+
+## Quantization Quality Metrics
+
+### Perplexity (Lower is Better)
+
+| Model | FP16 | INT8 | NF4 | NF4+DQ |
+|-------|------|------|-----|--------|
+| Llama 2 7B | 5.12 | 5.14 | 5.18 | 5.19 |
+| Llama 2 13B | 4.88 | 4.90 | 4.93 | 4.94 |
+| Llama 2 70B | 3.32 | 3.33 | 3.35 | 3.36 |
+
+**Conclusion**: <1% degradation for all quantization methods
+
+### MMLU Accuracy (Higher is Better)
+
+| Model | FP16 | INT8 | NF4 | FP4 |
+|-------|------|------|-----|-----|
+| Llama 2 7B | 45.9% | 45.7% | 45.2% | 43.8% |
+| Llama 2 13B | 54.8% | 54.6% | 54.1% | 52.9% |
+| Llama 2 70B | 68.9% | 68.7% | 68.4% | 67.2% |
+
+**Conclusion**: NF4 is significantly better than FP4 for transformers
+
+## Troubleshooting
+
+### "Quantization failed" Error
+
+Try different quant type:
+```python
+# If NF4 fails
+bnb_4bit_quant_type="fp4"
+```
+
+### Numerical Instability
+
+Use BF16 compute:
+```python
+bnb_4bit_compute_dtype=torch.bfloat16
+```
+
+### Poor Quality with 4-bit
+
+1. Try 8-bit instead:
+ ```python
+ load_in_8bit=True
+ ```
+
+2. Enable double quantization:
+ ```python
+ bnb_4bit_use_double_quant=True
+ ```
+
+3. Use BF16 compute dtype
+
+### FSDP Errors
+
+Ensure quant_storage matches torch_dtype:
+```python
+bnb_4bit_quant_storage=torch.bfloat16
+torch_dtype=torch.bfloat16 # Must match!
+```
+
+## References
+
+- LLM.int8() paper: "LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale" (2022)
+- QLoRA paper: "QLoRA: Efficient Finetuning of Quantized LLMs" (2023)
+- bitsandbytes GitHub: https://github.com/bitsandbytes-foundation/bitsandbytes
+- HuggingFace quantization docs: https://huggingface.co/docs/transformers/quantization/bitsandbytes
diff --git a/skills/clip/SKILL.md b/skills/clip/SKILL.md
new file mode 100644
index 0000000..e5282ae
--- /dev/null
+++ b/skills/clip/SKILL.md
@@ -0,0 +1,253 @@
+---
+name: clip
+description: OpenAI's model connecting vision and language. Enables zero-shot image classification, image-text matching, and cross-modal retrieval. Trained on 400M image-text pairs. Use for image search, content moderation, or vision-language tasks without fine-tuning. Best for general-purpose image understanding.
+version: 1.0.0
+author: Orchestra Research
+license: MIT
+tags: [Multimodal, CLIP, Vision-Language, Zero-Shot, Image Classification, OpenAI, Image Search, Cross-Modal Retrieval, Content Moderation]
+dependencies: [transformers, torch, pillow]
+---
+
+# CLIP - Contrastive Language-Image Pre-Training
+
+OpenAI's model that understands images from natural language.
+
+## When to use CLIP
+
+**Use when:**
+- Zero-shot image classification (no training data needed)
+- Image-text similarity/matching
+- Semantic image search
+- Content moderation (detect NSFW, violence)
+- Visual question answering
+- Cross-modal retrieval (image→text, text→image)
+
+**Metrics**:
+- **25,300+ GitHub stars**
+- Trained on 400M image-text pairs
+- Matches ResNet-50 on ImageNet (zero-shot)
+- MIT License
+
+**Use alternatives instead**:
+- **BLIP-2**: Better captioning
+- **LLaVA**: Vision-language chat
+- **Segment Anything**: Image segmentation
+
+## Quick start
+
+### Installation
+
+```bash
+pip install git+https://github.com/openai/CLIP.git
+pip install torch torchvision ftfy regex tqdm
+```
+
+### Zero-shot classification
+
+```python
+import torch
+import clip
+from PIL import Image
+
+# Load model
+device = "cuda" if torch.cuda.is_available() else "cpu"
+model, preprocess = clip.load("ViT-B/32", device=device)
+
+# Load image
+image = preprocess(Image.open("photo.jpg")).unsqueeze(0).to(device)
+
+# Define possible labels
+text = clip.tokenize(["a dog", "a cat", "a bird", "a car"]).to(device)
+
+# Compute similarity
+with torch.no_grad():
+ image_features = model.encode_image(image)
+ text_features = model.encode_text(text)
+
+ # Cosine similarity
+ logits_per_image, logits_per_text = model(image, text)
+ probs = logits_per_image.softmax(dim=-1).cpu().numpy()
+
+# Print results
+labels = ["a dog", "a cat", "a bird", "a car"]
+for label, prob in zip(labels, probs[0]):
+ print(f"{label}: {prob:.2%}")
+```
+
+## Available models
+
+```python
+# Models (sorted by size)
+models = [
+ "RN50", # ResNet-50
+ "RN101", # ResNet-101
+ "ViT-B/32", # Vision Transformer (recommended)
+ "ViT-B/16", # Better quality, slower
+ "ViT-L/14", # Best quality, slowest
+]
+
+model, preprocess = clip.load("ViT-B/32")
+```
+
+| Model | Parameters | Speed | Quality |
+|-------|------------|-------|---------|
+| RN50 | 102M | Fast | Good |
+| ViT-B/32 | 151M | Medium | Better |
+| ViT-L/14 | 428M | Slow | Best |
+
+## Image-text similarity
+
+```python
+# Compute embeddings
+image_features = model.encode_image(image)
+text_features = model.encode_text(text)
+
+# Normalize
+image_features /= image_features.norm(dim=-1, keepdim=True)
+text_features /= text_features.norm(dim=-1, keepdim=True)
+
+# Cosine similarity
+similarity = (image_features @ text_features.T).item()
+print(f"Similarity: {similarity:.4f}")
+```
+
+## Semantic image search
+
+```python
+# Index images
+image_paths = ["img1.jpg", "img2.jpg", "img3.jpg"]
+image_embeddings = []
+
+for img_path in image_paths:
+ image = preprocess(Image.open(img_path)).unsqueeze(0).to(device)
+ with torch.no_grad():
+ embedding = model.encode_image(image)
+ embedding /= embedding.norm(dim=-1, keepdim=True)
+ image_embeddings.append(embedding)
+
+image_embeddings = torch.cat(image_embeddings)
+
+# Search with text query
+query = "a sunset over the ocean"
+text_input = clip.tokenize([query]).to(device)
+with torch.no_grad():
+ text_embedding = model.encode_text(text_input)
+ text_embedding /= text_embedding.norm(dim=-1, keepdim=True)
+
+# Find most similar images
+similarities = (text_embedding @ image_embeddings.T).squeeze(0)
+top_k = similarities.topk(3)
+
+for idx, score in zip(top_k.indices, top_k.values):
+ print(f"{image_paths[idx]}: {score:.3f}")
+```
+
+## Content moderation
+
+```python
+# Define categories
+categories = [
+ "safe for work",
+ "not safe for work",
+ "violent content",
+ "graphic content"
+]
+
+text = clip.tokenize(categories).to(device)
+
+# Check image
+with torch.no_grad():
+ logits_per_image, _ = model(image, text)
+ probs = logits_per_image.softmax(dim=-1)
+
+# Get classification
+max_idx = probs.argmax().item()
+max_prob = probs[0, max_idx].item()
+
+print(f"Category: {categories[max_idx]} ({max_prob:.2%})")
+```
+
+## Batch processing
+
+```python
+# Process multiple images
+images = [preprocess(Image.open(f"img{i}.jpg")) for i in range(10)]
+images = torch.stack(images).to(device)
+
+with torch.no_grad():
+ image_features = model.encode_image(images)
+ image_features /= image_features.norm(dim=-1, keepdim=True)
+
+# Batch text
+texts = ["a dog", "a cat", "a bird"]
+text_tokens = clip.tokenize(texts).to(device)
+
+with torch.no_grad():
+ text_features = model.encode_text(text_tokens)
+ text_features /= text_features.norm(dim=-1, keepdim=True)
+
+# Similarity matrix (10 images × 3 texts)
+similarities = image_features @ text_features.T
+print(similarities.shape) # (10, 3)
+```
+
+## Integration with vector databases
+
+```python
+# Store CLIP embeddings in Chroma/FAISS
+import chromadb
+
+client = chromadb.Client()
+collection = client.create_collection("image_embeddings")
+
+# Add image embeddings
+for img_path, embedding in zip(image_paths, image_embeddings):
+ collection.add(
+ embeddings=[embedding.cpu().numpy().tolist()],
+ metadatas=[{"path": img_path}],
+ ids=[img_path]
+ )
+
+# Query with text
+query = "a sunset"
+text_embedding = model.encode_text(clip.tokenize([query]))
+results = collection.query(
+ query_embeddings=[text_embedding.cpu().numpy().tolist()],
+ n_results=5
+)
+```
+
+## Best practices
+
+1. **Use ViT-B/32 for most cases** - Good balance
+2. **Normalize embeddings** - Required for cosine similarity
+3. **Batch processing** - More efficient
+4. **Cache embeddings** - Expensive to recompute
+5. **Use descriptive labels** - Better zero-shot performance
+6. **GPU recommended** - 10-50× faster
+7. **Preprocess images** - Use provided preprocess function
+
+## Performance
+
+| Operation | CPU | GPU (V100) |
+|-----------|-----|------------|
+| Image encoding | ~200ms | ~20ms |
+| Text encoding | ~50ms | ~5ms |
+| Similarity compute | <1ms | <1ms |
+
+## Limitations
+
+1. **Not for fine-grained tasks** - Best for broad categories
+2. **Requires descriptive text** - Vague labels perform poorly
+3. **Biased on web data** - May have dataset biases
+4. **No bounding boxes** - Whole image only
+5. **Limited spatial understanding** - Position/counting weak
+
+## Resources
+
+- **GitHub**: https://github.com/openai/CLIP ⭐ 25,300+
+- **Paper**: https://arxiv.org/abs/2103.00020
+- **Colab**: https://colab.research.google.com/github/openai/clip/
+- **License**: MIT
+
+
diff --git a/skills/clip/references/applications.md b/skills/clip/references/applications.md
new file mode 100644
index 0000000..38e9a05
--- /dev/null
+++ b/skills/clip/references/applications.md
@@ -0,0 +1,207 @@
+# CLIP Applications Guide
+
+Practical applications and use cases for CLIP.
+
+## Zero-shot image classification
+
+```python
+import torch
+import clip
+from PIL import Image
+
+model, preprocess = clip.load("ViT-B/32")
+
+# Define categories
+categories = [
+ "a photo of a dog",
+ "a photo of a cat",
+ "a photo of a bird",
+ "a photo of a car",
+ "a photo of a person"
+]
+
+# Prepare image
+image = preprocess(Image.open("photo.jpg")).unsqueeze(0)
+text = clip.tokenize(categories)
+
+# Classify
+with torch.no_grad():
+ image_features = model.encode_image(image)
+ text_features = model.encode_text(text)
+
+ logits_per_image, _ = model(image, text)
+ probs = logits_per_image.softmax(dim=-1).cpu().numpy()
+
+# Print results
+for category, prob in zip(categories, probs[0]):
+ print(f"{category}: {prob:.2%}")
+```
+
+## Semantic image search
+
+```python
+# Index images
+image_database = []
+image_paths = ["img1.jpg", "img2.jpg", "img3.jpg"]
+
+for img_path in image_paths:
+ image = preprocess(Image.open(img_path)).unsqueeze(0)
+ with torch.no_grad():
+ features = model.encode_image(image)
+ features /= features.norm(dim=-1, keepdim=True)
+ image_database.append((img_path, features))
+
+# Search with text
+query = "a sunset over mountains"
+text_input = clip.tokenize([query])
+
+with torch.no_grad():
+ text_features = model.encode_text(text_input)
+ text_features /= text_features.norm(dim=-1, keepdim=True)
+
+# Find matches
+similarities = []
+for img_path, img_features in image_database:
+ similarity = (text_features @ img_features.T).item()
+ similarities.append((img_path, similarity))
+
+# Sort by similarity
+similarities.sort(key=lambda x: x[1], reverse=True)
+for img_path, score in similarities[:3]:
+ print(f"{img_path}: {score:.3f}")
+```
+
+## Content moderation
+
+```python
+# Define safety categories
+categories = [
+ "safe for work content",
+ "not safe for work content",
+ "violent or graphic content",
+ "hate speech or offensive content",
+ "spam or misleading content"
+]
+
+text = clip.tokenize(categories)
+
+# Check image
+with torch.no_grad():
+ logits, _ = model(image, text)
+ probs = logits.softmax(dim=-1)
+
+# Get classification
+max_idx = probs.argmax().item()
+confidence = probs[0, max_idx].item()
+
+if confidence > 0.7:
+ print(f"Classified as: {categories[max_idx]} ({confidence:.2%})")
+else:
+ print(f"Uncertain classification (confidence: {confidence:.2%})")
+```
+
+## Image-to-text retrieval
+
+```python
+# Text database
+captions = [
+ "A beautiful sunset over the ocean",
+ "A cute dog playing in the park",
+ "A modern city skyline at night",
+ "A delicious pizza with toppings"
+]
+
+# Encode captions
+caption_features = []
+for caption in captions:
+ text = clip.tokenize([caption])
+ with torch.no_grad():
+ features = model.encode_text(text)
+ features /= features.norm(dim=-1, keepdim=True)
+ caption_features.append(features)
+
+caption_features = torch.cat(caption_features)
+
+# Find matching captions for image
+with torch.no_grad():
+ image_features = model.encode_image(image)
+ image_features /= image_features.norm(dim=-1, keepdim=True)
+
+similarities = (image_features @ caption_features.T).squeeze(0)
+top_k = similarities.topk(3)
+
+for idx, score in zip(top_k.indices, top_k.values):
+ print(f"{captions[idx]}: {score:.3f}")
+```
+
+## Visual question answering
+
+```python
+# Create yes/no questions
+image = preprocess(Image.open("photo.jpg")).unsqueeze(0)
+
+questions = [
+ "a photo showing people",
+ "a photo showing animals",
+ "a photo taken indoors",
+ "a photo taken outdoors",
+ "a photo taken during daytime",
+ "a photo taken at night"
+]
+
+text = clip.tokenize(questions)
+
+with torch.no_grad():
+ logits, _ = model(image, text)
+ probs = logits.softmax(dim=-1)
+
+# Answer questions
+for question, prob in zip(questions, probs[0]):
+ answer = "Yes" if prob > 0.5 else "No"
+ print(f"{question}: {answer} ({prob:.2%})")
+```
+
+## Image deduplication
+
+```python
+# Detect duplicate/similar images
+def compute_similarity(img1_path, img2_path):
+ img1 = preprocess(Image.open(img1_path)).unsqueeze(0)
+ img2 = preprocess(Image.open(img2_path)).unsqueeze(0)
+
+ with torch.no_grad():
+ feat1 = model.encode_image(img1)
+ feat2 = model.encode_image(img2)
+
+ feat1 /= feat1.norm(dim=-1, keepdim=True)
+ feat2 /= feat2.norm(dim=-1, keepdim=True)
+
+ similarity = (feat1 @ feat2.T).item()
+
+ return similarity
+
+# Check for duplicates
+threshold = 0.95
+image_pairs = [("img1.jpg", "img2.jpg"), ("img1.jpg", "img3.jpg")]
+
+for img1, img2 in image_pairs:
+ sim = compute_similarity(img1, img2)
+ if sim > threshold:
+ print(f"{img1} and {img2} are duplicates (similarity: {sim:.3f})")
+```
+
+## Best practices
+
+1. **Use descriptive labels** - "a photo of X" works better than just "X"
+2. **Normalize embeddings** - Always normalize for cosine similarity
+3. **Batch processing** - Process multiple images/texts together
+4. **Cache embeddings** - Expensive to recompute
+5. **Set appropriate thresholds** - Test on validation data
+6. **Use GPU** - 10-50× faster than CPU
+7. **Consider model size** - ViT-B/32 good default, ViT-L/14 for best quality
+
+## Resources
+
+- **Paper**: https://arxiv.org/abs/2103.00020
+- **GitHub**: https://github.com/openai/CLIP
+- **Colab**: https://colab.research.google.com/github/openai/clip/
diff --git a/skills/find-skills/SKILL.md b/skills/find-skills/SKILL.md
new file mode 100644
index 0000000..206dae7
--- /dev/null
+++ b/skills/find-skills/SKILL.md
@@ -0,0 +1,133 @@
+---
+name: find-skills
+description: Helps users discover and install agent skills when they ask questions like "how do I do X", "find a skill for X", or express interest in extending capabilities. Uses a non-interactive installer script suitable for automated agents.
+---
+
+# Find Skills
+
+This skill helps you discover and install skills from the open agent skills ecosystem.
+
+## When to Use This Skill
+
+Use this skill when the user:
+
+- Asks "how do I do X" where X might be a common task with an existing skill
+- Says "find a skill for X" or "is there a skill for X"
+- Wants to search for tools, templates, or workflows
+- Expresses interest in extending agent capabilities
+- Mentions they wish they had help with a specific domain (design, testing, deployment, etc.)
+
+## Step 1: Search for Skills
+
+Use `npx -y skills find` with a relevant keyword to search the ecosystem:
+
+```bash
+npx -y skills find [query]
+```
+
+Examples:
+- User asks "help me with React performance" → `npx -y skills find react performance`
+- User asks "is there a skill for PR reviews?" → `npx -y skills find pr review`
+- User asks "I need to create a changelog" → `npx -y skills find changelog`
+
+The search results will show installable skills like:
+
+```
+vercel-labs/agent-skills@vercel-react-best-practices
+└ https://skills.sh/vercel-labs/agent-skills/vercel-react-best-practices
+```
+
+Browse all available skills at: https://skills.sh/
+
+## Step 2: Present Options
+
+When you find relevant skills, present them to the user with:
+1. The skill name and what it does
+2. A link to learn more on skills.sh
+
+Ask the user which skill(s) they want to install. All skills are installed to `./skills/` in the current working directory.
+
+## Step 3: Install with the Script
+
+**IMPORTANT: Do NOT use `npx -y skills add` for installation** — it requires interactive prompts.
+
+Use the bundled installer script instead:
+
+```bash
+python /skills/find-skills/scripts/install_skill.py --url
+```
+
+### Install Commands
+
+**From a GitHub URL** (most common — copy the URL from search results):
+```bash
+python /skills/find-skills/scripts/install_skill.py \
+ --url https://github.com/owner/repo/tree/main/skill-name
+```
+
+**From skills.sh shorthand** (owner/repo@skill):
+```bash
+python /skills/find-skills/scripts/install_skill.py \
+ --url vercel-labs/agent-skills@vercel-react-best-practices
+```
+
+**From repo + path** (install specific skills from a multi-skill repo):
+```bash
+# Single skill
+python /skills/find-skills/scripts/install_skill.py \
+ --repo owner/repo --path skill-name
+
+# Multiple skills from same repo
+python /skills/find-skills/scripts/install_skill.py \
+ --repo owner/repo --path skill-a --path skill-b
+```
+
+**With a specific git branch or tag**:
+```bash
+python /skills/find-skills/scripts/install_skill.py \
+ --repo owner/repo --path skill-name --ref v2.0
+```
+
+### Installer Options
+
+| Option | Description |
+|--------|-------------|
+| `--url` | GitHub URL or owner/repo@skill shorthand |
+| `--repo` | GitHub repo (owner/repo format) |
+| `--path` | Path to skill inside repo (repeatable) |
+| `--ref` | Git branch or tag |
+| `--dest` | Custom destination directory (default: `./skills`) |
+
+## Step 4: Confirm Installation
+
+After installation, verify by listing the skills directory:
+
+```bash
+ls /skills/ # all skills (system + user merged)
+```
+
+Then read the installed skill's SKILL.md to confirm it loaded correctly:
+
+```bash
+read_file /skills//SKILL.md
+```
+
+## Common Skill Categories
+
+| Category | Example Queries |
+|----------|----------------|
+| Web Development | react, nextjs, typescript, css, tailwind |
+| Testing | testing, jest, playwright, e2e |
+| DevOps | deploy, docker, kubernetes, ci-cd |
+| Documentation | docs, readme, changelog, api-docs |
+| Code Quality | review, lint, refactor, best-practices |
+| Design | ui, ux, design-system, accessibility |
+| Productivity | workflow, automation, git |
+
+## When No Skills Are Found
+
+If no relevant skills exist:
+
+1. Acknowledge that no existing skill was found
+2. Offer to help with the task directly using your general capabilities
+3. Mention the user could create their own skill with `npx -y skills init`
diff --git a/skills/find-skills/scripts/install_skill.py b/skills/find-skills/scripts/install_skill.py
new file mode 100644
index 0000000..f89194a
--- /dev/null
+++ b/skills/find-skills/scripts/install_skill.py
@@ -0,0 +1,211 @@
+#!/usr/bin/env python3
+"""Install a skill from GitHub into a local skills directory.
+
+Self-contained installer — no external dependencies beyond git.
+
+Usage examples:
+ # Install from a GitHub URL (auto-detects repo, ref, path)
+ python install_skill.py --url https://github.com/anthropics/skills/tree/main/excel
+
+ # Install from repo + path
+ python install_skill.py --repo anthropics/skills --path excel
+
+ # Install multiple skills from the same repo
+ python install_skill.py --repo anthropics/skills --path excel --path pdf
+
+ # Install with a specific git ref
+ python install_skill.py --repo org/repo --path my-skill --ref v2.0
+"""
+
+from __future__ import annotations
+
+import argparse
+import os
+import re
+import shutil
+import subprocess
+import sys
+import tempfile
+
+
+def parse_github_url(url: str) -> tuple[str, str | None, str | None]:
+ """Parse a GitHub URL into (repo, ref, path).
+
+ Supports formats:
+ https://github.com/owner/repo
+ https://github.com/owner/repo/tree/main/path/to/skill
+ github.com/owner/repo/tree/branch/path
+ owner/repo@skill-name (shorthand from skills.sh)
+
+ Returns:
+ (repo, ref_or_none, path_or_none)
+ """
+ # Shorthand: owner/repo@path
+ if "@" in url and "://" not in url:
+ repo, path = url.split("@", 1)
+ return repo.strip(), None, path.strip()
+
+ # Strip protocol and github.com prefix
+ cleaned = re.sub(r"^https?://", "", url)
+ cleaned = re.sub(r"^github\.com/", "", cleaned)
+ cleaned = cleaned.rstrip("/")
+
+ # Match: owner/repo/tree/ref/path...
+ m = re.match(r"^([^/]+/[^/]+)/tree/([^/]+)(?:/(.+))?$", cleaned)
+ if m:
+ return m.group(1), m.group(2), m.group(3)
+
+ # Match: owner/repo (no tree)
+ m = re.match(r"^([^/]+/[^/]+)$", cleaned)
+ if m:
+ return m.group(1), None, None
+
+ raise ValueError(f"Cannot parse GitHub URL: {url}")
+
+
+def clone_repo(repo: str, ref: str | None, dest: str) -> None:
+ """Shallow-clone a GitHub repo."""
+ clone_url = f"https://github.com/{repo}.git"
+ cmd = ["git", "clone", "--depth", "1"]
+ if ref:
+ cmd += ["--branch", ref]
+ cmd += [clone_url, dest]
+
+ result = subprocess.run(cmd, capture_output=True, text=True)
+ if result.returncode != 0:
+ raise RuntimeError(f"git clone failed: {result.stderr.strip()}")
+
+
+def copy_skill(src: str, dest_dir: str) -> str:
+ """Copy a skill directory to the destination.
+
+ Returns:
+ The skill name (directory basename).
+ """
+ skill_name = os.path.basename(src.rstrip("/"))
+ target = os.path.join(dest_dir, skill_name)
+
+ if os.path.exists(target):
+ shutil.rmtree(target)
+ print(f" Replaced existing: {skill_name}")
+
+ shutil.copytree(src, target)
+ return skill_name
+
+
+def validate_skill(path: str) -> bool:
+ """Check that a directory looks like a valid skill (has SKILL.md)."""
+ return os.path.isfile(os.path.join(path, "SKILL.md"))
+
+
+def install(
+ repo: str,
+ paths: list[str],
+ ref: str | None,
+ dest: str,
+) -> list[str]:
+ """Install skill(s) from a GitHub repo.
+
+ Returns:
+ List of installed skill names.
+ """
+ os.makedirs(dest, exist_ok=True)
+ installed: list[str] = []
+
+ with tempfile.TemporaryDirectory(prefix="skill-install-") as tmp:
+ clone_dir = os.path.join(tmp, "repo")
+ print(f"Cloning {repo}" + (f" @{ref}" if ref else "") + "...")
+ clone_repo(repo, ref, clone_dir)
+
+ if not paths:
+ # No path specified — treat entire repo as a single skill
+ if validate_skill(clone_dir):
+ name = copy_skill(clone_dir, dest)
+ installed.append(name)
+ else:
+ # List top-level directories that look like skills
+ for entry in sorted(os.listdir(clone_dir)):
+ entry_path = os.path.join(clone_dir, entry)
+ if os.path.isdir(entry_path) and validate_skill(entry_path):
+ name = copy_skill(entry_path, dest)
+ installed.append(name)
+
+ if not installed:
+ print("No valid skills found in repository root.", file=sys.stderr)
+ else:
+ for p in paths:
+ skill_path = os.path.join(clone_dir, p.strip("/"))
+ if not os.path.isdir(skill_path):
+ print(f" Path not found: {p}", file=sys.stderr)
+ continue
+ if not validate_skill(skill_path):
+ print(f" No SKILL.md in: {p}", file=sys.stderr)
+ continue
+ name = copy_skill(skill_path, dest)
+ installed.append(name)
+
+ return installed
+
+
+def main() -> int:
+ parser = argparse.ArgumentParser(
+ description="Install skills from GitHub into a local skills directory.",
+ )
+ src = parser.add_mutually_exclusive_group(required=True)
+ src.add_argument(
+ "--url",
+ help="GitHub URL (e.g. https://github.com/owner/repo/tree/main/skill-name)",
+ )
+ src.add_argument(
+ "--repo",
+ help="GitHub repo (e.g. owner/repo)",
+ )
+ parser.add_argument(
+ "--path",
+ action="append",
+ default=[],
+ help="Path to skill inside repo (repeatable)",
+ )
+ parser.add_argument(
+ "--ref",
+ default=None,
+ help="Git branch or tag (default: repo default branch)",
+ )
+ parser.add_argument(
+ "--dest",
+ default="./skills",
+ help="Destination directory (default: ./skills)",
+ )
+
+ args = parser.parse_args()
+ dest = args.dest
+
+ # Parse source
+ if args.url:
+ repo, ref, path = parse_github_url(args.url)
+ ref = args.ref or ref
+ paths = [path] if path else args.path
+ else:
+ repo = args.repo
+ ref = args.ref
+ paths = args.path
+
+ try:
+ installed = install(repo, paths, ref, dest)
+ except RuntimeError as e:
+ print(f"Error: {e}", file=sys.stderr)
+ return 1
+
+ if installed:
+ print(f"\nInstalled {len(installed)} skill(s) to {dest}/:")
+ for name in installed:
+ print(f" - {name}")
+ else:
+ print("No skills were installed.", file=sys.stderr)
+ return 1
+
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/skills/flash-attention/SKILL.md b/skills/flash-attention/SKILL.md
new file mode 100644
index 0000000..71e9319
--- /dev/null
+++ b/skills/flash-attention/SKILL.md
@@ -0,0 +1,367 @@
+---
+name: flash-attention
+description: Optimizes transformer attention with Flash Attention for 2-4x speedup and 10-20x memory reduction. Use when training/running transformers with long sequences (>512 tokens), encountering GPU memory issues with attention, or need faster inference. Supports PyTorch native SDPA, flash-attn library, H100 FP8, and sliding window attention.
+version: 1.0.0
+author: Orchestra Research
+license: MIT
+tags: [Optimization, Flash Attention, Attention Optimization, Memory Efficiency, Speed Optimization, Long Context, PyTorch, SDPA, H100, FP8, Transformers]
+dependencies: [flash-attn, torch, transformers]
+---
+
+# Flash Attention - Fast Memory-Efficient Attention
+
+## Quick start
+
+Flash Attention provides 2-4x speedup and 10-20x memory reduction for transformer attention through IO-aware tiling and recomputation.
+
+**PyTorch native (easiest, PyTorch 2.2+)**:
+```python
+import torch
+import torch.nn.functional as F
+
+q = torch.randn(2, 8, 512, 64, device='cuda', dtype=torch.float16) # [batch, heads, seq, dim]
+k = torch.randn(2, 8, 512, 64, device='cuda', dtype=torch.float16)
+v = torch.randn(2, 8, 512, 64, device='cuda', dtype=torch.float16)
+
+# Automatically uses Flash Attention if available
+out = F.scaled_dot_product_attention(q, k, v)
+```
+
+**flash-attn library (more features)**:
+```bash
+pip install flash-attn --no-build-isolation
+```
+
+```python
+from flash_attn import flash_attn_func
+
+# q, k, v: [batch, seqlen, nheads, headdim]
+out = flash_attn_func(q, k, v, dropout_p=0.0, causal=True)
+```
+
+## Common workflows
+
+### Workflow 1: Enable in existing PyTorch model
+
+Copy this checklist:
+
+```
+Flash Attention Integration:
+- [ ] Step 1: Check PyTorch version (≥2.2)
+- [ ] Step 2: Enable Flash Attention backend
+- [ ] Step 3: Verify speedup with profiling
+- [ ] Step 4: Test accuracy matches baseline
+```
+
+**Step 1: Check PyTorch version**
+
+```bash
+python -c "import torch; print(torch.__version__)"
+# Should be ≥2.2.0
+```
+
+If <2.2, upgrade:
+```bash
+pip install --upgrade torch
+```
+
+**Step 2: Enable Flash Attention backend**
+
+Replace standard attention:
+```python
+# Before (standard attention)
+attn_weights = torch.softmax(q @ k.transpose(-2, -1) / math.sqrt(d_k), dim=-1)
+out = attn_weights @ v
+
+# After (Flash Attention)
+import torch.nn.functional as F
+out = F.scaled_dot_product_attention(q, k, v, attn_mask=mask)
+```
+
+Force Flash Attention backend:
+```python
+with torch.backends.cuda.sdp_kernel(
+ enable_flash=True,
+ enable_math=False,
+ enable_mem_efficient=False
+):
+ out = F.scaled_dot_product_attention(q, k, v)
+```
+
+**Step 3: Verify speedup with profiling**
+
+```python
+import torch.utils.benchmark as benchmark
+
+def test_attention(use_flash):
+ q, k, v = [torch.randn(2, 8, 2048, 64, device='cuda', dtype=torch.float16) for _ in range(3)]
+
+ if use_flash:
+ with torch.backends.cuda.sdp_kernel(enable_flash=True):
+ return F.scaled_dot_product_attention(q, k, v)
+ else:
+ attn = (q @ k.transpose(-2, -1) / 8.0).softmax(dim=-1)
+ return attn @ v
+
+# Benchmark
+t_flash = benchmark.Timer(stmt='test_attention(True)', globals=globals())
+t_standard = benchmark.Timer(stmt='test_attention(False)', globals=globals())
+
+print(f"Flash: {t_flash.timeit(100).mean:.3f}s")
+print(f"Standard: {t_standard.timeit(100).mean:.3f}s")
+```
+
+Expected: 2-4x speedup for sequences >512 tokens.
+
+**Step 4: Test accuracy matches baseline**
+
+```python
+# Compare outputs
+q, k, v = [torch.randn(1, 8, 512, 64, device='cuda', dtype=torch.float16) for _ in range(3)]
+
+# Flash Attention
+out_flash = F.scaled_dot_product_attention(q, k, v)
+
+# Standard attention
+attn_weights = torch.softmax(q @ k.transpose(-2, -1) / 8.0, dim=-1)
+out_standard = attn_weights @ v
+
+# Check difference
+diff = (out_flash - out_standard).abs().max()
+print(f"Max difference: {diff:.6f}")
+# Should be <1e-3 for float16
+```
+
+### Workflow 2: Use flash-attn library for advanced features
+
+For multi-query attention, sliding window, or H100 FP8.
+
+Copy this checklist:
+
+```
+flash-attn Library Setup:
+- [ ] Step 1: Install flash-attn library
+- [ ] Step 2: Modify attention code
+- [ ] Step 3: Enable advanced features
+- [ ] Step 4: Benchmark performance
+```
+
+**Step 1: Install flash-attn library**
+
+```bash
+# NVIDIA GPUs (CUDA 12.0+)
+pip install flash-attn --no-build-isolation
+
+# Verify installation
+python -c "from flash_attn import flash_attn_func; print('Success')"
+```
+
+**Step 2: Modify attention code**
+
+```python
+from flash_attn import flash_attn_func
+
+# Input: [batch_size, seq_len, num_heads, head_dim]
+# Transpose from [batch, heads, seq, dim] if needed
+q = q.transpose(1, 2) # [batch, seq, heads, dim]
+k = k.transpose(1, 2)
+v = v.transpose(1, 2)
+
+out = flash_attn_func(
+ q, k, v,
+ dropout_p=0.1,
+ causal=True, # For autoregressive models
+ window_size=(-1, -1), # No sliding window
+ softmax_scale=None # Auto-scale
+)
+
+out = out.transpose(1, 2) # Back to [batch, heads, seq, dim]
+```
+
+**Step 3: Enable advanced features**
+
+Multi-query attention (shared K/V across heads):
+```python
+from flash_attn import flash_attn_func
+
+# q: [batch, seq, num_q_heads, dim]
+# k, v: [batch, seq, num_kv_heads, dim] # Fewer KV heads
+out = flash_attn_func(q, k, v) # Automatically handles MQA
+```
+
+Sliding window attention (local attention):
+```python
+# Only attend to window of 256 tokens before/after
+out = flash_attn_func(
+ q, k, v,
+ window_size=(256, 256), # (left, right) window
+ causal=True
+)
+```
+
+**Step 4: Benchmark performance**
+
+```python
+import torch
+from flash_attn import flash_attn_func
+import time
+
+q, k, v = [torch.randn(4, 4096, 32, 64, device='cuda', dtype=torch.float16) for _ in range(3)]
+
+# Warmup
+for _ in range(10):
+ _ = flash_attn_func(q, k, v)
+
+# Benchmark
+torch.cuda.synchronize()
+start = time.time()
+for _ in range(100):
+ out = flash_attn_func(q, k, v)
+ torch.cuda.synchronize()
+end = time.time()
+
+print(f"Time per iteration: {(end-start)/100*1000:.2f}ms")
+print(f"Memory allocated: {torch.cuda.max_memory_allocated()/1e9:.2f}GB")
+```
+
+### Workflow 3: H100 FP8 optimization (FlashAttention-3)
+
+For maximum performance on H100 GPUs.
+
+```
+FP8 Setup:
+- [ ] Step 1: Verify H100 GPU available
+- [ ] Step 2: Install flash-attn with FP8 support
+- [ ] Step 3: Convert inputs to FP8
+- [ ] Step 4: Run with FP8 attention
+```
+
+**Step 1: Verify H100 GPU**
+
+```bash
+nvidia-smi --query-gpu=name --format=csv
+# Should show "H100" or "H800"
+```
+
+**Step 2: Install flash-attn with FP8 support**
+
+```bash
+pip install flash-attn --no-build-isolation
+# FP8 support included for H100
+```
+
+**Step 3: Convert inputs to FP8**
+
+```python
+import torch
+
+q = torch.randn(2, 4096, 32, 64, device='cuda', dtype=torch.float16)
+k = torch.randn(2, 4096, 32, 64, device='cuda', dtype=torch.float16)
+v = torch.randn(2, 4096, 32, 64, device='cuda', dtype=torch.float16)
+
+# Convert to float8_e4m3 (FP8)
+q_fp8 = q.to(torch.float8_e4m3fn)
+k_fp8 = k.to(torch.float8_e4m3fn)
+v_fp8 = v.to(torch.float8_e4m3fn)
+```
+
+**Step 4: Run with FP8 attention**
+
+```python
+from flash_attn import flash_attn_func
+
+# FlashAttention-3 automatically uses FP8 kernels on H100
+out = flash_attn_func(q_fp8, k_fp8, v_fp8)
+# Result: ~1.2 PFLOPS, 1.5-2x faster than FP16
+```
+
+## When to use vs alternatives
+
+**Use Flash Attention when:**
+- Training transformers with sequences >512 tokens
+- Running inference with long context (>2K tokens)
+- GPU memory constrained (OOM with standard attention)
+- Need 2-4x speedup without accuracy loss
+- Using PyTorch 2.2+ or can install flash-attn
+
+**Use alternatives instead:**
+- **Standard attention**: Sequences <256 tokens (overhead not worth it)
+- **xFormers**: Need more attention variants (not just speed)
+- **Memory-efficient attention**: CPU inference (Flash Attention needs GPU)
+
+## Common issues
+
+**Issue: ImportError: cannot import flash_attn**
+
+Install with no-build-isolation flag:
+```bash
+pip install flash-attn --no-build-isolation
+```
+
+Or install CUDA toolkit first:
+```bash
+conda install cuda -c nvidia
+pip install flash-attn --no-build-isolation
+```
+
+**Issue: Slower than expected (no speedup)**
+
+Flash Attention benefits increase with sequence length:
+- <512 tokens: Minimal speedup (10-20%)
+- 512-2K tokens: 2-3x speedup
+- >2K tokens: 3-4x speedup
+
+Check sequence length is sufficient.
+
+**Issue: RuntimeError: CUDA error**
+
+Verify GPU supports Flash Attention:
+```python
+import torch
+print(torch.cuda.get_device_capability())
+# Should be ≥(7, 5) for Turing+
+```
+
+Flash Attention requires:
+- Ampere (A100, A10): ✅ Full support
+- Turing (T4): ✅ Supported
+- Volta (V100): ❌ Not supported
+
+**Issue: Accuracy degradation**
+
+Check dtype is float16 or bfloat16 (not float32):
+```python
+q = q.to(torch.float16) # Or torch.bfloat16
+```
+
+Flash Attention uses float16/bfloat16 for speed. Float32 not supported.
+
+## Advanced topics
+
+**Integration with HuggingFace Transformers**: See [references/transformers-integration.md](references/transformers-integration.md) for enabling Flash Attention in BERT, GPT, Llama models.
+
+**Performance benchmarks**: See [references/benchmarks.md](references/benchmarks.md) for detailed speed and memory comparisons across GPUs and sequence lengths.
+
+**Algorithm details**: See [references/algorithm.md](references/algorithm.md) for tiling strategy, recomputation, and IO complexity analysis.
+
+**Advanced features**: See [references/advanced-features.md](references/advanced-features.md) for rotary embeddings, ALiBi, paged KV cache, and custom attention masks.
+
+## Hardware requirements
+
+- **GPU**: NVIDIA Ampere+ (A100, A10, A30) or AMD MI200+
+- **VRAM**: Same as standard attention (Flash Attention doesn't increase memory)
+- **CUDA**: 12.0+ (11.8 minimum)
+- **PyTorch**: 2.2+ for native support
+
+**Not supported**: V100 (Volta), CPU inference
+
+## Resources
+
+- Paper: "FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness" (NeurIPS 2022)
+- Paper: "FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning" (ICLR 2024)
+- Blog: https://tridao.me/blog/2024/flash3/
+- GitHub: https://github.com/Dao-AILab/flash-attention
+- PyTorch docs: https://pytorch.org/docs/stable/generated/torch.nn.functional.scaled_dot_product_attention.html
+
+
+
diff --git a/skills/flash-attention/references/benchmarks.md b/skills/flash-attention/references/benchmarks.md
new file mode 100644
index 0000000..f798a6d
--- /dev/null
+++ b/skills/flash-attention/references/benchmarks.md
@@ -0,0 +1,215 @@
+# Performance Benchmarks
+
+## Contents
+- Speed comparisons across GPUs
+- Memory usage analysis
+- Scaling with sequence length
+- Training vs inference performance
+- Flash Attention versions comparison
+
+## Speed comparisons across GPUs
+
+### A100 80GB (Ampere)
+
+**Forward pass time** (milliseconds, batch=8, heads=32, dim=64):
+
+| Seq Length | Standard | Flash Attn 2 | Flash Attn 3 | Speedup (FA2) |
+|------------|----------|--------------|--------------|---------------|
+| 512 | 1.2 | 0.9 | N/A | 1.3x |
+| 1024 | 3.8 | 1.4 | N/A | 2.7x |
+| 2048 | 14.2 | 4.8 | N/A | 3.0x |
+| 4096 | 55.1 | 17.3 | N/A | 3.2x |
+| 8192 | 218.5 | 66.2 | N/A | 3.3x |
+
+### H100 80GB (Hopper)
+
+**Forward pass time** (milliseconds, same config):
+
+| Seq Length | Standard | Flash Attn 2 | Flash Attn 3 (FP16) | Flash Attn 3 (FP8) | Best Speedup |
+|------------|----------|--------------|---------------------|--------------------|--------------|
+| 512 | 0.8 | 0.6 | 0.4 | 0.3 | 2.7x |
+| 1024 | 2.6 | 1.0 | 0.6 | 0.4 | 6.5x |
+| 2048 | 9.8 | 3.4 | 2.0 | 1.3 | 7.5x |
+| 4096 | 38.2 | 12.5 | 7.2 | 4.8 | 8.0x |
+| 8192 | 151.4 | 47.8 | 27.1 | 18.2 | 8.3x |
+
+**Key insight**: Flash Attention 3 on H100 with FP8 achieves ~1.2 PFLOPS (75% of theoretical max).
+
+### A10G 24GB (Ampere)
+
+**Forward pass time** (milliseconds, batch=4):
+
+| Seq Length | Standard | Flash Attn 2 | Speedup |
+|------------|----------|--------------|---------|
+| 512 | 2.1 | 1.6 | 1.3x |
+| 1024 | 6.8 | 2.8 | 2.4x |
+| 2048 | 25.9 | 9.4 | 2.8x |
+| 4096 | 102.1 | 35.2 | 2.9x |
+
+## Memory usage analysis
+
+### GPU memory consumption (batch=8, heads=32, dim=64)
+
+**Standard attention memory**:
+
+| Seq Length | Attention Matrix | KV Cache | Total | Notes |
+|------------|------------------|----------|-------|-------|
+| 512 | 8 MB | 32 MB | 40 MB | Manageable |
+| 2048 | 128 MB | 128 MB | 256 MB | Growing |
+| 8192 | 2048 MB (2 GB) | 512 MB | 2.5 GB | Large |
+| 32768 | 32768 MB (32 GB) | 2048 MB | 34 GB | OOM on 24GB GPUs |
+
+**Flash Attention 2 memory**:
+
+| Seq Length | Attention (on-chip) | KV Cache | Total | Reduction |
+|------------|---------------------|----------|-------|-----------|
+| 512 | 0 MB (recomputed) | 32 MB | 32 MB | 20% |
+| 2048 | 0 MB | 128 MB | 128 MB | 50% |
+| 8192 | 0 MB | 512 MB | 512 MB | 80% |
+| 32768 | 0 MB | 2048 MB | 2 GB | 94% |
+
+**Key insight**: Flash Attention doesn't materialize attention matrix, saving O(N²) memory.
+
+### Memory scaling comparison
+
+**Llama 2 7B model memory** (float16, batch=1):
+
+| Context Length | Standard Attention | Flash Attention 2 | Can Fit 24GB GPU? |
+|----------------|-------------------|-------------------|-------------------|
+| 2K | 3.2 GB | 2.1 GB | Both: Yes |
+| 4K | 5.8 GB | 2.8 GB | Both: Yes |
+| 8K | 12.1 GB | 4.2 GB | Both: Yes |
+| 16K | 26.3 GB (OOM) | 7.8 GB | Only Flash: Yes |
+| 32K | OOM | 14.2 GB | Only Flash: Yes |
+
+### Training memory (Llama 2 7B, batch=4)
+
+| Context | Standard (GB) | Flash Attn (GB) | Reduction |
+|---------|---------------|-----------------|-----------|
+| 2K | 18.2 | 12.4 | 32% |
+| 4K | 34.8 | 16.8 | 52% |
+| 8K | OOM (>40GB) | 26.2 | Fits! |
+
+## Scaling with sequence length
+
+### Computational complexity
+
+**Standard attention**:
+- Time: O(N² × d)
+- Memory: O(N² + N × d)
+
+**Flash Attention**:
+- Time: O(N² × d) (same, but with better constants)
+- Memory: O(N × d) (linear!)
+
+### Empirical scaling (A100, batch=1, heads=32, dim=64)
+
+**Time per token (milliseconds)**:
+
+| Sequence | 512 | 1K | 2K | 4K | 8K | 16K |
+|----------|-----|-----|-----|-----|-----|------|
+| Standard | 0.15 | 0.37 | 1.11 | 3.44 | 13.4 | 52.8 |
+| Flash Attn 2 | 0.11 | 0.14 | 0.24 | 0.43 | 0.83 | 1.64 |
+| Speedup | 1.4x | 2.6x | 4.6x | 8.0x | 16.1x | 32.2x |
+
+**Observation**: Speedup increases quadratically with sequence length!
+
+### Memory per token (MB)
+
+| Sequence | 512 | 1K | 2K | 4K | 8K | 16K |
+|----------|-----|-----|-----|-----|-----|------|
+| Standard | 0.08 | 0.13 | 0.25 | 0.64 | 2.05 | 8.13 |
+| Flash Attn 2 | 0.06 | 0.06 | 0.06 | 0.06 | 0.06 | 0.06 |
+
+**Observation**: Flash Attention memory per token is constant!
+
+## Training vs inference performance
+
+### Training (forward + backward, Llama 2 7B, A100)
+
+| Batch × Seq | Standard (samples/sec) | Flash Attn (samples/sec) | Speedup |
+|-------------|------------------------|--------------------------|---------|
+| 4 × 2K | 1.2 | 3.1 | 2.6x |
+| 8 × 2K | 2.1 | 5.8 | 2.8x |
+| 4 × 4K | 0.4 | 1.3 | 3.3x |
+| 8 × 4K | OOM | 2.4 | Enabled |
+| 2 × 8K | 0.1 | 0.4 | 4.0x |
+
+### Inference (generation, Llama 2 7B, A100)
+
+| Context Length | Standard (tokens/sec) | Flash Attn (tokens/sec) | Speedup |
+|----------------|----------------------|-------------------------|---------|
+| 512 | 48 | 52 | 1.1x |
+| 2K | 42 | 62 | 1.5x |
+| 4K | 31 | 58 | 1.9x |
+| 8K | 18 | 51 | 2.8x |
+| 16K | OOM | 42 | Enabled |
+
+**Note**: Inference speedup less dramatic than training because generation is memory-bound (KV cache accesses).
+
+## Flash Attention versions comparison
+
+### Flash Attention 1 vs 2 vs 3 (H100, seq=4096, batch=8)
+
+| Metric | FA1 | FA2 | FA3 (FP16) | FA3 (FP8) |
+|--------|-----|-----|------------|-----------|
+| Forward time (ms) | 28.4 | 12.5 | 7.2 | 4.8 |
+| Memory (GB) | 4.8 | 4.2 | 4.2 | 2.8 |
+| TFLOPS | 180 | 420 | 740 | 1150 |
+| GPU util % | 35% | 55% | 75% | 82% |
+
+**Key improvements**:
+- FA2: 2.3x faster than FA1 (better parallelism)
+- FA3 (FP16): 1.7x faster than FA2 (H100 async optimizations)
+- FA3 (FP8): 2.6x faster than FA2 (low precision)
+
+### Features by version
+
+| Feature | FA1 | FA2 | FA3 |
+|---------|-----|-----|-----|
+| Basic attention | ✅ | ✅ | ✅ |
+| Causal masking | ✅ | ✅ | ✅ |
+| Multi-query attention | ❌ | ✅ | ✅ |
+| Sliding window | ❌ | ✅ | ✅ |
+| Paged KV cache | ❌ | ✅ | ✅ |
+| FP8 support | ❌ | ❌ | ✅ (H100 only) |
+| Work partitioning | Basic | Advanced | Optimal |
+
+## Real-world model benchmarks
+
+### Llama 2 models (A100 80GB, batch=4, seq=2048)
+
+| Model | Params | Standard (samples/sec) | Flash Attn (samples/sec) | Speedup |
+|-------|--------|------------------------|--------------------------|---------|
+| Llama 2 7B | 7B | 1.2 | 3.1 | 2.6x |
+| Llama 2 13B | 13B | 0.6 | 1.7 | 2.8x |
+| Llama 2 70B | 70B | 0.12 | 0.34 | 2.8x |
+
+### GPT-style models (seq=1024)
+
+| Model | Standard (tokens/sec) | Flash Attn (tokens/sec) | Speedup |
+|-------|----------------------|-------------------------|---------|
+| GPT-2 (124M) | 520 | 680 | 1.3x |
+| GPT-J (6B) | 42 | 98 | 2.3x |
+| GPT-NeoX (20B) | 8 | 22 | 2.75x |
+
+## Recommendations by use case
+
+**Training large models (>7B parameters)**:
+- Use Flash Attention 2 on A100
+- Use Flash Attention 3 FP8 on H100 for maximum speed
+- Expected: 2.5-3x speedup
+
+**Long context inference (>4K tokens)**:
+- Flash Attention essential (enables contexts standard attention can't handle)
+- Expected: 2-4x speedup, 5-10x memory reduction
+
+**Short sequences (<512 tokens)**:
+- Flash Attention provides 1.2-1.5x speedup
+- Minimal memory benefit
+- Still worth enabling (no downside)
+
+**Multi-user serving**:
+- Flash Attention reduces per-request memory
+- Allows higher concurrent batch sizes
+- Can serve 2-3x more users on same hardware
diff --git a/skills/flash-attention/references/transformers-integration.md b/skills/flash-attention/references/transformers-integration.md
new file mode 100644
index 0000000..4873675
--- /dev/null
+++ b/skills/flash-attention/references/transformers-integration.md
@@ -0,0 +1,293 @@
+# HuggingFace Transformers Integration
+
+## Contents
+- Enabling Flash Attention in Transformers
+- Supported model architectures
+- Configuration examples
+- Performance comparisons
+- Troubleshooting model-specific issues
+
+## Enabling Flash Attention in Transformers
+
+HuggingFace Transformers (v4.36+) supports Flash Attention 2 natively.
+
+**Simple enable for any supported model**:
+```python
+from transformers import AutoModel
+
+model = AutoModel.from_pretrained(
+ "meta-llama/Llama-2-7b-hf",
+ attn_implementation="flash_attention_2",
+ torch_dtype=torch.float16,
+ device_map="auto"
+)
+```
+
+**Install requirements**:
+```bash
+pip install transformers>=4.36
+pip install flash-attn --no-build-isolation
+```
+
+## Supported model architectures
+
+As of Transformers 4.40:
+
+**Fully supported**:
+- Llama / Llama 2 / Llama 3
+- Mistral / Mixtral
+- Falcon
+- GPT-NeoX
+- Phi / Phi-2 / Phi-3
+- Qwen / Qwen2
+- Gemma
+- Starcoder2
+- GPT-J
+- OPT
+- BLOOM
+
+**Partially supported** (encoder-decoder):
+- BART
+- T5 / Flan-T5
+- Whisper
+
+**Check support**:
+```python
+from transformers import AutoConfig
+
+config = AutoConfig.from_pretrained("model-name")
+print(config._attn_implementation_internal)
+# 'flash_attention_2' if supported
+```
+
+## Configuration examples
+
+### Llama 2 with Flash Attention
+
+```python
+from transformers import AutoModelForCausalLM, AutoTokenizer
+import torch
+
+model_id = "meta-llama/Llama-2-7b-hf"
+
+model = AutoModelForCausalLM.from_pretrained(
+ model_id,
+ attn_implementation="flash_attention_2",
+ torch_dtype=torch.float16,
+ device_map="auto"
+)
+
+tokenizer = AutoTokenizer.from_pretrained(model_id)
+
+# Generate
+inputs = tokenizer("Once upon a time", return_tensors="pt").to("cuda")
+outputs = model.generate(**inputs, max_length=100)
+print(tokenizer.decode(outputs[0]))
+```
+
+### Mistral with Flash Attention for long context
+
+```python
+from transformers import AutoModelForCausalLM
+import torch
+
+model = AutoModelForCausalLM.from_pretrained(
+ "mistralai/Mistral-7B-v0.1",
+ attn_implementation="flash_attention_2",
+ torch_dtype=torch.bfloat16, # Better for long context
+ device_map="auto",
+ max_position_embeddings=32768 # Extended context
+)
+
+# Process long document (32K tokens)
+long_text = "..." * 10000
+inputs = tokenizer(long_text, return_tensors="pt", truncation=False).to("cuda")
+outputs = model.generate(**inputs, max_new_tokens=512)
+```
+
+### Fine-tuning with Flash Attention
+
+```python
+from transformers import Trainer, TrainingArguments
+from transformers import AutoModelForCausalLM
+
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-7b-hf",
+ attn_implementation="flash_attention_2",
+ torch_dtype=torch.float16
+)
+
+training_args = TrainingArguments(
+ output_dir="./results",
+ per_device_train_batch_size=4,
+ gradient_accumulation_steps=4,
+ num_train_epochs=3,
+ fp16=True, # Must match model dtype
+ optim="adamw_torch_fused" # Fast optimizer
+)
+
+trainer = Trainer(
+ model=model,
+ args=training_args,
+ train_dataset=train_dataset
+)
+
+trainer.train()
+```
+
+### Multi-GPU training
+
+```python
+from transformers import AutoModelForCausalLM
+import torch
+
+# Model parallelism with Flash Attention
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-13b-hf",
+ attn_implementation="flash_attention_2",
+ torch_dtype=torch.float16,
+ device_map="auto", # Automatic multi-GPU placement
+ max_memory={0: "20GB", 1: "20GB"} # Limit per GPU
+)
+```
+
+## Performance comparisons
+
+### Memory usage (Llama 2 7B, batch=1)
+
+| Sequence Length | Standard Attention | Flash Attention 2 | Reduction |
+|-----------------|-------------------|-------------------|-----------|
+| 512 | 1.2 GB | 0.9 GB | 25% |
+| 2048 | 3.8 GB | 1.4 GB | 63% |
+| 8192 | 14.2 GB | 3.2 GB | 77% |
+| 32768 | OOM (>24GB) | 10.8 GB | Fits! |
+
+### Speed (tokens/sec, A100 80GB)
+
+| Model | Standard | Flash Attn 2 | Speedup |
+|-------|----------|--------------|---------|
+| Llama 2 7B (seq=2048) | 42 | 118 | 2.8x |
+| Llama 2 13B (seq=4096) | 18 | 52 | 2.9x |
+| Llama 2 70B (seq=2048) | 4 | 11 | 2.75x |
+
+### Training throughput (samples/sec)
+
+| Model | Batch Size | Standard | Flash Attn 2 | Speedup |
+|-------|------------|----------|--------------|---------|
+| Llama 2 7B | 4 | 1.2 | 3.1 | 2.6x |
+| Llama 2 7B | 8 | 2.1 | 5.8 | 2.8x |
+| Llama 2 13B | 2 | 0.6 | 1.7 | 2.8x |
+
+## Troubleshooting model-specific issues
+
+### Issue: Model doesn't support Flash Attention
+
+Check support list above. If not supported, use PyTorch SDPA as fallback:
+
+```python
+model = AutoModelForCausalLM.from_pretrained(
+ "model-name",
+ attn_implementation="sdpa", # PyTorch native (still faster)
+ torch_dtype=torch.float16
+)
+```
+
+### Issue: CUDA out of memory during loading
+
+Reduce memory footprint:
+
+```python
+model = AutoModelForCausalLM.from_pretrained(
+ "model-name",
+ attn_implementation="flash_attention_2",
+ torch_dtype=torch.float16,
+ device_map="auto",
+ max_memory={0: "18GB"}, # Reserve memory for KV cache
+ low_cpu_mem_usage=True
+)
+```
+
+### Issue: Slower inference than expected
+
+Ensure dtype matches:
+
+```python
+# Model and inputs must both be float16/bfloat16
+model = model.to(torch.float16)
+inputs = tokenizer(..., return_tensors="pt").to("cuda")
+inputs = {k: v.to(torch.float16) if v.dtype == torch.float32 else v
+ for k, v in inputs.items()}
+```
+
+### Issue: Different outputs vs standard attention
+
+Flash Attention is numerically equivalent but uses different computation order. Small differences (<1e-3) are normal:
+
+```python
+# Compare outputs
+model_standard = AutoModelForCausalLM.from_pretrained("model-name", torch_dtype=torch.float16)
+model_flash = AutoModelForCausalLM.from_pretrained(
+ "model-name",
+ attn_implementation="flash_attention_2",
+ torch_dtype=torch.float16
+)
+
+inputs = tokenizer("Test", return_tensors="pt").to("cuda")
+
+with torch.no_grad():
+ out_standard = model_standard(**inputs).logits
+ out_flash = model_flash(**inputs).logits
+
+diff = (out_standard - out_flash).abs().max()
+print(f"Max diff: {diff:.6f}") # Should be ~1e-3 to 1e-4
+```
+
+### Issue: ImportError during model loading
+
+Install flash-attn:
+```bash
+pip install flash-attn --no-build-isolation
+```
+
+Or disable Flash Attention:
+```python
+model = AutoModelForCausalLM.from_pretrained(
+ "model-name",
+ attn_implementation="eager", # Standard PyTorch
+ torch_dtype=torch.float16
+)
+```
+
+## Best practices
+
+1. **Always use float16/bfloat16** with Flash Attention (not float32)
+2. **Set device_map="auto"** for automatic memory management
+3. **Use bfloat16 for long context** (better numerical stability)
+4. **Enable gradient checkpointing** for training large models
+5. **Monitor memory** with `torch.cuda.max_memory_allocated()`
+
+**Example with all best practices**:
+```python
+from transformers import AutoModelForCausalLM, TrainingArguments
+
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-2-7b-hf",
+ attn_implementation="flash_attention_2",
+ torch_dtype=torch.bfloat16, # Better for training
+ device_map="auto",
+ low_cpu_mem_usage=True
+)
+
+# Enable gradient checkpointing for memory
+model.gradient_checkpointing_enable()
+
+# Training with optimizations
+training_args = TrainingArguments(
+ output_dir="./results",
+ per_device_train_batch_size=8,
+ gradient_accumulation_steps=2,
+ bf16=True, # Match model dtype
+ optim="adamw_torch_fused",
+ gradient_checkpointing=True
+)
+```
diff --git a/skills/langgraph-docs/SKILL.md b/skills/langgraph-docs/SKILL.md
new file mode 100644
index 0000000..360a17f
--- /dev/null
+++ b/skills/langgraph-docs/SKILL.md
@@ -0,0 +1,36 @@
+---
+name: langgraph-docs
+description: Use this skill for requests related to LangGraph in order to fetch relevant documentation to provide accurate, up-to-date guidance.
+---
+
+# langgraph-docs
+
+## Overview
+
+This skill explains how to access LangGraph Python documentation to help answer questions and guide implementation.
+
+## Instructions
+
+### 1. Fetch the Documentation Index
+
+Use the fetch_url tool to read the following URL:
+https://docs.langchain.com/llms.txt
+
+This provides a structured list of all available documentation with descriptions.
+
+### 2. Select Relevant Documentation
+
+Based on the question, identify 2-4 most relevant documentation URLs from the index. Prioritize:
+
+- Specific how-to guides for implementation questions
+- Core concept pages for understanding questions
+- Tutorials for end-to-end examples
+- Reference docs for API details
+
+### 3. Fetch Selected Documentation
+
+Use the fetch_url tool to read the selected documentation URLs.
+
+### 4. Provide Accurate Guidance
+
+After reading the documentation, complete the user's request.
\ No newline at end of file
diff --git a/skills/llama-cpp/SKILL.md b/skills/llama-cpp/SKILL.md
new file mode 100644
index 0000000..ed41a5d
--- /dev/null
+++ b/skills/llama-cpp/SKILL.md
@@ -0,0 +1,258 @@
+---
+name: llama-cpp
+description: Runs LLM inference on CPU, Apple Silicon, and consumer GPUs without NVIDIA hardware. Use for edge deployment, M1/M2/M3 Macs, AMD/Intel GPUs, or when CUDA is unavailable. Supports GGUF quantization (1.5-8 bit) for reduced memory and 4-10× speedup vs PyTorch on CPU.
+version: 1.0.0
+author: Orchestra Research
+license: MIT
+tags: [Inference Serving, Llama.cpp, CPU Inference, Apple Silicon, Edge Deployment, GGUF, Quantization, Non-NVIDIA, AMD GPUs, Intel GPUs, Embedded]
+dependencies: [llama-cpp-python]
+---
+
+# llama.cpp
+
+Pure C/C++ LLM inference with minimal dependencies, optimized for CPUs and non-NVIDIA hardware.
+
+## When to use llama.cpp
+
+**Use llama.cpp when:**
+- Running on CPU-only machines
+- Deploying on Apple Silicon (M1/M2/M3/M4)
+- Using AMD or Intel GPUs (no CUDA)
+- Edge deployment (Raspberry Pi, embedded systems)
+- Need simple deployment without Docker/Python
+
+**Use TensorRT-LLM instead when:**
+- Have NVIDIA GPUs (A100/H100)
+- Need maximum throughput (100K+ tok/s)
+- Running in datacenter with CUDA
+
+**Use vLLM instead when:**
+- Have NVIDIA GPUs
+- Need Python-first API
+- Want PagedAttention
+
+## Quick start
+
+### Installation
+
+```bash
+# macOS/Linux
+brew install llama.cpp
+
+# Or build from source
+git clone https://github.com/ggerganov/llama.cpp
+cd llama.cpp
+make
+
+# With Metal (Apple Silicon)
+make LLAMA_METAL=1
+
+# With CUDA (NVIDIA)
+make LLAMA_CUDA=1
+
+# With ROCm (AMD)
+make LLAMA_HIP=1
+```
+
+### Download model
+
+```bash
+# Download from HuggingFace (GGUF format)
+huggingface-cli download \
+ TheBloke/Llama-2-7B-Chat-GGUF \
+ llama-2-7b-chat.Q4_K_M.gguf \
+ --local-dir models/
+
+# Or convert from HuggingFace
+python convert_hf_to_gguf.py models/llama-2-7b-chat/
+```
+
+### Run inference
+
+```bash
+# Simple chat
+./llama-cli \
+ -m models/llama-2-7b-chat.Q4_K_M.gguf \
+ -p "Explain quantum computing" \
+ -n 256 # Max tokens
+
+# Interactive chat
+./llama-cli \
+ -m models/llama-2-7b-chat.Q4_K_M.gguf \
+ --interactive
+```
+
+### Server mode
+
+```bash
+# Start OpenAI-compatible server
+./llama-server \
+ -m models/llama-2-7b-chat.Q4_K_M.gguf \
+ --host 0.0.0.0 \
+ --port 8080 \
+ -ngl 32 # Offload 32 layers to GPU
+
+# Client request
+curl http://localhost:8080/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -d '{
+ "model": "llama-2-7b-chat",
+ "messages": [{"role": "user", "content": "Hello!"}],
+ "temperature": 0.7,
+ "max_tokens": 100
+ }'
+```
+
+## Quantization formats
+
+### GGUF format overview
+
+| Format | Bits | Size (7B) | Speed | Quality | Use Case |
+|--------|------|-----------|-------|---------|----------|
+| **Q4_K_M** | 4.5 | 4.1 GB | Fast | Good | **Recommended default** |
+| Q4_K_S | 4.3 | 3.9 GB | Faster | Lower | Speed critical |
+| Q5_K_M | 5.5 | 4.8 GB | Medium | Better | Quality critical |
+| Q6_K | 6.5 | 5.5 GB | Slower | Best | Maximum quality |
+| Q8_0 | 8.0 | 7.0 GB | Slow | Excellent | Minimal degradation |
+| Q2_K | 2.5 | 2.7 GB | Fastest | Poor | Testing only |
+
+### Choosing quantization
+
+```bash
+# General use (balanced)
+Q4_K_M # 4-bit, medium quality
+
+# Maximum speed (more degradation)
+Q2_K or Q3_K_M
+
+# Maximum quality (slower)
+Q6_K or Q8_0
+
+# Very large models (70B, 405B)
+Q3_K_M or Q4_K_S # Lower bits to fit in memory
+```
+
+## Hardware acceleration
+
+### Apple Silicon (Metal)
+
+```bash
+# Build with Metal
+make LLAMA_METAL=1
+
+# Run with GPU acceleration (automatic)
+./llama-cli -m model.gguf -ngl 999 # Offload all layers
+
+# Performance: M3 Max 40-60 tokens/sec (Llama 2-7B Q4_K_M)
+```
+
+### NVIDIA GPUs (CUDA)
+
+```bash
+# Build with CUDA
+make LLAMA_CUDA=1
+
+# Offload layers to GPU
+./llama-cli -m model.gguf -ngl 35 # Offload 35/40 layers
+
+# Hybrid CPU+GPU for large models
+./llama-cli -m llama-70b.Q4_K_M.gguf -ngl 20 # GPU: 20 layers, CPU: rest
+```
+
+### AMD GPUs (ROCm)
+
+```bash
+# Build with ROCm
+make LLAMA_HIP=1
+
+# Run with AMD GPU
+./llama-cli -m model.gguf -ngl 999
+```
+
+## Common patterns
+
+### Batch processing
+
+```bash
+# Process multiple prompts from file
+cat prompts.txt | ./llama-cli \
+ -m model.gguf \
+ --batch-size 512 \
+ -n 100
+```
+
+### Constrained generation
+
+```bash
+# JSON output with grammar
+./llama-cli \
+ -m model.gguf \
+ -p "Generate a person: " \
+ --grammar-file grammars/json.gbnf
+
+# Outputs valid JSON only
+```
+
+### Context size
+
+```bash
+# Increase context (default 512)
+./llama-cli \
+ -m model.gguf \
+ -c 4096 # 4K context window
+
+# Very long context (if model supports)
+./llama-cli -m model.gguf -c 32768 # 32K context
+```
+
+## Performance benchmarks
+
+### CPU performance (Llama 2-7B Q4_K_M)
+
+| CPU | Threads | Speed | Cost |
+|-----|---------|-------|------|
+| Apple M3 Max | 16 | 50 tok/s | $0 (local) |
+| AMD Ryzen 9 7950X | 32 | 35 tok/s | $0.50/hour |
+| Intel i9-13900K | 32 | 30 tok/s | $0.40/hour |
+| AWS c7i.16xlarge | 64 | 40 tok/s | $2.88/hour |
+
+### GPU acceleration (Llama 2-7B Q4_K_M)
+
+| GPU | Speed | vs CPU | Cost |
+|-----|-------|--------|------|
+| NVIDIA RTX 4090 | 120 tok/s | 3-4× | $0 (local) |
+| NVIDIA A10 | 80 tok/s | 2-3× | $1.00/hour |
+| AMD MI250 | 70 tok/s | 2× | $2.00/hour |
+| Apple M3 Max (Metal) | 50 tok/s | ~Same | $0 (local) |
+
+## Supported models
+
+**LLaMA family**:
+- Llama 2 (7B, 13B, 70B)
+- Llama 3 (8B, 70B, 405B)
+- Code Llama
+
+**Mistral family**:
+- Mistral 7B
+- Mixtral 8x7B, 8x22B
+
+**Other**:
+- Falcon, BLOOM, GPT-J
+- Phi-3, Gemma, Qwen
+- LLaVA (vision), Whisper (audio)
+
+**Find models**: https://huggingface.co/models?library=gguf
+
+## References
+
+- **[Quantization Guide](references/quantization.md)** - GGUF formats, conversion, quality comparison
+- **[Server Deployment](references/server.md)** - API endpoints, Docker, monitoring
+- **[Optimization](references/optimization.md)** - Performance tuning, hybrid CPU+GPU
+
+## Resources
+
+- **GitHub**: https://github.com/ggerganov/llama.cpp
+- **Models**: https://huggingface.co/models?library=gguf
+- **Discord**: https://discord.gg/llama-cpp
+
+
diff --git a/skills/llama-cpp/references/optimization.md b/skills/llama-cpp/references/optimization.md
new file mode 100644
index 0000000..dbe870c
--- /dev/null
+++ b/skills/llama-cpp/references/optimization.md
@@ -0,0 +1,89 @@
+# Performance Optimization Guide
+
+Maximize llama.cpp inference speed and efficiency.
+
+## CPU Optimization
+
+### Thread tuning
+```bash
+# Set threads (default: physical cores)
+./llama-cli -m model.gguf -t 8
+
+# For AMD Ryzen 9 7950X (16 cores, 32 threads)
+-t 16 # Best: physical cores
+
+# Avoid hyperthreading (slower for matrix ops)
+```
+
+### BLAS acceleration
+```bash
+# OpenBLAS (faster matrix ops)
+make LLAMA_OPENBLAS=1
+
+# BLAS gives 2-3× speedup
+```
+
+## GPU Offloading
+
+### Layer offloading
+```bash
+# Offload 35 layers to GPU (hybrid mode)
+./llama-cli -m model.gguf -ngl 35
+
+# Offload all layers
+./llama-cli -m model.gguf -ngl 999
+
+# Find optimal value:
+# Start with -ngl 999
+# If OOM, reduce by 5 until fits
+```
+
+### Memory usage
+```bash
+# Check VRAM usage
+nvidia-smi dmon
+
+# Reduce context if needed
+./llama-cli -m model.gguf -c 2048 # 2K context instead of 4K
+```
+
+## Batch Processing
+
+```bash
+# Increase batch size for throughput
+./llama-cli -m model.gguf -b 512 # Default: 512
+
+# Physical batch (GPU)
+--ubatch 128 # Process 128 tokens at once
+```
+
+## Context Management
+
+```bash
+# Default context (512 tokens)
+-c 512
+
+# Longer context (slower, more memory)
+-c 4096
+
+# Very long context (if model supports)
+-c 32768
+```
+
+## Benchmarks
+
+### CPU Performance (Llama 2-7B Q4_K_M)
+
+| Setup | Speed | Notes |
+|-------|-------|-------|
+| Apple M3 Max | 50 tok/s | Metal acceleration |
+| AMD 7950X (16c) | 35 tok/s | OpenBLAS |
+| Intel i9-13900K | 30 tok/s | AVX2 |
+
+### GPU Offloading (RTX 4090)
+
+| Layers GPU | Speed | VRAM |
+|------------|-------|------|
+| 0 (CPU only) | 30 tok/s | 0 GB |
+| 20 (hybrid) | 80 tok/s | 8 GB |
+| 35 (all) | 120 tok/s | 12 GB |
diff --git a/skills/llama-cpp/references/quantization.md b/skills/llama-cpp/references/quantization.md
new file mode 100644
index 0000000..8620463
--- /dev/null
+++ b/skills/llama-cpp/references/quantization.md
@@ -0,0 +1,213 @@
+# GGUF Quantization Guide
+
+Complete guide to GGUF quantization formats and model conversion.
+
+## Quantization Overview
+
+**GGUF** (GPT-Generated Unified Format) - Standard format for llama.cpp models.
+
+### Format Comparison
+
+| Format | Perplexity | Size (7B) | Tokens/sec | Notes |
+|--------|------------|-----------|------------|-------|
+| FP16 | 5.9565 (baseline) | 13.0 GB | 15 tok/s | Original quality |
+| Q8_0 | 5.9584 (+0.03%) | 7.0 GB | 25 tok/s | Nearly lossless |
+| **Q6_K** | 5.9642 (+0.13%) | 5.5 GB | 30 tok/s | Best quality/size |
+| **Q5_K_M** | 5.9796 (+0.39%) | 4.8 GB | 35 tok/s | Balanced |
+| **Q4_K_M** | 6.0565 (+1.68%) | 4.1 GB | 40 tok/s | **Recommended** |
+| Q4_K_S | 6.1125 (+2.62%) | 3.9 GB | 42 tok/s | Faster, lower quality |
+| Q3_K_M | 6.3184 (+6.07%) | 3.3 GB | 45 tok/s | Small models only |
+| Q2_K | 6.8673 (+15.3%) | 2.7 GB | 50 tok/s | Not recommended |
+
+**Recommendation**: Use **Q4_K_M** for best balance of quality and speed.
+
+## Converting Models
+
+### HuggingFace to GGUF
+
+```bash
+# 1. Download HuggingFace model
+huggingface-cli download meta-llama/Llama-2-7b-chat-hf \
+ --local-dir models/llama-2-7b-chat/
+
+# 2. Convert to FP16 GGUF
+python convert_hf_to_gguf.py \
+ models/llama-2-7b-chat/ \
+ --outtype f16 \
+ --outfile models/llama-2-7b-chat-f16.gguf
+
+# 3. Quantize to Q4_K_M
+./llama-quantize \
+ models/llama-2-7b-chat-f16.gguf \
+ models/llama-2-7b-chat-Q4_K_M.gguf \
+ Q4_K_M
+```
+
+### Batch quantization
+
+```bash
+# Quantize to multiple formats
+for quant in Q4_K_M Q5_K_M Q6_K Q8_0; do
+ ./llama-quantize \
+ model-f16.gguf \
+ model-${quant}.gguf \
+ $quant
+done
+```
+
+## K-Quantization Methods
+
+**K-quants** use mixed precision for better quality:
+- Attention weights: Higher precision
+- Feed-forward weights: Lower precision
+
+**Variants**:
+- `_S` (Small): Faster, lower quality
+- `_M` (Medium): Balanced (recommended)
+- `_L` (Large): Better quality, larger size
+
+**Example**: `Q4_K_M`
+- `Q4`: 4-bit quantization
+- `K`: Mixed precision method
+- `M`: Medium quality
+
+## Quality Testing
+
+```bash
+# Calculate perplexity (quality metric)
+./llama-perplexity \
+ -m model.gguf \
+ -f wikitext-2-raw/wiki.test.raw \
+ -c 512
+
+# Lower perplexity = better quality
+# Baseline (FP16): ~5.96
+# Q4_K_M: ~6.06 (+1.7%)
+# Q2_K: ~6.87 (+15.3% - too much degradation)
+```
+
+## Use Case Guide
+
+### General purpose (chatbots, assistants)
+```
+Q4_K_M - Best balance
+Q5_K_M - If you have extra RAM
+```
+
+### Code generation
+```
+Q5_K_M or Q6_K - Higher precision helps with code
+```
+
+### Creative writing
+```
+Q4_K_M - Sufficient quality
+Q3_K_M - Acceptable for draft generation
+```
+
+### Technical/medical
+```
+Q6_K or Q8_0 - Maximum accuracy
+```
+
+### Edge devices (Raspberry Pi)
+```
+Q2_K or Q3_K_S - Fit in limited RAM
+```
+
+## Model Size Scaling
+
+### 7B parameter models
+
+| Format | Size | RAM needed |
+|--------|------|------------|
+| Q2_K | 2.7 GB | 5 GB |
+| Q3_K_M | 3.3 GB | 6 GB |
+| Q4_K_M | 4.1 GB | 7 GB |
+| Q5_K_M | 4.8 GB | 8 GB |
+| Q6_K | 5.5 GB | 9 GB |
+| Q8_0 | 7.0 GB | 11 GB |
+
+### 13B parameter models
+
+| Format | Size | RAM needed |
+|--------|------|------------|
+| Q2_K | 5.1 GB | 8 GB |
+| Q3_K_M | 6.2 GB | 10 GB |
+| Q4_K_M | 7.9 GB | 12 GB |
+| Q5_K_M | 9.2 GB | 14 GB |
+| Q6_K | 10.7 GB | 16 GB |
+
+### 70B parameter models
+
+| Format | Size | RAM needed |
+|--------|------|------------|
+| Q2_K | 26 GB | 32 GB |
+| Q3_K_M | 32 GB | 40 GB |
+| Q4_K_M | 41 GB | 48 GB |
+| Q4_K_S | 39 GB | 46 GB |
+| Q5_K_M | 48 GB | 56 GB |
+
+**Recommendation for 70B**: Use Q3_K_M or Q4_K_S to fit in consumer hardware.
+
+## Finding Pre-Quantized Models
+
+**TheBloke** on HuggingFace:
+- https://huggingface.co/TheBloke
+- Most models available in all GGUF formats
+- No conversion needed
+
+**Example**:
+```bash
+# Download pre-quantized Llama 2-7B
+huggingface-cli download \
+ TheBloke/Llama-2-7B-Chat-GGUF \
+ llama-2-7b-chat.Q4_K_M.gguf \
+ --local-dir models/
+```
+
+## Importance Matrices (imatrix)
+
+**What**: Calibration data to improve quantization quality.
+
+**Benefits**:
+- 10-20% perplexity improvement with Q4
+- Essential for Q3 and below
+
+**Usage**:
+```bash
+# 1. Generate importance matrix
+./llama-imatrix \
+ -m model-f16.gguf \
+ -f calibration-data.txt \
+ -o model.imatrix
+
+# 2. Quantize with imatrix
+./llama-quantize \
+ --imatrix model.imatrix \
+ model-f16.gguf \
+ model-Q4_K_M.gguf \
+ Q4_K_M
+```
+
+**Calibration data**:
+- Use domain-specific text (e.g., code for code models)
+- ~100MB of representative text
+- Higher quality data = better quantization
+
+## Troubleshooting
+
+**Model outputs gibberish**:
+- Quantization too aggressive (Q2_K)
+- Try Q4_K_M or Q5_K_M
+- Verify model converted correctly
+
+**Out of memory**:
+- Use lower quantization (Q4_K_S instead of Q5_K_M)
+- Offload fewer layers to GPU (`-ngl`)
+- Use smaller context (`-c 2048`)
+
+**Slow inference**:
+- Higher quantization uses more compute
+- Q8_0 much slower than Q4_K_M
+- Consider speed vs quality trade-off
diff --git a/skills/llama-cpp/references/server.md b/skills/llama-cpp/references/server.md
new file mode 100644
index 0000000..19dba47
--- /dev/null
+++ b/skills/llama-cpp/references/server.md
@@ -0,0 +1,125 @@
+# Server Deployment Guide
+
+Production deployment of llama.cpp server with OpenAI-compatible API.
+
+## Server Modes
+
+### llama-server
+
+```bash
+# Basic server
+./llama-server \
+ -m models/llama-2-7b-chat.Q4_K_M.gguf \
+ --host 0.0.0.0 \
+ --port 8080 \
+ -c 4096 # Context size
+
+# With GPU acceleration
+./llama-server \
+ -m models/llama-2-70b.Q4_K_M.gguf \
+ -ngl 40 # Offload 40 layers to GPU
+```
+
+## OpenAI-Compatible API
+
+### Chat completions
+```bash
+curl http://localhost:8080/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -d '{
+ "model": "llama-2",
+ "messages": [
+ {"role": "system", "content": "You are helpful"},
+ {"role": "user", "content": "Hello"}
+ ],
+ "temperature": 0.7,
+ "max_tokens": 100
+ }'
+```
+
+### Streaming
+```bash
+curl http://localhost:8080/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -d '{
+ "model": "llama-2",
+ "messages": [{"role": "user", "content": "Count to 10"}],
+ "stream": true
+ }'
+```
+
+## Docker Deployment
+
+**Dockerfile**:
+```dockerfile
+FROM ubuntu:22.04
+RUN apt-get update && apt-get install -y git build-essential
+RUN git clone https://github.com/ggerganov/llama.cpp
+WORKDIR /llama.cpp
+RUN make LLAMA_CUDA=1
+COPY models/ /models/
+EXPOSE 8080
+CMD ["./llama-server", "-m", "/models/model.gguf", "--host", "0.0.0.0", "--port", "8080"]
+```
+
+**Run**:
+```bash
+docker run --gpus all -p 8080:8080 llama-cpp:latest
+```
+
+## Monitoring
+
+```bash
+# Server metrics endpoint
+curl http://localhost:8080/metrics
+
+# Health check
+curl http://localhost:8080/health
+```
+
+**Metrics**:
+- requests_total
+- tokens_generated
+- prompt_tokens
+- completion_tokens
+- kv_cache_tokens
+
+## Load Balancing
+
+**NGINX**:
+```nginx
+upstream llama_cpp {
+ server llama1:8080;
+ server llama2:8080;
+}
+
+server {
+ location / {
+ proxy_pass http://llama_cpp;
+ proxy_read_timeout 300s;
+ }
+}
+```
+
+## Performance Tuning
+
+**Parallel requests**:
+```bash
+./llama-server \
+ -m model.gguf \
+ -np 4 # 4 parallel slots
+```
+
+**Continuous batching**:
+```bash
+./llama-server \
+ -m model.gguf \
+ --cont-batching # Enable continuous batching
+```
+
+**Context caching**:
+```bash
+./llama-server \
+ -m model.gguf \
+ --cache-prompt # Cache processed prompts
+```
diff --git a/skills/lm-evaluation-harness/SKILL.md b/skills/lm-evaluation-harness/SKILL.md
new file mode 100644
index 0000000..01b994f
--- /dev/null
+++ b/skills/lm-evaluation-harness/SKILL.md
@@ -0,0 +1,490 @@
+---
+name: lm-evaluation-harness
+description: Evaluates LLMs across 60+ academic benchmarks (MMLU, HumanEval, GSM8K, TruthfulQA, HellaSwag). Use when benchmarking model quality, comparing models, reporting academic results, or tracking training progress. Industry standard used by EleutherAI, HuggingFace, and major labs. Supports HuggingFace, vLLM, APIs.
+version: 1.0.0
+author: Orchestra Research
+license: MIT
+tags: [Evaluation, LM Evaluation Harness, Benchmarking, MMLU, HumanEval, GSM8K, EleutherAI, Model Quality, Academic Benchmarks, Industry Standard]
+dependencies: [lm-eval, transformers, vllm]
+---
+
+# lm-evaluation-harness - LLM Benchmarking
+
+## Quick start
+
+lm-evaluation-harness evaluates LLMs across 60+ academic benchmarks using standardized prompts and metrics.
+
+**Installation**:
+```bash
+pip install lm-eval
+```
+
+**Evaluate any HuggingFace model**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks mmlu,gsm8k,hellaswag \
+ --device cuda:0 \
+ --batch_size 8
+```
+
+**View available tasks**:
+```bash
+lm_eval --tasks list
+```
+
+## Common workflows
+
+### Workflow 1: Standard benchmark evaluation
+
+Evaluate model on core benchmarks (MMLU, GSM8K, HumanEval).
+
+Copy this checklist:
+
+```
+Benchmark Evaluation:
+- [ ] Step 1: Choose benchmark suite
+- [ ] Step 2: Configure model
+- [ ] Step 3: Run evaluation
+- [ ] Step 4: Analyze results
+```
+
+**Step 1: Choose benchmark suite**
+
+**Core reasoning benchmarks**:
+- **MMLU** (Massive Multitask Language Understanding) - 57 subjects, multiple choice
+- **GSM8K** - Grade school math word problems
+- **HellaSwag** - Common sense reasoning
+- **TruthfulQA** - Truthfulness and factuality
+- **ARC** (AI2 Reasoning Challenge) - Science questions
+
+**Code benchmarks**:
+- **HumanEval** - Python code generation (164 problems)
+- **MBPP** (Mostly Basic Python Problems) - Python coding
+
+**Standard suite** (recommended for model releases):
+```bash
+--tasks mmlu,gsm8k,hellaswag,truthfulqa,arc_challenge
+```
+
+**Step 2: Configure model**
+
+**HuggingFace model**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf,dtype=bfloat16 \
+ --tasks mmlu \
+ --device cuda:0 \
+ --batch_size auto # Auto-detect optimal batch size
+```
+
+**Quantized model (4-bit/8-bit)**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf,load_in_4bit=True \
+ --tasks mmlu \
+ --device cuda:0
+```
+
+**Custom checkpoint**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=/path/to/my-model,tokenizer=/path/to/tokenizer \
+ --tasks mmlu \
+ --device cuda:0
+```
+
+**Step 3: Run evaluation**
+
+```bash
+# Full MMLU evaluation (57 subjects)
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks mmlu \
+ --num_fewshot 5 \ # 5-shot evaluation (standard)
+ --batch_size 8 \
+ --output_path results/ \
+ --log_samples # Save individual predictions
+
+# Multiple benchmarks at once
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks mmlu,gsm8k,hellaswag,truthfulqa,arc_challenge \
+ --num_fewshot 5 \
+ --batch_size 8 \
+ --output_path results/llama2-7b-eval.json
+```
+
+**Step 4: Analyze results**
+
+Results saved to `results/llama2-7b-eval.json`:
+
+```json
+{
+ "results": {
+ "mmlu": {
+ "acc": 0.459,
+ "acc_stderr": 0.004
+ },
+ "gsm8k": {
+ "exact_match": 0.142,
+ "exact_match_stderr": 0.006
+ },
+ "hellaswag": {
+ "acc_norm": 0.765,
+ "acc_norm_stderr": 0.004
+ }
+ },
+ "config": {
+ "model": "hf",
+ "model_args": "pretrained=meta-llama/Llama-2-7b-hf",
+ "num_fewshot": 5
+ }
+}
+```
+
+### Workflow 2: Track training progress
+
+Evaluate checkpoints during training.
+
+```
+Training Progress Tracking:
+- [ ] Step 1: Set up periodic evaluation
+- [ ] Step 2: Choose quick benchmarks
+- [ ] Step 3: Automate evaluation
+- [ ] Step 4: Plot learning curves
+```
+
+**Step 1: Set up periodic evaluation**
+
+Evaluate every N training steps:
+
+```bash
+#!/bin/bash
+# eval_checkpoint.sh
+
+CHECKPOINT_DIR=$1
+STEP=$2
+
+lm_eval --model hf \
+ --model_args pretrained=$CHECKPOINT_DIR/checkpoint-$STEP \
+ --tasks gsm8k,hellaswag \
+ --num_fewshot 0 \ # 0-shot for speed
+ --batch_size 16 \
+ --output_path results/step-$STEP.json
+```
+
+**Step 2: Choose quick benchmarks**
+
+Fast benchmarks for frequent evaluation:
+- **HellaSwag**: ~10 minutes on 1 GPU
+- **GSM8K**: ~5 minutes
+- **PIQA**: ~2 minutes
+
+Avoid for frequent eval (too slow):
+- **MMLU**: ~2 hours (57 subjects)
+- **HumanEval**: Requires code execution
+
+**Step 3: Automate evaluation**
+
+Integrate with training script:
+
+```python
+# In training loop
+if step % eval_interval == 0:
+ model.save_pretrained(f"checkpoints/step-{step}")
+
+ # Run evaluation
+ os.system(f"./eval_checkpoint.sh checkpoints step-{step}")
+```
+
+Or use PyTorch Lightning callbacks:
+
+```python
+from pytorch_lightning import Callback
+
+class EvalHarnessCallback(Callback):
+ def on_validation_epoch_end(self, trainer, pl_module):
+ step = trainer.global_step
+ checkpoint_path = f"checkpoints/step-{step}"
+
+ # Save checkpoint
+ trainer.save_checkpoint(checkpoint_path)
+
+ # Run lm-eval
+ os.system(f"lm_eval --model hf --model_args pretrained={checkpoint_path} ...")
+```
+
+**Step 4: Plot learning curves**
+
+```python
+import json
+import matplotlib.pyplot as plt
+
+# Load all results
+steps = []
+mmlu_scores = []
+
+for file in sorted(glob.glob("results/step-*.json")):
+ with open(file) as f:
+ data = json.load(f)
+ step = int(file.split("-")[1].split(".")[0])
+ steps.append(step)
+ mmlu_scores.append(data["results"]["mmlu"]["acc"])
+
+# Plot
+plt.plot(steps, mmlu_scores)
+plt.xlabel("Training Step")
+plt.ylabel("MMLU Accuracy")
+plt.title("Training Progress")
+plt.savefig("training_curve.png")
+```
+
+### Workflow 3: Compare multiple models
+
+Benchmark suite for model comparison.
+
+```
+Model Comparison:
+- [ ] Step 1: Define model list
+- [ ] Step 2: Run evaluations
+- [ ] Step 3: Generate comparison table
+```
+
+**Step 1: Define model list**
+
+```bash
+# models.txt
+meta-llama/Llama-2-7b-hf
+meta-llama/Llama-2-13b-hf
+mistralai/Mistral-7B-v0.1
+microsoft/phi-2
+```
+
+**Step 2: Run evaluations**
+
+```bash
+#!/bin/bash
+# eval_all_models.sh
+
+TASKS="mmlu,gsm8k,hellaswag,truthfulqa"
+
+while read model; do
+ echo "Evaluating $model"
+
+ # Extract model name for output file
+ model_name=$(echo $model | sed 's/\//-/g')
+
+ lm_eval --model hf \
+ --model_args pretrained=$model,dtype=bfloat16 \
+ --tasks $TASKS \
+ --num_fewshot 5 \
+ --batch_size auto \
+ --output_path results/$model_name.json
+
+done < models.txt
+```
+
+**Step 3: Generate comparison table**
+
+```python
+import json
+import pandas as pd
+
+models = [
+ "meta-llama-Llama-2-7b-hf",
+ "meta-llama-Llama-2-13b-hf",
+ "mistralai-Mistral-7B-v0.1",
+ "microsoft-phi-2"
+]
+
+tasks = ["mmlu", "gsm8k", "hellaswag", "truthfulqa"]
+
+results = []
+for model in models:
+ with open(f"results/{model}.json") as f:
+ data = json.load(f)
+ row = {"Model": model.replace("-", "/")}
+ for task in tasks:
+ # Get primary metric for each task
+ metrics = data["results"][task]
+ if "acc" in metrics:
+ row[task.upper()] = f"{metrics['acc']:.3f}"
+ elif "exact_match" in metrics:
+ row[task.upper()] = f"{metrics['exact_match']:.3f}"
+ results.append(row)
+
+df = pd.DataFrame(results)
+print(df.to_markdown(index=False))
+```
+
+Output:
+```
+| Model | MMLU | GSM8K | HELLASWAG | TRUTHFULQA |
+|------------------------|-------|-------|-----------|------------|
+| meta-llama/Llama-2-7b | 0.459 | 0.142 | 0.765 | 0.391 |
+| meta-llama/Llama-2-13b | 0.549 | 0.287 | 0.801 | 0.430 |
+| mistralai/Mistral-7B | 0.626 | 0.395 | 0.812 | 0.428 |
+| microsoft/phi-2 | 0.560 | 0.613 | 0.682 | 0.447 |
+```
+
+### Workflow 4: Evaluate with vLLM (faster inference)
+
+Use vLLM backend for 5-10x faster evaluation.
+
+```
+vLLM Evaluation:
+- [ ] Step 1: Install vLLM
+- [ ] Step 2: Configure vLLM backend
+- [ ] Step 3: Run evaluation
+```
+
+**Step 1: Install vLLM**
+
+```bash
+pip install vllm
+```
+
+**Step 2: Configure vLLM backend**
+
+```bash
+lm_eval --model vllm \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf,tensor_parallel_size=1,dtype=auto,gpu_memory_utilization=0.8 \
+ --tasks mmlu \
+ --batch_size auto
+```
+
+**Step 3: Run evaluation**
+
+vLLM is 5-10× faster than standard HuggingFace:
+
+```bash
+# Standard HF: ~2 hours for MMLU on 7B model
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks mmlu \
+ --batch_size 8
+
+# vLLM: ~15-20 minutes for MMLU on 7B model
+lm_eval --model vllm \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf,tensor_parallel_size=2 \
+ --tasks mmlu \
+ --batch_size auto
+```
+
+## When to use vs alternatives
+
+**Use lm-evaluation-harness when:**
+- Benchmarking models for academic papers
+- Comparing model quality across standard tasks
+- Tracking training progress
+- Reporting standardized metrics (everyone uses same prompts)
+- Need reproducible evaluation
+
+**Use alternatives instead:**
+- **HELM** (Stanford): Broader evaluation (fairness, efficiency, calibration)
+- **AlpacaEval**: Instruction-following evaluation with LLM judges
+- **MT-Bench**: Conversational multi-turn evaluation
+- **Custom scripts**: Domain-specific evaluation
+
+## Common issues
+
+**Issue: Evaluation too slow**
+
+Use vLLM backend:
+```bash
+lm_eval --model vllm \
+ --model_args pretrained=model-name,tensor_parallel_size=2
+```
+
+Or reduce fewshot examples:
+```bash
+--num_fewshot 0 # Instead of 5
+```
+
+Or evaluate subset of MMLU:
+```bash
+--tasks mmlu_stem # Only STEM subjects
+```
+
+**Issue: Out of memory**
+
+Reduce batch size:
+```bash
+--batch_size 1 # Or --batch_size auto
+```
+
+Use quantization:
+```bash
+--model_args pretrained=model-name,load_in_8bit=True
+```
+
+Enable CPU offloading:
+```bash
+--model_args pretrained=model-name,device_map=auto,offload_folder=offload
+```
+
+**Issue: Different results than reported**
+
+Check fewshot count:
+```bash
+--num_fewshot 5 # Most papers use 5-shot
+```
+
+Check exact task name:
+```bash
+--tasks mmlu # Not mmlu_direct or mmlu_fewshot
+```
+
+Verify model and tokenizer match:
+```bash
+--model_args pretrained=model-name,tokenizer=same-model-name
+```
+
+**Issue: HumanEval not executing code**
+
+Install execution dependencies:
+```bash
+pip install human-eval
+```
+
+Enable code execution:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=model-name \
+ --tasks humaneval \
+ --allow_code_execution # Required for HumanEval
+```
+
+## Advanced topics
+
+**Benchmark descriptions**: See [references/benchmark-guide.md](references/benchmark-guide.md) for detailed description of all 60+ tasks, what they measure, and interpretation.
+
+**Custom tasks**: See [references/custom-tasks.md](references/custom-tasks.md) for creating domain-specific evaluation tasks.
+
+**API evaluation**: See [references/api-evaluation.md](references/api-evaluation.md) for evaluating OpenAI, Anthropic, and other API models.
+
+**Multi-GPU strategies**: See [references/distributed-eval.md](references/distributed-eval.md) for data parallel and tensor parallel evaluation.
+
+## Hardware requirements
+
+- **GPU**: NVIDIA (CUDA 11.8+), works on CPU (very slow)
+- **VRAM**:
+ - 7B model: 16GB (bf16) or 8GB (8-bit)
+ - 13B model: 28GB (bf16) or 14GB (8-bit)
+ - 70B model: Requires multi-GPU or quantization
+- **Time** (7B model, single A100):
+ - HellaSwag: 10 minutes
+ - GSM8K: 5 minutes
+ - MMLU (full): 2 hours
+ - HumanEval: 20 minutes
+
+## Resources
+
+- GitHub: https://github.com/EleutherAI/lm-evaluation-harness
+- Docs: https://github.com/EleutherAI/lm-evaluation-harness/tree/main/docs
+- Task library: 60+ tasks including MMLU, GSM8K, HumanEval, TruthfulQA, HellaSwag, ARC, WinoGrande, etc.
+- Leaderboard: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard (uses this harness)
+
+
+
diff --git a/skills/lm-evaluation-harness/references/api-evaluation.md b/skills/lm-evaluation-harness/references/api-evaluation.md
new file mode 100644
index 0000000..db77f61
--- /dev/null
+++ b/skills/lm-evaluation-harness/references/api-evaluation.md
@@ -0,0 +1,490 @@
+# API Evaluation
+
+Guide to evaluating OpenAI, Anthropic, and other API-based language models.
+
+## Overview
+
+The lm-evaluation-harness supports evaluating API-based models through a unified `TemplateAPI` interface. This allows benchmarking of:
+- OpenAI models (GPT-4, GPT-3.5, etc.)
+- Anthropic models (Claude 3, Claude 2, etc.)
+- Local OpenAI-compatible APIs
+- Custom API endpoints
+
+**Why evaluate API models**:
+- Benchmark closed-source models
+- Compare API models to open models
+- Validate API performance
+- Track model updates over time
+
+## Supported API Models
+
+| Provider | Model Type | Request Types | Logprobs |
+|----------|------------|---------------|----------|
+| OpenAI (completions) | `openai-completions` | All | ✅ Yes |
+| OpenAI (chat) | `openai-chat-completions` | `generate_until` only | ❌ No |
+| Anthropic (completions) | `anthropic-completions` | All | ❌ No |
+| Anthropic (chat) | `anthropic-chat` | `generate_until` only | ❌ No |
+| Local (OpenAI-compatible) | `local-completions` | Depends on server | Varies |
+
+**Note**: Models without logprobs can only be evaluated on generation tasks, not perplexity or loglikelihood tasks.
+
+## OpenAI Models
+
+### Setup
+
+```bash
+export OPENAI_API_KEY=sk-...
+```
+
+### Completion Models (Legacy)
+
+**Available models**: `davinci-002`, `babbage-002`
+
+```bash
+lm_eval --model openai-completions \
+ --model_args model=davinci-002 \
+ --tasks lambada_openai,hellaswag \
+ --batch_size auto
+```
+
+**Supports**:
+- `generate_until`: ✅
+- `loglikelihood`: ✅
+- `loglikelihood_rolling`: ✅
+
+### Chat Models
+
+**Available models**: `gpt-4`, `gpt-4-turbo`, `gpt-3.5-turbo`
+
+```bash
+lm_eval --model openai-chat-completions \
+ --model_args model=gpt-4-turbo \
+ --tasks mmlu,gsm8k,humaneval \
+ --num_fewshot 5 \
+ --batch_size auto
+```
+
+**Supports**:
+- `generate_until`: ✅
+- `loglikelihood`: ❌ (no logprobs)
+- `loglikelihood_rolling`: ❌
+
+**Important**: Chat models don't provide logprobs, so they can only be used with generation tasks (MMLU, GSM8K, HumanEval), not perplexity tasks.
+
+### Configuration Options
+
+```bash
+lm_eval --model openai-chat-completions \
+ --model_args \
+ model=gpt-4-turbo,\
+ base_url=https://api.openai.com/v1,\
+ num_concurrent=5,\
+ max_retries=3,\
+ timeout=60,\
+ batch_size=auto
+```
+
+**Parameters**:
+- `model`: Model identifier (required)
+- `base_url`: API endpoint (default: OpenAI)
+- `num_concurrent`: Concurrent requests (default: 5)
+- `max_retries`: Retry failed requests (default: 3)
+- `timeout`: Request timeout in seconds (default: 60)
+- `tokenizer`: Tokenizer to use (default: matches model)
+- `tokenizer_backend`: `"tiktoken"` or `"huggingface"`
+
+### Cost Management
+
+OpenAI charges per token. Estimate costs before running:
+
+```python
+# Rough estimate
+num_samples = 1000
+avg_tokens_per_sample = 500 # input + output
+cost_per_1k_tokens = 0.01 # GPT-3.5 Turbo
+
+total_cost = (num_samples * avg_tokens_per_sample / 1000) * cost_per_1k_tokens
+print(f"Estimated cost: ${total_cost:.2f}")
+```
+
+**Cost-saving tips**:
+- Use `--limit N` for testing
+- Start with `gpt-3.5-turbo` before `gpt-4`
+- Set `max_gen_toks` to minimum needed
+- Use `num_fewshot=0` for zero-shot when possible
+
+## Anthropic Models
+
+### Setup
+
+```bash
+export ANTHROPIC_API_KEY=sk-ant-...
+```
+
+### Completion Models (Legacy)
+
+```bash
+lm_eval --model anthropic-completions \
+ --model_args model=claude-2.1 \
+ --tasks lambada_openai,hellaswag \
+ --batch_size auto
+```
+
+### Chat Models (Recommended)
+
+**Available models**: `claude-3-5-sonnet-20241022`, `claude-3-opus-20240229`, `claude-3-sonnet-20240229`, `claude-3-haiku-20240307`
+
+```bash
+lm_eval --model anthropic-chat \
+ --model_args model=claude-3-5-sonnet-20241022 \
+ --tasks mmlu,gsm8k,humaneval \
+ --num_fewshot 5 \
+ --batch_size auto
+```
+
+**Aliases**: `anthropic-chat-completions` (same as `anthropic-chat`)
+
+### Configuration Options
+
+```bash
+lm_eval --model anthropic-chat \
+ --model_args \
+ model=claude-3-5-sonnet-20241022,\
+ base_url=https://api.anthropic.com,\
+ num_concurrent=5,\
+ max_retries=3,\
+ timeout=60
+```
+
+### Cost Management
+
+Anthropic pricing (as of 2024):
+- Claude 3.5 Sonnet: $3.00 / 1M input, $15.00 / 1M output
+- Claude 3 Opus: $15.00 / 1M input, $75.00 / 1M output
+- Claude 3 Haiku: $0.25 / 1M input, $1.25 / 1M output
+
+**Budget-friendly strategy**:
+```bash
+# Test on small sample first
+lm_eval --model anthropic-chat \
+ --model_args model=claude-3-haiku-20240307 \
+ --tasks mmlu \
+ --limit 100
+
+# Then run full eval on best model
+lm_eval --model anthropic-chat \
+ --model_args model=claude-3-5-sonnet-20241022 \
+ --tasks mmlu \
+ --num_fewshot 5
+```
+
+## Local OpenAI-Compatible APIs
+
+Many local inference servers expose OpenAI-compatible APIs (vLLM, Text Generation Inference, llama.cpp, Ollama).
+
+### vLLM Local Server
+
+**Start server**:
+```bash
+vllm serve meta-llama/Llama-2-7b-hf \
+ --host 0.0.0.0 \
+ --port 8000
+```
+
+**Evaluate**:
+```bash
+lm_eval --model local-completions \
+ --model_args \
+ model=meta-llama/Llama-2-7b-hf,\
+ base_url=http://localhost:8000/v1,\
+ num_concurrent=1 \
+ --tasks mmlu,gsm8k \
+ --batch_size auto
+```
+
+### Text Generation Inference (TGI)
+
+**Start server**:
+```bash
+docker run --gpus all --shm-size 1g -p 8080:80 \
+ ghcr.io/huggingface/text-generation-inference:latest \
+ --model-id meta-llama/Llama-2-7b-hf
+```
+
+**Evaluate**:
+```bash
+lm_eval --model local-completions \
+ --model_args \
+ model=meta-llama/Llama-2-7b-hf,\
+ base_url=http://localhost:8080/v1 \
+ --tasks hellaswag,arc_challenge
+```
+
+### Ollama
+
+**Start server**:
+```bash
+ollama serve
+ollama pull llama2:7b
+```
+
+**Evaluate**:
+```bash
+lm_eval --model local-completions \
+ --model_args \
+ model=llama2:7b,\
+ base_url=http://localhost:11434/v1 \
+ --tasks mmlu
+```
+
+### llama.cpp Server
+
+**Start server**:
+```bash
+./server -m models/llama-2-7b.gguf --host 0.0.0.0 --port 8080
+```
+
+**Evaluate**:
+```bash
+lm_eval --model local-completions \
+ --model_args \
+ model=llama2,\
+ base_url=http://localhost:8080/v1 \
+ --tasks gsm8k
+```
+
+## Custom API Implementation
+
+For custom API endpoints, subclass `TemplateAPI`:
+
+### Create `my_api.py`
+
+```python
+from lm_eval.models.api_models import TemplateAPI
+import requests
+
+class MyCustomAPI(TemplateAPI):
+ """Custom API model."""
+
+ def __init__(self, base_url, api_key, **kwargs):
+ super().__init__(base_url=base_url, **kwargs)
+ self.api_key = api_key
+
+ def _create_payload(self, messages, gen_kwargs):
+ """Create API request payload."""
+ return {
+ "messages": messages,
+ "api_key": self.api_key,
+ **gen_kwargs
+ }
+
+ def parse_generations(self, response):
+ """Parse generation response."""
+ return response.json()["choices"][0]["text"]
+
+ def parse_logprobs(self, response):
+ """Parse logprobs (if available)."""
+ # Return None if API doesn't provide logprobs
+ logprobs = response.json().get("logprobs")
+ if logprobs:
+ return logprobs["token_logprobs"]
+ return None
+```
+
+### Register and Use
+
+```python
+from lm_eval import evaluator
+from my_api import MyCustomAPI
+
+model = MyCustomAPI(
+ base_url="https://api.example.com/v1",
+ api_key="your-key"
+)
+
+results = evaluator.simple_evaluate(
+ model=model,
+ tasks=["mmlu", "gsm8k"],
+ num_fewshot=5,
+ batch_size="auto"
+)
+```
+
+## Comparing API and Open Models
+
+### Side-by-Side Evaluation
+
+```bash
+# Evaluate OpenAI GPT-4
+lm_eval --model openai-chat-completions \
+ --model_args model=gpt-4-turbo \
+ --tasks mmlu,gsm8k,hellaswag \
+ --num_fewshot 5 \
+ --output_path results/gpt4.json
+
+# Evaluate open Llama 2 70B
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-70b-hf,dtype=bfloat16 \
+ --tasks mmlu,gsm8k,hellaswag \
+ --num_fewshot 5 \
+ --output_path results/llama2-70b.json
+
+# Compare results
+python scripts/compare_results.py \
+ results/gpt4.json \
+ results/llama2-70b.json
+```
+
+### Typical Comparisons
+
+| Model | MMLU | GSM8K | HumanEval | Cost |
+|-------|------|-------|-----------|------|
+| GPT-4 Turbo | 86.4% | 92.0% | 67.0% | $$$$ |
+| Claude 3 Opus | 86.8% | 95.0% | 84.9% | $$$$ |
+| GPT-3.5 Turbo | 70.0% | 57.1% | 48.1% | $$ |
+| Llama 2 70B | 68.9% | 56.8% | 29.9% | Free (self-host) |
+| Mixtral 8x7B | 70.6% | 58.4% | 40.2% | Free (self-host) |
+
+## Best Practices
+
+### Rate Limiting
+
+Respect API rate limits:
+```bash
+lm_eval --model openai-chat-completions \
+ --model_args \
+ model=gpt-4-turbo,\
+ num_concurrent=3,\ # Lower concurrency
+ timeout=120 \ # Longer timeout
+ --tasks mmlu
+```
+
+### Reproducibility
+
+Set temperature to 0 for deterministic results:
+```bash
+lm_eval --model openai-chat-completions \
+ --model_args model=gpt-4-turbo \
+ --tasks mmlu \
+ --gen_kwargs temperature=0.0
+```
+
+Or use `seed` for sampling:
+```bash
+lm_eval --model anthropic-chat \
+ --model_args model=claude-3-5-sonnet-20241022 \
+ --tasks gsm8k \
+ --gen_kwargs temperature=0.7,seed=42
+```
+
+### Caching
+
+API models automatically cache responses to avoid redundant calls:
+```bash
+# First run: makes API calls
+lm_eval --model openai-chat-completions \
+ --model_args model=gpt-4-turbo \
+ --tasks mmlu \
+ --limit 100
+
+# Second run: uses cache (instant, free)
+lm_eval --model openai-chat-completions \
+ --model_args model=gpt-4-turbo \
+ --tasks mmlu \
+ --limit 100
+```
+
+Cache location: `~/.cache/lm_eval/`
+
+### Error Handling
+
+APIs can fail. Use retries:
+```bash
+lm_eval --model openai-chat-completions \
+ --model_args \
+ model=gpt-4-turbo,\
+ max_retries=5,\
+ timeout=120 \
+ --tasks mmlu
+```
+
+## Troubleshooting
+
+### "Authentication failed"
+
+Check API key:
+```bash
+echo $OPENAI_API_KEY # Should print sk-...
+echo $ANTHROPIC_API_KEY # Should print sk-ant-...
+```
+
+### "Rate limit exceeded"
+
+Reduce concurrency:
+```bash
+--model_args num_concurrent=1
+```
+
+Or add delays between requests.
+
+### "Timeout error"
+
+Increase timeout:
+```bash
+--model_args timeout=180
+```
+
+### "Model not found"
+
+For local APIs, verify server is running:
+```bash
+curl http://localhost:8000/v1/models
+```
+
+### Cost Runaway
+
+Use `--limit` for testing:
+```bash
+lm_eval --model openai-chat-completions \
+ --model_args model=gpt-4-turbo \
+ --tasks mmlu \
+ --limit 50 # Only 50 samples
+```
+
+## Advanced Features
+
+### Custom Headers
+
+```bash
+lm_eval --model local-completions \
+ --model_args \
+ base_url=http://api.example.com/v1,\
+ header="Authorization: Bearer token,X-Custom: value"
+```
+
+### Disable SSL Verification (Development Only)
+
+```bash
+lm_eval --model local-completions \
+ --model_args \
+ base_url=https://localhost:8000/v1,\
+ verify_certificate=false
+```
+
+### Custom Tokenizer
+
+```bash
+lm_eval --model openai-chat-completions \
+ --model_args \
+ model=gpt-4-turbo,\
+ tokenizer=gpt2,\
+ tokenizer_backend=huggingface
+```
+
+## References
+
+- OpenAI API: https://platform.openai.com/docs/api-reference
+- Anthropic API: https://docs.anthropic.com/claude/reference
+- TemplateAPI: `lm_eval/models/api_models.py`
+- OpenAI models: `lm_eval/models/openai_completions.py`
+- Anthropic models: `lm_eval/models/anthropic_llms.py`
diff --git a/skills/lm-evaluation-harness/references/benchmark-guide.md b/skills/lm-evaluation-harness/references/benchmark-guide.md
new file mode 100644
index 0000000..e3031ec
--- /dev/null
+++ b/skills/lm-evaluation-harness/references/benchmark-guide.md
@@ -0,0 +1,488 @@
+# Benchmark Guide
+
+Complete guide to all 60+ evaluation tasks in lm-evaluation-harness, what they measure, and how to interpret results.
+
+## Overview
+
+The lm-evaluation-harness includes 60+ benchmarks spanning:
+- Language understanding (MMLU, GLUE)
+- Mathematical reasoning (GSM8K, MATH)
+- Code generation (HumanEval, MBPP)
+- Instruction following (IFEval, AlpacaEval)
+- Long-context understanding (LongBench)
+- Multilingual capabilities (AfroBench, NorEval)
+- Reasoning (BBH, ARC)
+- Truthfulness (TruthfulQA)
+
+**List all tasks**:
+```bash
+lm_eval --tasks list
+```
+
+## Major Benchmarks
+
+### MMLU (Massive Multitask Language Understanding)
+
+**What it measures**: Broad knowledge across 57 subjects (STEM, humanities, social sciences, law).
+
+**Task variants**:
+- `mmlu`: Original 57-subject benchmark
+- `mmlu_pro`: More challenging version with reasoning-focused questions
+- `mmlu_prox`: Multilingual extension
+
+**Format**: Multiple choice (4 options)
+
+**Example**:
+```
+Question: What is the capital of France?
+A. Berlin
+B. Paris
+C. London
+D. Madrid
+Answer: B
+```
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks mmlu \
+ --num_fewshot 5
+```
+
+**Interpretation**:
+- Random: 25% (chance)
+- GPT-3 (175B): 43.9%
+- GPT-4: 86.4%
+- Human expert: ~90%
+
+**Good for**: Assessing general knowledge and domain expertise.
+
+### GSM8K (Grade School Math 8K)
+
+**What it measures**: Mathematical reasoning on grade-school level word problems.
+
+**Task variants**:
+- `gsm8k`: Base task
+- `gsm8k_cot`: With chain-of-thought prompting
+- `gsm_plus`: Adversarial variant with perturbations
+
+**Format**: Free-form generation, extract numerical answer
+
+**Example**:
+```
+Question: A baker made 200 cookies. He sold 3/5 of them in the morning and 1/4 of the remaining in the afternoon. How many cookies does he have left?
+Answer: 60
+```
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks gsm8k \
+ --num_fewshot 5
+```
+
+**Interpretation**:
+- Random: ~0%
+- GPT-3 (175B): 17.0%
+- GPT-4: 92.0%
+- Llama 2 70B: 56.8%
+
+**Good for**: Testing multi-step reasoning and arithmetic.
+
+### HumanEval
+
+**What it measures**: Python code generation from docstrings (functional correctness).
+
+**Task variants**:
+- `humaneval`: Standard benchmark
+- `humaneval_instruct`: For instruction-tuned models
+
+**Format**: Code generation, execution-based evaluation
+
+**Example**:
+```python
+def has_close_elements(numbers: List[float], threshold: float) -> bool:
+ """ Check if in given list of numbers, are any two numbers closer to each other than
+ given threshold.
+ >>> has_close_elements([1.0, 2.0, 3.0], 0.5)
+ False
+ >>> has_close_elements([1.0, 2.8, 3.0, 4.0, 5.0, 2.0], 0.3)
+ True
+ """
+```
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=codellama/CodeLlama-7b-hf \
+ --tasks humaneval \
+ --batch_size 1
+```
+
+**Interpretation**:
+- Random: 0%
+- GPT-3 (175B): 0%
+- Codex: 28.8%
+- GPT-4: 67.0%
+- Code Llama 34B: 53.7%
+
+**Good for**: Evaluating code generation capabilities.
+
+### BBH (BIG-Bench Hard)
+
+**What it measures**: 23 challenging reasoning tasks where models previously failed to beat humans.
+
+**Categories**:
+- Logical reasoning
+- Math word problems
+- Social understanding
+- Algorithmic reasoning
+
+**Format**: Multiple choice and free-form
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks bbh \
+ --num_fewshot 3
+```
+
+**Interpretation**:
+- Random: ~25%
+- GPT-3 (175B): 33.9%
+- PaLM 540B: 58.3%
+- GPT-4: 86.7%
+
+**Good for**: Testing advanced reasoning capabilities.
+
+### IFEval (Instruction-Following Evaluation)
+
+**What it measures**: Ability to follow specific, verifiable instructions.
+
+**Instruction types**:
+- Format constraints (e.g., "answer in 3 sentences")
+- Length constraints (e.g., "use at least 100 words")
+- Content constraints (e.g., "include the word 'banana'")
+- Structural constraints (e.g., "use bullet points")
+
+**Format**: Free-form generation with rule-based verification
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-chat-hf \
+ --tasks ifeval \
+ --batch_size auto
+```
+
+**Interpretation**:
+- Measures: Instruction adherence (not quality)
+- GPT-4: 86% instruction following
+- Claude 2: 84%
+
+**Good for**: Evaluating chat/instruct models.
+
+### GLUE (General Language Understanding Evaluation)
+
+**What it measures**: Natural language understanding across 9 tasks.
+
+**Tasks**:
+- `cola`: Grammatical acceptability
+- `sst2`: Sentiment analysis
+- `mrpc`: Paraphrase detection
+- `qqp`: Question pairs
+- `stsb`: Semantic similarity
+- `mnli`: Natural language inference
+- `qnli`: Question answering NLI
+- `rte`: Recognizing textual entailment
+- `wnli`: Winograd schemas
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=bert-base-uncased \
+ --tasks glue \
+ --num_fewshot 0
+```
+
+**Interpretation**:
+- BERT Base: 78.3 (GLUE score)
+- RoBERTa Large: 88.5
+- Human baseline: 87.1
+
+**Good for**: Encoder-only models, fine-tuning baselines.
+
+### LongBench
+
+**What it measures**: Long-context understanding (4K-32K tokens).
+
+**21 tasks covering**:
+- Single-document QA
+- Multi-document QA
+- Summarization
+- Few-shot learning
+- Code completion
+- Synthetic tasks
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks longbench \
+ --batch_size 1
+```
+
+**Interpretation**:
+- Tests context utilization
+- Many models struggle beyond 4K tokens
+- GPT-4 Turbo: 54.3%
+
+**Good for**: Evaluating long-context models.
+
+## Additional Benchmarks
+
+### TruthfulQA
+
+**What it measures**: Model's propensity to be truthful vs. generate plausible-sounding falsehoods.
+
+**Format**: Multiple choice with 4-5 options
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks truthfulqa_mc2 \
+ --batch_size auto
+```
+
+**Interpretation**:
+- Larger models often score worse (more convincing lies)
+- GPT-3: 58.8%
+- GPT-4: 59.0%
+- Human: ~94%
+
+### ARC (AI2 Reasoning Challenge)
+
+**What it measures**: Grade-school science questions.
+
+**Variants**:
+- `arc_easy`: Easier questions
+- `arc_challenge`: Harder questions requiring reasoning
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks arc_challenge \
+ --num_fewshot 25
+```
+
+**Interpretation**:
+- ARC-Easy: Most models >80%
+- ARC-Challenge random: 25%
+- GPT-4: 96.3%
+
+### HellaSwag
+
+**What it measures**: Commonsense reasoning about everyday situations.
+
+**Format**: Choose most plausible continuation
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks hellaswag \
+ --num_fewshot 10
+```
+
+**Interpretation**:
+- Random: 25%
+- GPT-3: 78.9%
+- Llama 2 70B: 85.3%
+
+### WinoGrande
+
+**What it measures**: Commonsense reasoning via pronoun resolution.
+
+**Example**:
+```
+The trophy doesn't fit in the brown suitcase because _ is too large.
+A. the trophy
+B. the suitcase
+```
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks winogrande \
+ --num_fewshot 5
+```
+
+### PIQA
+
+**What it measures**: Physical commonsense reasoning.
+
+**Example**: "To clean a keyboard, use compressed air or..."
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks piqa
+```
+
+## Multilingual Benchmarks
+
+### AfroBench
+
+**What it measures**: Performance across 64 African languages.
+
+**15 tasks**: NLU, text generation, knowledge, QA, math reasoning
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks afrobench
+```
+
+### NorEval
+
+**What it measures**: Norwegian language understanding (9 task categories).
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=NbAiLab/nb-gpt-j-6B \
+ --tasks noreval
+```
+
+## Domain-Specific Benchmarks
+
+### MATH
+
+**What it measures**: High-school competition math problems.
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks math \
+ --num_fewshot 4
+```
+
+**Interpretation**:
+- Very challenging
+- GPT-4: 42.5%
+- Minerva 540B: 33.6%
+
+### MBPP (Mostly Basic Python Problems)
+
+**What it measures**: Python programming from natural language descriptions.
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=codellama/CodeLlama-7b-hf \
+ --tasks mbpp \
+ --batch_size 1
+```
+
+### DROP
+
+**What it measures**: Reading comprehension requiring discrete reasoning.
+
+**Command**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks drop
+```
+
+## Benchmark Selection Guide
+
+### For General Purpose Models
+
+Run this suite:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks mmlu,gsm8k,hellaswag,arc_challenge,truthfulqa_mc2 \
+ --num_fewshot 5
+```
+
+### For Code Models
+
+```bash
+lm_eval --model hf \
+ --model_args pretrained=codellama/CodeLlama-7b-hf \
+ --tasks humaneval,mbpp \
+ --batch_size 1
+```
+
+### For Chat/Instruct Models
+
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-chat-hf \
+ --tasks ifeval,mmlu,gsm8k_cot \
+ --batch_size auto
+```
+
+### For Long Context Models
+
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-3.1-8B \
+ --tasks longbench \
+ --batch_size 1
+```
+
+## Interpreting Results
+
+### Understanding Metrics
+
+**Accuracy**: Percentage of correct answers (most common)
+
+**Exact Match (EM)**: Requires exact string match (strict)
+
+**F1 Score**: Balances precision and recall
+
+**BLEU/ROUGE**: Text generation similarity
+
+**Pass@k**: Percentage passing when generating k samples
+
+### Typical Score Ranges
+
+| Model Size | MMLU | GSM8K | HumanEval | HellaSwag |
+|------------|------|-------|-----------|-----------|
+| 7B | 40-50% | 10-20% | 5-15% | 70-80% |
+| 13B | 45-55% | 20-35% | 15-25% | 75-82% |
+| 70B | 60-70% | 50-65% | 35-50% | 82-87% |
+| GPT-4 | 86% | 92% | 67% | 95% |
+
+### Red Flags
+
+- **All tasks at random chance**: Model not trained properly
+- **Exact 0% on generation tasks**: Likely format/parsing issue
+- **Huge variance across runs**: Check seed/sampling settings
+- **Better than GPT-4 on everything**: Likely contamination
+
+## Best Practices
+
+1. **Always report few-shot setting**: 0-shot, 5-shot, etc.
+2. **Run multiple seeds**: Report mean ± std
+3. **Check for data contamination**: Search training data for benchmark examples
+4. **Compare to published baselines**: Validate your setup
+5. **Report all hyperparameters**: Model, batch size, max tokens, temperature
+
+## References
+
+- Task list: `lm_eval --tasks list`
+- Task README: `lm_eval/tasks/README.md`
+- Papers: See individual benchmark papers
diff --git a/skills/lm-evaluation-harness/references/custom-tasks.md b/skills/lm-evaluation-harness/references/custom-tasks.md
new file mode 100644
index 0000000..c5c1e89
--- /dev/null
+++ b/skills/lm-evaluation-harness/references/custom-tasks.md
@@ -0,0 +1,602 @@
+# Custom Tasks
+
+Complete guide to creating domain-specific evaluation tasks in lm-evaluation-harness.
+
+## Overview
+
+Custom tasks allow you to evaluate models on your own datasets and metrics. Tasks are defined using YAML configuration files with optional Python utilities for complex logic.
+
+**Why create custom tasks**:
+- Evaluate on proprietary/domain-specific data
+- Test specific capabilities not covered by existing benchmarks
+- Create evaluation pipelines for internal models
+- Reproduce research experiments
+
+## Quick Start
+
+### Minimal Custom Task
+
+Create `my_tasks/simple_qa.yaml`:
+
+```yaml
+task: simple_qa
+dataset_path: data/simple_qa.jsonl
+output_type: generate_until
+doc_to_text: "Question: {{question}}\nAnswer:"
+doc_to_target: "{{answer}}"
+metric_list:
+ - metric: exact_match
+ aggregation: mean
+ higher_is_better: true
+```
+
+**Run it**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks simple_qa \
+ --include_path my_tasks/
+```
+
+## Task Configuration Reference
+
+### Essential Fields
+
+```yaml
+# Task identification
+task: my_custom_task # Unique task name (required)
+task_alias: "My Task" # Display name
+tag: # Tags for grouping
+ - custom
+ - domain_specific
+
+# Dataset configuration
+dataset_path: data/my_data.jsonl # HuggingFace dataset or local path
+dataset_name: default # Subset name (if applicable)
+training_split: train
+validation_split: validation
+test_split: test
+
+# Evaluation configuration
+output_type: generate_until # or loglikelihood, multiple_choice
+num_fewshot: 5 # Number of few-shot examples
+batch_size: auto # Batch size
+
+# Prompt templates (Jinja2)
+doc_to_text: "Question: {{question}}"
+doc_to_target: "{{answer}}"
+
+# Metrics
+metric_list:
+ - metric: exact_match
+ aggregation: mean
+ higher_is_better: true
+
+# Metadata
+metadata:
+ version: 1.0
+```
+
+### Output Types
+
+**`generate_until`**: Free-form generation
+```yaml
+output_type: generate_until
+generation_kwargs:
+ max_gen_toks: 256
+ until:
+ - "\n"
+ - "."
+ temperature: 0.0
+```
+
+**`loglikelihood`**: Compute log probability of targets
+```yaml
+output_type: loglikelihood
+# Used for perplexity, classification
+```
+
+**`multiple_choice`**: Choose from options
+```yaml
+output_type: multiple_choice
+doc_to_choice: "{{choices}}" # List of choices
+```
+
+## Data Formats
+
+### Local JSONL File
+
+`data/my_data.jsonl`:
+```json
+{"question": "What is 2+2?", "answer": "4"}
+{"question": "Capital of France?", "answer": "Paris"}
+```
+
+**Task config**:
+```yaml
+dataset_path: data/my_data.jsonl
+dataset_kwargs:
+ data_files:
+ test: data/my_data.jsonl
+```
+
+### HuggingFace Dataset
+
+```yaml
+dataset_path: squad
+dataset_name: plain_text
+test_split: validation
+```
+
+### CSV File
+
+`data/my_data.csv`:
+```csv
+question,answer,category
+What is 2+2?,4,math
+Capital of France?,Paris,geography
+```
+
+**Task config**:
+```yaml
+dataset_path: data/my_data.csv
+dataset_kwargs:
+ data_files:
+ test: data/my_data.csv
+```
+
+## Prompt Engineering
+
+### Simple Template
+
+```yaml
+doc_to_text: "Question: {{question}}\nAnswer:"
+doc_to_target: "{{answer}}"
+```
+
+### Conditional Logic
+
+```yaml
+doc_to_text: |
+ {% if context %}
+ Context: {{context}}
+ {% endif %}
+ Question: {{question}}
+ Answer:
+```
+
+### Multiple Choice
+
+```yaml
+doc_to_text: |
+ Question: {{question}}
+ A. {{choices[0]}}
+ B. {{choices[1]}}
+ C. {{choices[2]}}
+ D. {{choices[3]}}
+ Answer:
+
+doc_to_target: "{{ 'ABCD'[answer_idx] }}"
+doc_to_choice: ["A", "B", "C", "D"]
+```
+
+### Few-Shot Formatting
+
+```yaml
+fewshot_delimiter: "\n\n" # Between examples
+target_delimiter: " " # Between question and answer
+doc_to_text: "Q: {{question}}"
+doc_to_target: "A: {{answer}}"
+```
+
+## Custom Python Functions
+
+For complex logic, use Python functions in `utils.py`.
+
+### Create `my_tasks/utils.py`
+
+```python
+def process_docs(dataset):
+ """Preprocess documents."""
+ def _process(doc):
+ # Custom preprocessing
+ doc["question"] = doc["question"].strip().lower()
+ return doc
+
+ return dataset.map(_process)
+
+def doc_to_text(doc):
+ """Custom prompt formatting."""
+ context = doc.get("context", "")
+ question = doc["question"]
+
+ if context:
+ return f"Context: {context}\nQuestion: {question}\nAnswer:"
+ return f"Question: {question}\nAnswer:"
+
+def doc_to_target(doc):
+ """Custom target extraction."""
+ return doc["answer"].strip().lower()
+
+def aggregate_scores(items):
+ """Custom metric aggregation."""
+ correct = sum(1 for item in items if item == 1.0)
+ total = len(items)
+ return correct / total if total > 0 else 0.0
+```
+
+### Use in Task Config
+
+```yaml
+task: my_custom_task
+dataset_path: data/my_data.jsonl
+
+# Use Python functions
+process_docs: !function utils.process_docs
+doc_to_text: !function utils.doc_to_text
+doc_to_target: !function utils.doc_to_target
+
+metric_list:
+ - metric: exact_match
+ aggregation: !function utils.aggregate_scores
+ higher_is_better: true
+```
+
+## Real-World Examples
+
+### Example 1: Domain QA Task
+
+**Goal**: Evaluate medical question answering.
+
+`medical_qa/medical_qa.yaml`:
+```yaml
+task: medical_qa
+dataset_path: data/medical_qa.jsonl
+output_type: generate_until
+num_fewshot: 3
+
+doc_to_text: |
+ Medical Question: {{question}}
+ Context: {{context}}
+ Answer (be concise):
+
+doc_to_target: "{{answer}}"
+
+generation_kwargs:
+ max_gen_toks: 100
+ until:
+ - "\n\n"
+ temperature: 0.0
+
+metric_list:
+ - metric: exact_match
+ aggregation: mean
+ higher_is_better: true
+ - metric: !function utils.medical_f1
+ aggregation: mean
+ higher_is_better: true
+
+filter_list:
+ - name: lowercase
+ filter:
+ - function: lowercase
+ - function: remove_whitespace
+
+metadata:
+ version: 1.0
+ domain: medical
+```
+
+`medical_qa/utils.py`:
+```python
+from sklearn.metrics import f1_score
+import re
+
+def medical_f1(predictions, references):
+ """Custom F1 for medical terms."""
+ pred_terms = set(extract_medical_terms(predictions[0]))
+ ref_terms = set(extract_medical_terms(references[0]))
+
+ if not pred_terms and not ref_terms:
+ return 1.0
+ if not pred_terms or not ref_terms:
+ return 0.0
+
+ tp = len(pred_terms & ref_terms)
+ fp = len(pred_terms - ref_terms)
+ fn = len(ref_terms - pred_terms)
+
+ precision = tp / (tp + fp) if (tp + fp) > 0 else 0
+ recall = tp / (tp + fn) if (tp + fn) > 0 else 0
+
+ return 2 * (precision * recall) / (precision + recall) if (precision + recall) > 0 else 0
+
+def extract_medical_terms(text):
+ """Extract medical terminology."""
+ # Custom logic
+ return re.findall(r'\b[A-Z][a-z]+(?:[A-Z][a-z]+)*\b', text)
+```
+
+### Example 2: Code Evaluation
+
+`code_eval/python_challenges.yaml`:
+```yaml
+task: python_challenges
+dataset_path: data/python_problems.jsonl
+output_type: generate_until
+num_fewshot: 0
+
+doc_to_text: |
+ Write a Python function to solve:
+ {{problem_statement}}
+
+ Function signature:
+ {{function_signature}}
+
+doc_to_target: "{{canonical_solution}}"
+
+generation_kwargs:
+ max_gen_toks: 512
+ until:
+ - "\n\nclass"
+ - "\n\ndef"
+ temperature: 0.2
+
+metric_list:
+ - metric: !function utils.execute_code
+ aggregation: mean
+ higher_is_better: true
+
+process_results: !function utils.process_code_results
+
+metadata:
+ version: 1.0
+```
+
+`code_eval/utils.py`:
+```python
+import subprocess
+import json
+
+def execute_code(predictions, references):
+ """Execute generated code against test cases."""
+ generated_code = predictions[0]
+ test_cases = json.loads(references[0])
+
+ try:
+ # Execute code with test cases
+ for test_input, expected_output in test_cases:
+ result = execute_with_timeout(generated_code, test_input, timeout=5)
+ if result != expected_output:
+ return 0.0
+ return 1.0
+ except Exception:
+ return 0.0
+
+def execute_with_timeout(code, input_data, timeout=5):
+ """Safely execute code with timeout."""
+ # Implementation with subprocess and timeout
+ pass
+
+def process_code_results(doc, results):
+ """Process code execution results."""
+ return {
+ "passed": results[0] == 1.0,
+ "generated_code": results[1]
+ }
+```
+
+### Example 3: Instruction Following
+
+`instruction_eval/instruction_eval.yaml`:
+```yaml
+task: instruction_following
+dataset_path: data/instructions.jsonl
+output_type: generate_until
+num_fewshot: 0
+
+doc_to_text: |
+ Instruction: {{instruction}}
+ {% if constraints %}
+ Constraints: {{constraints}}
+ {% endif %}
+ Response:
+
+doc_to_target: "{{expected_response}}"
+
+generation_kwargs:
+ max_gen_toks: 256
+ temperature: 0.7
+
+metric_list:
+ - metric: !function utils.check_constraints
+ aggregation: mean
+ higher_is_better: true
+ - metric: !function utils.semantic_similarity
+ aggregation: mean
+ higher_is_better: true
+
+process_docs: !function utils.add_constraint_checkers
+```
+
+`instruction_eval/utils.py`:
+```python
+from sentence_transformers import SentenceTransformer, util
+
+model = SentenceTransformer('all-MiniLM-L6-v2')
+
+def check_constraints(predictions, references):
+ """Check if response satisfies constraints."""
+ response = predictions[0]
+ constraints = json.loads(references[0])
+
+ satisfied = 0
+ total = len(constraints)
+
+ for constraint in constraints:
+ if verify_constraint(response, constraint):
+ satisfied += 1
+
+ return satisfied / total if total > 0 else 1.0
+
+def verify_constraint(response, constraint):
+ """Verify single constraint."""
+ if constraint["type"] == "length":
+ return len(response.split()) >= constraint["min_words"]
+ elif constraint["type"] == "contains":
+ return constraint["keyword"] in response.lower()
+ # Add more constraint types
+ return True
+
+def semantic_similarity(predictions, references):
+ """Compute semantic similarity."""
+ pred_embedding = model.encode(predictions[0])
+ ref_embedding = model.encode(references[0])
+ return float(util.cos_sim(pred_embedding, ref_embedding))
+
+def add_constraint_checkers(dataset):
+ """Parse constraints into verifiable format."""
+ def _parse(doc):
+ # Parse constraint string into structured format
+ doc["parsed_constraints"] = parse_constraints(doc.get("constraints", ""))
+ return doc
+ return dataset.map(_parse)
+```
+
+## Advanced Features
+
+### Output Filtering
+
+```yaml
+filter_list:
+ - name: extract_answer
+ filter:
+ - function: regex
+ regex_pattern: "Answer: (.*)"
+ group: 1
+ - function: lowercase
+ - function: strip_whitespace
+```
+
+### Multiple Metrics
+
+```yaml
+metric_list:
+ - metric: exact_match
+ aggregation: mean
+ higher_is_better: true
+ - metric: f1
+ aggregation: mean
+ higher_is_better: true
+ - metric: bleu
+ aggregation: mean
+ higher_is_better: true
+```
+
+### Task Groups
+
+Create `my_tasks/_default.yaml`:
+```yaml
+group: my_eval_suite
+task:
+ - simple_qa
+ - medical_qa
+ - python_challenges
+```
+
+**Run entire suite**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks my_eval_suite \
+ --include_path my_tasks/
+```
+
+## Testing Your Task
+
+### Validate Configuration
+
+```bash
+# Test task loading
+lm_eval --tasks my_custom_task --include_path my_tasks/ --limit 0
+
+# Run on 5 samples
+lm_eval --model hf \
+ --model_args pretrained=gpt2 \
+ --tasks my_custom_task \
+ --include_path my_tasks/ \
+ --limit 5
+```
+
+### Debug Mode
+
+```bash
+lm_eval --model hf \
+ --model_args pretrained=gpt2 \
+ --tasks my_custom_task \
+ --include_path my_tasks/ \
+ --limit 1 \
+ --log_samples # Save input/output samples
+```
+
+## Best Practices
+
+1. **Start simple**: Test with minimal config first
+2. **Version your tasks**: Use `metadata.version`
+3. **Document your metrics**: Explain custom metrics in comments
+4. **Test with multiple models**: Ensure robustness
+5. **Validate on known examples**: Include sanity checks
+6. **Use filters carefully**: Can hide errors
+7. **Handle edge cases**: Empty strings, missing fields
+
+## Common Patterns
+
+### Classification Task
+
+```yaml
+output_type: loglikelihood
+doc_to_text: "Text: {{text}}\nLabel:"
+doc_to_target: " {{label}}" # Space prefix important!
+metric_list:
+ - metric: acc
+ aggregation: mean
+```
+
+### Perplexity Evaluation
+
+```yaml
+output_type: loglikelihood_rolling
+doc_to_text: "{{text}}"
+metric_list:
+ - metric: perplexity
+ aggregation: perplexity
+```
+
+### Ranking Task
+
+```yaml
+output_type: loglikelihood
+doc_to_text: "Query: {{query}}\nPassage: {{passage}}\nRelevant:"
+doc_to_target: [" Yes", " No"]
+metric_list:
+ - metric: acc
+ aggregation: mean
+```
+
+## Troubleshooting
+
+**"Task not found"**: Check `--include_path` and task name
+
+**Empty results**: Verify `doc_to_text` and `doc_to_target` templates
+
+**Metric errors**: Ensure metric names are correct (exact_match, not exact-match)
+
+**Filter issues**: Test filters with `--log_samples`
+
+**Python function not found**: Check `!function module.function_name` syntax
+
+## References
+
+- Task system: EleutherAI/lm-evaluation-harness docs
+- Example tasks: `lm_eval/tasks/` directory
+- TaskConfig: `lm_eval/api/task.py`
diff --git a/skills/lm-evaluation-harness/references/distributed-eval.md b/skills/lm-evaluation-harness/references/distributed-eval.md
new file mode 100644
index 0000000..2132e5b
--- /dev/null
+++ b/skills/lm-evaluation-harness/references/distributed-eval.md
@@ -0,0 +1,519 @@
+# Distributed Evaluation
+
+Guide to running evaluation across multiple GPUs using data parallelism and tensor/pipeline parallelism.
+
+## Overview
+
+Distributed evaluation speeds up benchmarking by:
+- **Data Parallelism**: Split evaluation samples across GPUs (each GPU has full model copy)
+- **Tensor Parallelism**: Split model weights across GPUs (for large models)
+- **Pipeline Parallelism**: Split model layers across GPUs (for very large models)
+
+**When to use**:
+- Data Parallel: Model fits on single GPU, want faster evaluation
+- Tensor/Pipeline Parallel: Model too large for single GPU
+
+## HuggingFace Models (`hf`)
+
+### Data Parallelism (Recommended)
+
+Each GPU loads a full copy of the model and processes a subset of evaluation data.
+
+**Single Node (8 GPUs)**:
+```bash
+accelerate launch --multi_gpu --num_processes 8 \
+ -m lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf,dtype=bfloat16 \
+ --tasks mmlu,gsm8k,hellaswag \
+ --batch_size 16
+```
+
+**Speedup**: Near-linear (8 GPUs = ~8× faster)
+
+**Memory**: Each GPU needs full model (7B model ≈ 14GB × 8 = 112GB total)
+
+### Tensor Parallelism (Model Sharding)
+
+Split model weights across GPUs for models too large for single GPU.
+
+**Without accelerate launcher**:
+```bash
+lm_eval --model hf \
+ --model_args \
+ pretrained=meta-llama/Llama-2-70b-hf,\
+ parallelize=True,\
+ dtype=bfloat16 \
+ --tasks mmlu,gsm8k \
+ --batch_size 8
+```
+
+**With 8 GPUs**: 70B model (140GB) / 8 = 17.5GB per GPU ✅
+
+**Advanced sharding**:
+```bash
+lm_eval --model hf \
+ --model_args \
+ pretrained=meta-llama/Llama-2-70b-hf,\
+ parallelize=True,\
+ device_map_option=auto,\
+ max_memory_per_gpu=40GB,\
+ max_cpu_memory=100GB,\
+ dtype=bfloat16 \
+ --tasks mmlu
+```
+
+**Options**:
+- `device_map_option`: `"auto"` (default), `"balanced"`, `"balanced_low_0"`
+- `max_memory_per_gpu`: Max memory per GPU (e.g., `"40GB"`)
+- `max_cpu_memory`: Max CPU memory for offloading
+- `offload_folder`: Disk offloading directory
+
+### Combined Data + Tensor Parallelism
+
+Use both for very large models.
+
+**Example: 70B model on 16 GPUs (2 copies, 8 GPUs each)**:
+```bash
+accelerate launch --multi_gpu --num_processes 2 \
+ -m lm_eval --model hf \
+ --model_args \
+ pretrained=meta-llama/Llama-2-70b-hf,\
+ parallelize=True,\
+ dtype=bfloat16 \
+ --tasks mmlu \
+ --batch_size 8
+```
+
+**Result**: 2× speedup from data parallelism, 70B model fits via tensor parallelism
+
+### Configuration with `accelerate config`
+
+Create `~/.cache/huggingface/accelerate/default_config.yaml`:
+```yaml
+compute_environment: LOCAL_MACHINE
+distributed_type: MULTI_GPU
+num_machines: 1
+num_processes: 8
+gpu_ids: all
+mixed_precision: bf16
+```
+
+**Then run**:
+```bash
+accelerate launch -m lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks mmlu
+```
+
+## vLLM Models (`vllm`)
+
+vLLM provides highly optimized distributed inference.
+
+### Tensor Parallelism
+
+**Single Node (4 GPUs)**:
+```bash
+lm_eval --model vllm \
+ --model_args \
+ pretrained=meta-llama/Llama-2-70b-hf,\
+ tensor_parallel_size=4,\
+ dtype=auto,\
+ gpu_memory_utilization=0.9 \
+ --tasks mmlu,gsm8k \
+ --batch_size auto
+```
+
+**Memory**: 70B model split across 4 GPUs = ~35GB per GPU
+
+### Data Parallelism
+
+**Multiple model replicas**:
+```bash
+lm_eval --model vllm \
+ --model_args \
+ pretrained=meta-llama/Llama-2-7b-hf,\
+ data_parallel_size=4,\
+ dtype=auto,\
+ gpu_memory_utilization=0.8 \
+ --tasks hellaswag,arc_challenge \
+ --batch_size auto
+```
+
+**Result**: 4 model replicas = 4× throughput
+
+### Combined Tensor + Data Parallelism
+
+**Example: 8 GPUs = 4 TP × 2 DP**:
+```bash
+lm_eval --model vllm \
+ --model_args \
+ pretrained=meta-llama/Llama-2-70b-hf,\
+ tensor_parallel_size=4,\
+ data_parallel_size=2,\
+ dtype=auto,\
+ gpu_memory_utilization=0.85 \
+ --tasks mmlu \
+ --batch_size auto
+```
+
+**Result**: 70B model fits (TP=4), 2× speedup (DP=2)
+
+### Multi-Node vLLM
+
+vLLM doesn't natively support multi-node. Use Ray:
+
+```bash
+# Start Ray cluster
+ray start --head --port=6379
+
+# Run evaluation
+lm_eval --model vllm \
+ --model_args \
+ pretrained=meta-llama/Llama-2-70b-hf,\
+ tensor_parallel_size=8,\
+ dtype=auto \
+ --tasks mmlu
+```
+
+## NVIDIA NeMo Models (`nemo_lm`)
+
+### Data Replication
+
+**8 replicas on 8 GPUs**:
+```bash
+torchrun --nproc-per-node=8 --no-python \
+ lm_eval --model nemo_lm \
+ --model_args \
+ path=/path/to/model.nemo,\
+ devices=8 \
+ --tasks hellaswag,arc_challenge \
+ --batch_size 32
+```
+
+**Speedup**: Near-linear (8× faster)
+
+### Tensor Parallelism
+
+**4-way tensor parallelism**:
+```bash
+torchrun --nproc-per-node=4 --no-python \
+ lm_eval --model nemo_lm \
+ --model_args \
+ path=/path/to/70b_model.nemo,\
+ devices=4,\
+ tensor_model_parallel_size=4 \
+ --tasks mmlu,gsm8k \
+ --batch_size 16
+```
+
+### Pipeline Parallelism
+
+**2 TP × 2 PP on 4 GPUs**:
+```bash
+torchrun --nproc-per-node=4 --no-python \
+ lm_eval --model nemo_lm \
+ --model_args \
+ path=/path/to/model.nemo,\
+ devices=4,\
+ tensor_model_parallel_size=2,\
+ pipeline_model_parallel_size=2 \
+ --tasks mmlu \
+ --batch_size 8
+```
+
+**Constraint**: `devices = TP × PP`
+
+### Multi-Node NeMo
+
+Currently not supported by lm-evaluation-harness.
+
+## SGLang Models (`sglang`)
+
+### Tensor Parallelism
+
+```bash
+lm_eval --model sglang \
+ --model_args \
+ pretrained=meta-llama/Llama-2-70b-hf,\
+ tp_size=4,\
+ dtype=auto \
+ --tasks gsm8k \
+ --batch_size auto
+```
+
+### Data Parallelism (Deprecated)
+
+**Note**: SGLang is deprecating data parallelism. Use tensor parallelism instead.
+
+```bash
+lm_eval --model sglang \
+ --model_args \
+ pretrained=meta-llama/Llama-2-7b-hf,\
+ dp_size=4,\
+ dtype=auto \
+ --tasks mmlu
+```
+
+## Performance Comparison
+
+### 70B Model Evaluation (MMLU, 5-shot)
+
+| Method | GPUs | Time | Memory/GPU | Notes |
+|--------|------|------|------------|-------|
+| HF (no parallel) | 1 | 8 hours | 140GB (OOM) | Won't fit |
+| HF (TP=8) | 8 | 2 hours | 17.5GB | Slower, fits |
+| HF (DP=8) | 8 | 1 hour | 140GB (OOM) | Won't fit |
+| vLLM (TP=4) | 4 | 30 min | 35GB | Fast! |
+| vLLM (TP=4, DP=2) | 8 | 15 min | 35GB | Fastest |
+
+### 7B Model Evaluation (Multiple Tasks)
+
+| Method | GPUs | Time | Speedup |
+|--------|------|------|---------|
+| HF (single) | 1 | 4 hours | 1× |
+| HF (DP=4) | 4 | 1 hour | 4× |
+| HF (DP=8) | 8 | 30 min | 8× |
+| vLLM (DP=8) | 8 | 15 min | 16× |
+
+**Takeaway**: vLLM is significantly faster than HuggingFace for inference.
+
+## Choosing Parallelism Strategy
+
+### Decision Tree
+
+```
+Model fits on single GPU?
+├─ YES: Use data parallelism
+│ ├─ HF: accelerate launch --multi_gpu --num_processes N
+│ └─ vLLM: data_parallel_size=N (fastest)
+│
+└─ NO: Use tensor/pipeline parallelism
+ ├─ Model < 70B:
+ │ └─ vLLM: tensor_parallel_size=4
+ ├─ Model 70-175B:
+ │ ├─ vLLM: tensor_parallel_size=8
+ │ └─ Or HF: parallelize=True
+ └─ Model > 175B:
+ └─ Contact framework authors
+```
+
+### Memory Estimation
+
+**Rule of thumb**:
+```
+Memory (GB) = Parameters (B) × Precision (bytes) × 1.2 (overhead)
+```
+
+**Examples**:
+- 7B FP16: 7 × 2 × 1.2 = 16.8GB ✅ Fits A100 40GB
+- 13B FP16: 13 × 2 × 1.2 = 31.2GB ✅ Fits A100 40GB
+- 70B FP16: 70 × 2 × 1.2 = 168GB ❌ Need TP=4 or TP=8
+- 70B BF16: 70 × 2 × 1.2 = 168GB (same as FP16)
+
+**With tensor parallelism**:
+```
+Memory per GPU = Total Memory / TP
+```
+
+- 70B on 4 GPUs: 168GB / 4 = 42GB per GPU ✅
+- 70B on 8 GPUs: 168GB / 8 = 21GB per GPU ✅
+
+## Multi-Node Evaluation
+
+### HuggingFace with SLURM
+
+**Submit job**:
+```bash
+#!/bin/bash
+#SBATCH --nodes=4
+#SBATCH --gpus-per-node=8
+#SBATCH --ntasks-per-node=1
+
+srun accelerate launch --multi_gpu \
+ --num_processes $((SLURM_NNODES * 8)) \
+ -m lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks mmlu,gsm8k,hellaswag \
+ --batch_size 16
+```
+
+**Submit**:
+```bash
+sbatch eval_job.sh
+```
+
+### Manual Multi-Node Setup
+
+**On each node, run**:
+```bash
+accelerate launch \
+ --multi_gpu \
+ --num_machines 4 \
+ --num_processes 32 \
+ --main_process_ip $MASTER_IP \
+ --main_process_port 29500 \
+ --machine_rank $NODE_RANK \
+ -m lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks mmlu
+```
+
+**Environment variables**:
+- `MASTER_IP`: IP of rank 0 node
+- `NODE_RANK`: 0, 1, 2, 3 for each node
+
+## Best Practices
+
+### 1. Start Small
+
+Test on small sample first:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-70b-hf,parallelize=True \
+ --tasks mmlu \
+ --limit 100 # Just 100 samples
+```
+
+### 2. Monitor GPU Usage
+
+```bash
+# Terminal 1: Run evaluation
+lm_eval --model hf ...
+
+# Terminal 2: Monitor
+watch -n 1 nvidia-smi
+```
+
+Look for:
+- GPU utilization > 90%
+- Memory usage stable
+- All GPUs active
+
+### 3. Optimize Batch Size
+
+```bash
+# Auto batch size (recommended)
+--batch_size auto
+
+# Or tune manually
+--batch_size 16 # Start here
+--batch_size 32 # Increase if memory allows
+```
+
+### 4. Use Mixed Precision
+
+```bash
+--model_args dtype=bfloat16 # Faster, less memory
+```
+
+### 5. Check Communication
+
+For data parallelism, check network bandwidth:
+```bash
+# Should see InfiniBand or high-speed network
+nvidia-smi topo -m
+```
+
+## Troubleshooting
+
+### "CUDA out of memory"
+
+**Solutions**:
+1. Increase tensor parallelism:
+ ```bash
+ --model_args tensor_parallel_size=8 # Was 4
+ ```
+
+2. Reduce batch size:
+ ```bash
+ --batch_size 4 # Was 16
+ ```
+
+3. Lower precision:
+ ```bash
+ --model_args dtype=int8 # Quantization
+ ```
+
+### "NCCL error" or Hanging
+
+**Check**:
+1. All GPUs visible: `nvidia-smi`
+2. NCCL installed: `python -c "import torch; print(torch.cuda.nccl.version())"`
+3. Network connectivity between nodes
+
+**Fix**:
+```bash
+export NCCL_DEBUG=INFO # Enable debug logging
+export NCCL_IB_DISABLE=0 # Use InfiniBand if available
+```
+
+### Slow Evaluation
+
+**Possible causes**:
+1. **Data loading bottleneck**: Preprocess dataset
+2. **Low GPU utilization**: Increase batch size
+3. **Communication overhead**: Reduce parallelism degree
+
+**Profile**:
+```bash
+lm_eval --model hf \
+ --model_args pretrained=meta-llama/Llama-2-7b-hf \
+ --tasks mmlu \
+ --limit 100 \
+ --log_samples # Check timing
+```
+
+### GPUs Imbalanced
+
+**Symptom**: GPU 0 at 100%, others at 50%
+
+**Solution**: Use `device_map_option=balanced`:
+```bash
+--model_args parallelize=True,device_map_option=balanced
+```
+
+## Example Configurations
+
+### Small Model (7B) - Fast Evaluation
+
+```bash
+# 8 A100s, data parallel
+accelerate launch --multi_gpu --num_processes 8 \
+ -m lm_eval --model hf \
+ --model_args \
+ pretrained=meta-llama/Llama-2-7b-hf,\
+ dtype=bfloat16 \
+ --tasks mmlu,gsm8k,hellaswag,arc_challenge \
+ --num_fewshot 5 \
+ --batch_size 32
+
+# Time: ~30 minutes
+```
+
+### Large Model (70B) - vLLM
+
+```bash
+# 8 H100s, tensor parallel
+lm_eval --model vllm \
+ --model_args \
+ pretrained=meta-llama/Llama-2-70b-hf,\
+ tensor_parallel_size=8,\
+ dtype=auto,\
+ gpu_memory_utilization=0.9 \
+ --tasks mmlu,gsm8k,humaneval \
+ --num_fewshot 5 \
+ --batch_size auto
+
+# Time: ~1 hour
+```
+
+### Very Large Model (175B+)
+
+**Requires specialized setup - contact framework maintainers**
+
+## References
+
+- HuggingFace Accelerate: https://huggingface.co/docs/accelerate/
+- vLLM docs: https://docs.vllm.ai/
+- NeMo docs: https://docs.nvidia.com/nemo-framework/
+- lm-eval distributed guide: `docs/model_guide.md`
diff --git a/skills/peft/SKILL.md b/skills/peft/SKILL.md
new file mode 100644
index 0000000..609e0c5
--- /dev/null
+++ b/skills/peft/SKILL.md
@@ -0,0 +1,431 @@
+---
+name: peft
+description: Parameter-efficient fine-tuning for LLMs using LoRA, QLoRA, and 25+ methods. Use when fine-tuning large models (7B-70B) with limited GPU memory, when you need to train <1% of parameters with minimal accuracy loss, or for multi-adapter serving. HuggingFace's official library integrated with transformers ecosystem.
+version: 1.0.0
+author: Orchestra Research
+license: MIT
+tags: [Fine-Tuning, PEFT, LoRA, QLoRA, Parameter-Efficient, Adapters, Low-Rank, Memory Optimization, Multi-Adapter]
+dependencies: [peft>=0.13.0, transformers>=4.45.0, torch>=2.0.0, bitsandbytes>=0.43.0]
+---
+
+# PEFT (Parameter-Efficient Fine-Tuning)
+
+Fine-tune LLMs by training <1% of parameters using LoRA, QLoRA, and 25+ adapter methods.
+
+## When to use PEFT
+
+**Use PEFT/LoRA when:**
+- Fine-tuning 7B-70B models on consumer GPUs (RTX 4090, A100)
+- Need to train <1% parameters (6MB adapters vs 14GB full model)
+- Want fast iteration with multiple task-specific adapters
+- Deploying multiple fine-tuned variants from one base model
+
+**Use QLoRA (PEFT + quantization) when:**
+- Fine-tuning 70B models on single 24GB GPU
+- Memory is the primary constraint
+- Can accept ~5% quality trade-off vs full fine-tuning
+
+**Use full fine-tuning instead when:**
+- Training small models (<1B parameters)
+- Need maximum quality and have compute budget
+- Significant domain shift requires updating all weights
+
+## Quick start
+
+### Installation
+
+```bash
+# Basic installation
+pip install peft
+
+# With quantization support (recommended)
+pip install peft bitsandbytes
+
+# Full stack
+pip install peft transformers accelerate bitsandbytes datasets
+```
+
+### LoRA fine-tuning (standard)
+
+```python
+from transformers import AutoModelForCausalLM, AutoTokenizer, TrainingArguments, Trainer
+from peft import get_peft_model, LoraConfig, TaskType
+from datasets import load_dataset
+
+# Load base model
+model_name = "meta-llama/Llama-3.1-8B"
+model = AutoModelForCausalLM.from_pretrained(model_name, torch_dtype="auto", device_map="auto")
+tokenizer = AutoTokenizer.from_pretrained(model_name)
+tokenizer.pad_token = tokenizer.eos_token
+
+# LoRA configuration
+lora_config = LoraConfig(
+ task_type=TaskType.CAUSAL_LM,
+ r=16, # Rank (8-64, higher = more capacity)
+ lora_alpha=32, # Scaling factor (typically 2*r)
+ lora_dropout=0.05, # Dropout for regularization
+ target_modules=["q_proj", "v_proj", "k_proj", "o_proj"], # Attention layers
+ bias="none" # Don't train biases
+)
+
+# Apply LoRA
+model = get_peft_model(model, lora_config)
+model.print_trainable_parameters()
+# Output: trainable params: 13,631,488 || all params: 8,043,307,008 || trainable%: 0.17%
+
+# Prepare dataset
+dataset = load_dataset("databricks/databricks-dolly-15k", split="train")
+
+def tokenize(example):
+ text = f"### Instruction:\n{example['instruction']}\n\n### Response:\n{example['response']}"
+ return tokenizer(text, truncation=True, max_length=512, padding="max_length")
+
+tokenized = dataset.map(tokenize, remove_columns=dataset.column_names)
+
+# Training
+training_args = TrainingArguments(
+ output_dir="./lora-llama",
+ num_train_epochs=3,
+ per_device_train_batch_size=4,
+ gradient_accumulation_steps=4,
+ learning_rate=2e-4,
+ fp16=True,
+ logging_steps=10,
+ save_strategy="epoch"
+)
+
+trainer = Trainer(
+ model=model,
+ args=training_args,
+ train_dataset=tokenized,
+ data_collator=lambda data: {"input_ids": torch.stack([f["input_ids"] for f in data]),
+ "attention_mask": torch.stack([f["attention_mask"] for f in data]),
+ "labels": torch.stack([f["input_ids"] for f in data])}
+)
+
+trainer.train()
+
+# Save adapter only (6MB vs 16GB)
+model.save_pretrained("./lora-llama-adapter")
+```
+
+### QLoRA fine-tuning (memory-efficient)
+
+```python
+from transformers import AutoModelForCausalLM, BitsAndBytesConfig
+from peft import get_peft_model, LoraConfig, prepare_model_for_kbit_training
+
+# 4-bit quantization config
+bnb_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_quant_type="nf4", # NormalFloat4 (best for LLMs)
+ bnb_4bit_compute_dtype="bfloat16", # Compute in bf16
+ bnb_4bit_use_double_quant=True # Nested quantization
+)
+
+# Load quantized model
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-3.1-70B",
+ quantization_config=bnb_config,
+ device_map="auto"
+)
+
+# Prepare for training (enables gradient checkpointing)
+model = prepare_model_for_kbit_training(model)
+
+# LoRA config for QLoRA
+lora_config = LoraConfig(
+ r=64, # Higher rank for 70B
+ lora_alpha=128,
+ lora_dropout=0.1,
+ target_modules=["q_proj", "v_proj", "k_proj", "o_proj", "gate_proj", "up_proj", "down_proj"],
+ bias="none",
+ task_type="CAUSAL_LM"
+)
+
+model = get_peft_model(model, lora_config)
+# 70B model now fits on single 24GB GPU!
+```
+
+## LoRA parameter selection
+
+### Rank (r) - capacity vs efficiency
+
+| Rank | Trainable Params | Memory | Quality | Use Case |
+|------|-----------------|--------|---------|----------|
+| 4 | ~3M | Minimal | Lower | Simple tasks, prototyping |
+| **8** | ~7M | Low | Good | **Recommended starting point** |
+| **16** | ~14M | Medium | Better | **General fine-tuning** |
+| 32 | ~27M | Higher | High | Complex tasks |
+| 64 | ~54M | High | Highest | Domain adaptation, 70B models |
+
+### Alpha (lora_alpha) - scaling factor
+
+```python
+# Rule of thumb: alpha = 2 * rank
+LoraConfig(r=16, lora_alpha=32) # Standard
+LoraConfig(r=16, lora_alpha=16) # Conservative (lower learning rate effect)
+LoraConfig(r=16, lora_alpha=64) # Aggressive (higher learning rate effect)
+```
+
+### Target modules by architecture
+
+```python
+# Llama / Mistral / Qwen
+target_modules = ["q_proj", "v_proj", "k_proj", "o_proj", "gate_proj", "up_proj", "down_proj"]
+
+# GPT-2 / GPT-Neo
+target_modules = ["c_attn", "c_proj", "c_fc"]
+
+# Falcon
+target_modules = ["query_key_value", "dense", "dense_h_to_4h", "dense_4h_to_h"]
+
+# BLOOM
+target_modules = ["query_key_value", "dense", "dense_h_to_4h", "dense_4h_to_h"]
+
+# Auto-detect all linear layers
+target_modules = "all-linear" # PEFT 0.6.0+
+```
+
+## Loading and merging adapters
+
+### Load trained adapter
+
+```python
+from peft import PeftModel, AutoPeftModelForCausalLM
+from transformers import AutoModelForCausalLM
+
+# Option 1: Load with PeftModel
+base_model = AutoModelForCausalLM.from_pretrained("meta-llama/Llama-3.1-8B")
+model = PeftModel.from_pretrained(base_model, "./lora-llama-adapter")
+
+# Option 2: Load directly (recommended)
+model = AutoPeftModelForCausalLM.from_pretrained(
+ "./lora-llama-adapter",
+ device_map="auto"
+)
+```
+
+### Merge adapter into base model
+
+```python
+# Merge for deployment (no adapter overhead)
+merged_model = model.merge_and_unload()
+
+# Save merged model
+merged_model.save_pretrained("./llama-merged")
+tokenizer.save_pretrained("./llama-merged")
+
+# Push to Hub
+merged_model.push_to_hub("username/llama-finetuned")
+```
+
+### Multi-adapter serving
+
+```python
+from peft import PeftModel
+
+# Load base with first adapter
+model = AutoPeftModelForCausalLM.from_pretrained("./adapter-task1")
+
+# Load additional adapters
+model.load_adapter("./adapter-task2", adapter_name="task2")
+model.load_adapter("./adapter-task3", adapter_name="task3")
+
+# Switch between adapters at runtime
+model.set_adapter("task1") # Use task1 adapter
+output1 = model.generate(**inputs)
+
+model.set_adapter("task2") # Switch to task2
+output2 = model.generate(**inputs)
+
+# Disable adapters (use base model)
+with model.disable_adapter():
+ base_output = model.generate(**inputs)
+```
+
+## PEFT methods comparison
+
+| Method | Trainable % | Memory | Speed | Best For |
+|--------|------------|--------|-------|----------|
+| **LoRA** | 0.1-1% | Low | Fast | General fine-tuning |
+| **QLoRA** | 0.1-1% | Very Low | Medium | Memory-constrained |
+| AdaLoRA | 0.1-1% | Low | Medium | Automatic rank selection |
+| IA3 | 0.01% | Minimal | Fastest | Few-shot adaptation |
+| Prefix Tuning | 0.1% | Low | Medium | Generation control |
+| Prompt Tuning | 0.001% | Minimal | Fast | Simple task adaptation |
+| P-Tuning v2 | 0.1% | Low | Medium | NLU tasks |
+
+### IA3 (minimal parameters)
+
+```python
+from peft import IA3Config
+
+ia3_config = IA3Config(
+ target_modules=["q_proj", "v_proj", "k_proj", "down_proj"],
+ feedforward_modules=["down_proj"]
+)
+model = get_peft_model(model, ia3_config)
+# Trains only 0.01% of parameters!
+```
+
+### Prefix Tuning
+
+```python
+from peft import PrefixTuningConfig
+
+prefix_config = PrefixTuningConfig(
+ task_type="CAUSAL_LM",
+ num_virtual_tokens=20, # Prepended tokens
+ prefix_projection=True # Use MLP projection
+)
+model = get_peft_model(model, prefix_config)
+```
+
+## Integration patterns
+
+### With TRL (SFTTrainer)
+
+```python
+from trl import SFTTrainer, SFTConfig
+from peft import LoraConfig
+
+lora_config = LoraConfig(r=16, lora_alpha=32, target_modules="all-linear")
+
+trainer = SFTTrainer(
+ model=model,
+ args=SFTConfig(output_dir="./output", max_seq_length=512),
+ train_dataset=dataset,
+ peft_config=lora_config, # Pass LoRA config directly
+)
+trainer.train()
+```
+
+### With Axolotl (YAML config)
+
+```yaml
+# axolotl config.yaml
+adapter: lora
+lora_r: 16
+lora_alpha: 32
+lora_dropout: 0.05
+lora_target_modules:
+ - q_proj
+ - v_proj
+ - k_proj
+ - o_proj
+lora_target_linear: true # Target all linear layers
+```
+
+### With vLLM (inference)
+
+```python
+from vllm import LLM
+from vllm.lora.request import LoRARequest
+
+# Load base model with LoRA support
+llm = LLM(model="meta-llama/Llama-3.1-8B", enable_lora=True)
+
+# Serve with adapter
+outputs = llm.generate(
+ prompts,
+ lora_request=LoRARequest("adapter1", 1, "./lora-adapter")
+)
+```
+
+## Performance benchmarks
+
+### Memory usage (Llama 3.1 8B)
+
+| Method | GPU Memory | Trainable Params |
+|--------|-----------|------------------|
+| Full fine-tuning | 60+ GB | 8B (100%) |
+| LoRA r=16 | 18 GB | 14M (0.17%) |
+| QLoRA r=16 | 6 GB | 14M (0.17%) |
+| IA3 | 16 GB | 800K (0.01%) |
+
+### Training speed (A100 80GB)
+
+| Method | Tokens/sec | vs Full FT |
+|--------|-----------|------------|
+| Full FT | 2,500 | 1x |
+| LoRA | 3,200 | 1.3x |
+| QLoRA | 2,100 | 0.84x |
+
+### Quality (MMLU benchmark)
+
+| Model | Full FT | LoRA | QLoRA |
+|-------|---------|------|-------|
+| Llama 2-7B | 45.3 | 44.8 | 44.1 |
+| Llama 2-13B | 54.8 | 54.2 | 53.5 |
+
+## Common issues
+
+### CUDA OOM during training
+
+```python
+# Solution 1: Enable gradient checkpointing
+model.gradient_checkpointing_enable()
+
+# Solution 2: Reduce batch size + increase accumulation
+TrainingArguments(
+ per_device_train_batch_size=1,
+ gradient_accumulation_steps=16
+)
+
+# Solution 3: Use QLoRA
+from transformers import BitsAndBytesConfig
+bnb_config = BitsAndBytesConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4")
+```
+
+### Adapter not applying
+
+```python
+# Verify adapter is active
+print(model.active_adapters) # Should show adapter name
+
+# Check trainable parameters
+model.print_trainable_parameters()
+
+# Ensure model in training mode
+model.train()
+```
+
+### Quality degradation
+
+```python
+# Increase rank
+LoraConfig(r=32, lora_alpha=64)
+
+# Target more modules
+target_modules = "all-linear"
+
+# Use more training data and epochs
+TrainingArguments(num_train_epochs=5)
+
+# Lower learning rate
+TrainingArguments(learning_rate=1e-4)
+```
+
+## Best practices
+
+1. **Start with r=8-16**, increase if quality insufficient
+2. **Use alpha = 2 * rank** as starting point
+3. **Target attention + MLP layers** for best quality/efficiency
+4. **Enable gradient checkpointing** for memory savings
+5. **Save adapters frequently** (small files, easy rollback)
+6. **Evaluate on held-out data** before merging
+7. **Use QLoRA for 70B+ models** on consumer hardware
+
+## References
+
+- **[Advanced Usage](references/advanced-usage.md)** - DoRA, LoftQ, rank stabilization, custom modules
+- **[Troubleshooting](references/troubleshooting.md)** - Common errors, debugging, optimization
+
+## Resources
+
+- **GitHub**: https://github.com/huggingface/peft
+- **Docs**: https://huggingface.co/docs/peft
+- **LoRA Paper**: arXiv:2106.09685
+- **QLoRA Paper**: arXiv:2305.14314
+- **Models**: https://huggingface.co/models?library=peft
diff --git a/skills/peft/references/advanced-usage.md b/skills/peft/references/advanced-usage.md
new file mode 100644
index 0000000..d23c0d4
--- /dev/null
+++ b/skills/peft/references/advanced-usage.md
@@ -0,0 +1,514 @@
+# PEFT Advanced Usage Guide
+
+## Advanced LoRA Variants
+
+### DoRA (Weight-Decomposed Low-Rank Adaptation)
+
+DoRA decomposes weights into magnitude and direction components, often achieving better results than standard LoRA:
+
+```python
+from peft import LoraConfig
+
+dora_config = LoraConfig(
+ r=16,
+ lora_alpha=32,
+ target_modules=["q_proj", "v_proj", "k_proj", "o_proj"],
+ use_dora=True, # Enable DoRA
+ task_type="CAUSAL_LM"
+)
+
+model = get_peft_model(model, dora_config)
+```
+
+**When to use DoRA**:
+- Consistently outperforms LoRA on instruction-following tasks
+- Slightly higher memory (~10%) due to magnitude vectors
+- Best for quality-critical fine-tuning
+
+### AdaLoRA (Adaptive Rank)
+
+Automatically adjusts rank per layer based on importance:
+
+```python
+from peft import AdaLoraConfig
+
+adalora_config = AdaLoraConfig(
+ init_r=64, # Initial rank
+ target_r=16, # Target average rank
+ tinit=200, # Warmup steps
+ tfinal=1000, # Final pruning step
+ deltaT=10, # Rank update frequency
+ beta1=0.85,
+ beta2=0.85,
+ orth_reg_weight=0.5, # Orthogonality regularization
+ target_modules=["q_proj", "v_proj"],
+ task_type="CAUSAL_LM"
+)
+```
+
+**Benefits**:
+- Allocates more rank to important layers
+- Can reduce total parameters while maintaining quality
+- Good for exploring optimal rank distribution
+
+### LoRA+ (Asymmetric Learning Rates)
+
+Different learning rates for A and B matrices:
+
+```python
+from peft import LoraConfig
+
+# LoRA+ uses higher LR for B matrix
+lora_plus_config = LoraConfig(
+ r=16,
+ lora_alpha=32,
+ target_modules="all-linear",
+ use_rslora=True, # Rank-stabilized LoRA (related technique)
+)
+
+# Manual implementation of LoRA+
+from torch.optim import AdamW
+
+# Group parameters
+lora_A_params = [p for n, p in model.named_parameters() if "lora_A" in n]
+lora_B_params = [p for n, p in model.named_parameters() if "lora_B" in n]
+
+optimizer = AdamW([
+ {"params": lora_A_params, "lr": 1e-4},
+ {"params": lora_B_params, "lr": 1e-3}, # 10x higher for B
+])
+```
+
+### rsLoRA (Rank-Stabilized LoRA)
+
+Scales LoRA outputs to stabilize training with different ranks:
+
+```python
+lora_config = LoraConfig(
+ r=64,
+ lora_alpha=64,
+ use_rslora=True, # Enables rank-stabilized scaling
+ target_modules="all-linear"
+)
+```
+
+**When to use**:
+- When experimenting with different ranks
+- Helps maintain consistent behavior across rank values
+- Recommended for r > 32
+
+## LoftQ (LoRA-Fine-Tuning-aware Quantization)
+
+Initializes LoRA weights to compensate for quantization error:
+
+```python
+from peft import LoftQConfig, LoraConfig, get_peft_model
+from transformers import AutoModelForCausalLM, BitsAndBytesConfig
+
+# LoftQ configuration
+loftq_config = LoftQConfig(
+ loftq_bits=4, # Quantization bits
+ loftq_iter=5, # Alternating optimization iterations
+)
+
+# LoRA config with LoftQ initialization
+lora_config = LoraConfig(
+ r=16,
+ lora_alpha=32,
+ target_modules="all-linear",
+ init_lora_weights="loftq",
+ loftq_config=loftq_config,
+ task_type="CAUSAL_LM"
+)
+
+# Load quantized model
+bnb_config = BitsAndBytesConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4")
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-3.1-8B",
+ quantization_config=bnb_config
+)
+
+model = get_peft_model(model, lora_config)
+```
+
+**Benefits over standard QLoRA**:
+- Better initial quality after quantization
+- Faster convergence
+- ~1-2% better final accuracy on benchmarks
+
+## Custom Module Targeting
+
+### Target specific layers
+
+```python
+# Target only first and last transformer layers
+lora_config = LoraConfig(
+ r=16,
+ lora_alpha=32,
+ target_modules=["model.layers.0.self_attn.q_proj",
+ "model.layers.0.self_attn.v_proj",
+ "model.layers.31.self_attn.q_proj",
+ "model.layers.31.self_attn.v_proj"],
+ layers_to_transform=[0, 31] # Alternative approach
+)
+```
+
+### Layer pattern matching
+
+```python
+# Target layers 0-10 only
+lora_config = LoraConfig(
+ r=16,
+ lora_alpha=32,
+ target_modules="all-linear",
+ layers_to_transform=list(range(11)), # Layers 0-10
+ layers_pattern="model.layers"
+)
+```
+
+### Exclude specific layers
+
+```python
+lora_config = LoraConfig(
+ r=16,
+ target_modules="all-linear",
+ modules_to_save=["lm_head"], # Train these fully (not LoRA)
+)
+```
+
+## Embedding and LM Head Training
+
+### Train embeddings with LoRA
+
+```python
+from peft import LoraConfig
+
+# Include embeddings
+lora_config = LoraConfig(
+ r=16,
+ lora_alpha=32,
+ target_modules=["q_proj", "v_proj", "embed_tokens"], # Include embeddings
+ modules_to_save=["lm_head"], # Train lm_head fully
+)
+```
+
+### Extending vocabulary with LoRA
+
+```python
+from transformers import AutoModelForCausalLM, AutoTokenizer
+from peft import get_peft_model, LoraConfig
+
+# Add new tokens
+tokenizer = AutoTokenizer.from_pretrained("meta-llama/Llama-3.1-8B")
+new_tokens = ["", ""]
+tokenizer.add_tokens(new_tokens)
+
+# Resize model embeddings
+model = AutoModelForCausalLM.from_pretrained("meta-llama/Llama-3.1-8B")
+model.resize_token_embeddings(len(tokenizer))
+
+# Configure LoRA to train new embeddings
+lora_config = LoraConfig(
+ r=16,
+ target_modules="all-linear",
+ modules_to_save=["embed_tokens", "lm_head"], # Train these fully
+)
+
+model = get_peft_model(model, lora_config)
+```
+
+## Multi-Adapter Patterns
+
+### Adapter composition
+
+```python
+from peft import PeftModel
+
+# Load model with multiple adapters
+model = AutoPeftModelForCausalLM.from_pretrained("./base-adapter")
+model.load_adapter("./style-adapter", adapter_name="style")
+model.load_adapter("./task-adapter", adapter_name="task")
+
+# Combine adapters (weighted sum)
+model.add_weighted_adapter(
+ adapters=["style", "task"],
+ weights=[0.7, 0.3],
+ adapter_name="combined",
+ combination_type="linear" # or "cat", "svd"
+)
+
+model.set_adapter("combined")
+```
+
+### Adapter stacking
+
+```python
+# Stack adapters (apply sequentially)
+model.add_weighted_adapter(
+ adapters=["base", "domain", "task"],
+ weights=[1.0, 1.0, 1.0],
+ adapter_name="stacked",
+ combination_type="cat" # Concatenate adapter outputs
+)
+```
+
+### Dynamic adapter switching
+
+```python
+import torch
+
+class MultiAdapterModel:
+ def __init__(self, base_model_path, adapter_paths):
+ self.model = AutoPeftModelForCausalLM.from_pretrained(adapter_paths[0])
+ for name, path in adapter_paths[1:].items():
+ self.model.load_adapter(path, adapter_name=name)
+
+ def generate(self, prompt, adapter_name="default"):
+ self.model.set_adapter(adapter_name)
+ return self.model.generate(**self.tokenize(prompt))
+
+ def generate_ensemble(self, prompt, adapters, weights):
+ """Generate with weighted adapter ensemble"""
+ outputs = []
+ for adapter, weight in zip(adapters, weights):
+ self.model.set_adapter(adapter)
+ logits = self.model(**self.tokenize(prompt)).logits
+ outputs.append(weight * logits)
+ return torch.stack(outputs).sum(dim=0)
+```
+
+## Memory Optimization
+
+### Gradient checkpointing with LoRA
+
+```python
+from peft import prepare_model_for_kbit_training
+
+# Enable gradient checkpointing
+model = prepare_model_for_kbit_training(
+ model,
+ use_gradient_checkpointing=True,
+ gradient_checkpointing_kwargs={"use_reentrant": False}
+)
+```
+
+### CPU offloading for training
+
+```python
+from accelerate import Accelerator
+
+accelerator = Accelerator(
+ mixed_precision="bf16",
+ gradient_accumulation_steps=8,
+ cpu_offload=True # Offload optimizer states to CPU
+)
+
+model, optimizer, dataloader = accelerator.prepare(model, optimizer, dataloader)
+```
+
+### Memory-efficient attention with LoRA
+
+```python
+from transformers import AutoModelForCausalLM
+
+# Combine Flash Attention 2 with LoRA
+model = AutoModelForCausalLM.from_pretrained(
+ "meta-llama/Llama-3.1-8B",
+ attn_implementation="flash_attention_2",
+ torch_dtype=torch.bfloat16
+)
+
+# Apply LoRA
+model = get_peft_model(model, lora_config)
+```
+
+## Inference Optimization
+
+### Merge for deployment
+
+```python
+# Merge adapter weights into base model
+merged_model = model.merge_and_unload()
+
+# Quantize merged model for inference
+from transformers import BitsAndBytesConfig
+
+bnb_config = BitsAndBytesConfig(load_in_4bit=True)
+quantized_model = AutoModelForCausalLM.from_pretrained(
+ "./merged-model",
+ quantization_config=bnb_config
+)
+```
+
+### Export to different formats
+
+```python
+# Export to GGUF (llama.cpp)
+# First merge, then convert
+merged_model.save_pretrained("./merged-model")
+
+# Use llama.cpp converter
+# python convert-hf-to-gguf.py ./merged-model --outfile model.gguf
+
+# Export to ONNX
+from optimum.onnxruntime import ORTModelForCausalLM
+
+ort_model = ORTModelForCausalLM.from_pretrained(
+ "./merged-model",
+ export=True
+)
+ort_model.save_pretrained("./onnx-model")
+```
+
+### Batch adapter inference
+
+```python
+from vllm import LLM
+from vllm.lora.request import LoRARequest
+
+# Initialize with LoRA support
+llm = LLM(
+ model="meta-llama/Llama-3.1-8B",
+ enable_lora=True,
+ max_lora_rank=64,
+ max_loras=4 # Max concurrent adapters
+)
+
+# Batch with different adapters
+requests = [
+ ("prompt1", LoRARequest("adapter1", 1, "./adapter1")),
+ ("prompt2", LoRARequest("adapter2", 2, "./adapter2")),
+ ("prompt3", LoRARequest("adapter1", 1, "./adapter1")),
+]
+
+outputs = llm.generate(
+ [r[0] for r in requests],
+ lora_request=[r[1] for r in requests]
+)
+```
+
+## Training Recipes
+
+### Instruction tuning recipe
+
+```python
+lora_config = LoraConfig(
+ r=16,
+ lora_alpha=32,
+ lora_dropout=0.05,
+ target_modules="all-linear",
+ bias="none",
+ task_type="CAUSAL_LM"
+)
+
+training_args = TrainingArguments(
+ output_dir="./output",
+ num_train_epochs=3,
+ per_device_train_batch_size=4,
+ gradient_accumulation_steps=4,
+ learning_rate=2e-4,
+ lr_scheduler_type="cosine",
+ warmup_ratio=0.03,
+ bf16=True,
+ logging_steps=10,
+ save_strategy="steps",
+ save_steps=100,
+ eval_strategy="steps",
+ eval_steps=100,
+)
+```
+
+### Code generation recipe
+
+```python
+lora_config = LoraConfig(
+ r=32, # Higher rank for code
+ lora_alpha=64,
+ lora_dropout=0.1,
+ target_modules=["q_proj", "v_proj", "k_proj", "o_proj", "gate_proj", "up_proj", "down_proj"],
+ bias="none",
+ task_type="CAUSAL_LM"
+)
+
+training_args = TrainingArguments(
+ learning_rate=1e-4, # Lower LR for code
+ num_train_epochs=2,
+ max_seq_length=2048, # Longer sequences
+)
+```
+
+### Conversational/Chat recipe
+
+```python
+from trl import SFTTrainer
+
+lora_config = LoraConfig(
+ r=16,
+ lora_alpha=16, # alpha = r for chat
+ lora_dropout=0.05,
+ target_modules="all-linear"
+)
+
+# Use chat template
+def format_chat(example):
+ messages = [
+ {"role": "user", "content": example["instruction"]},
+ {"role": "assistant", "content": example["response"]}
+ ]
+ return tokenizer.apply_chat_template(messages, tokenize=False)
+
+trainer = SFTTrainer(
+ model=model,
+ peft_config=lora_config,
+ train_dataset=dataset.map(format_chat),
+ max_seq_length=1024,
+)
+```
+
+## Debugging and Validation
+
+### Verify adapter application
+
+```python
+# Check which modules have LoRA
+for name, module in model.named_modules():
+ if hasattr(module, "lora_A"):
+ print(f"LoRA applied to: {name}")
+
+# Print detailed config
+print(model.peft_config)
+
+# Check adapter state
+print(f"Active adapters: {model.active_adapters}")
+print(f"Trainable: {sum(p.numel() for p in model.parameters() if p.requires_grad)}")
+```
+
+### Compare with base model
+
+```python
+# Generate with adapter
+model.set_adapter("default")
+adapter_output = model.generate(**inputs)
+
+# Generate without adapter
+with model.disable_adapter():
+ base_output = model.generate(**inputs)
+
+print(f"Adapter: {tokenizer.decode(adapter_output[0])}")
+print(f"Base: {tokenizer.decode(base_output[0])}")
+```
+
+### Monitor training metrics
+
+```python
+from transformers import TrainerCallback
+
+class LoRACallback(TrainerCallback):
+ def on_log(self, args, state, control, logs=None, **kwargs):
+ if "loss" in logs:
+ # Log adapter-specific metrics
+ model = kwargs["model"]
+ lora_params = sum(p.numel() for n, p in model.named_parameters()
+ if "lora" in n and p.requires_grad)
+ print(f"Step {state.global_step}: loss={logs['loss']:.4f}, lora_params={lora_params}")
+```
diff --git a/skills/peft/references/troubleshooting.md b/skills/peft/references/troubleshooting.md
new file mode 100644
index 0000000..2200f75
--- /dev/null
+++ b/skills/peft/references/troubleshooting.md
@@ -0,0 +1,480 @@
+# PEFT Troubleshooting Guide
+
+## Installation Issues
+
+### bitsandbytes CUDA Error
+
+**Error**: `CUDA Setup failed despite GPU being available`
+
+**Fix**:
+```bash
+# Check CUDA version
+nvcc --version
+
+# Install matching bitsandbytes
+pip uninstall bitsandbytes
+pip install bitsandbytes --no-cache-dir
+
+# Or compile from source for specific CUDA
+git clone https://github.com/TimDettmers/bitsandbytes.git
+cd bitsandbytes
+CUDA_VERSION=118 make cuda11x # Adjust for your CUDA
+pip install .
+```
+
+### Triton Import Error
+
+**Error**: `ModuleNotFoundError: No module named 'triton'`
+
+**Fix**:
+```bash
+# Install triton (Linux only)
+pip install triton
+
+# Windows: Triton not supported, use CUDA backend
+# Set environment variable to disable triton
+export CUDA_VISIBLE_DEVICES=0
+```
+
+### PEFT Version Conflicts
+
+**Error**: `AttributeError: 'LoraConfig' object has no attribute 'use_dora'`
+
+**Fix**:
+```bash
+# Upgrade to latest PEFT
+pip install peft>=0.13.0 --upgrade
+
+# Check version
+python -c "import peft; print(peft.__version__)"
+```
+
+## Training Issues
+
+### CUDA Out of Memory
+
+**Error**: `torch.cuda.OutOfMemoryError: CUDA out of memory`
+
+**Solutions**:
+
+1. **Enable gradient checkpointing**:
+```python
+from peft import prepare_model_for_kbit_training
+model = prepare_model_for_kbit_training(model, use_gradient_checkpointing=True)
+```
+
+2. **Reduce batch size**:
+```python
+TrainingArguments(
+ per_device_train_batch_size=1,
+ gradient_accumulation_steps=16 # Maintain effective batch size
+)
+```
+
+3. **Use QLoRA**:
+```python
+from transformers import BitsAndBytesConfig
+
+bnb_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_quant_type="nf4",
+ bnb_4bit_use_double_quant=True
+)
+model = AutoModelForCausalLM.from_pretrained(model_name, quantization_config=bnb_config)
+```
+
+4. **Lower LoRA rank**:
+```python
+LoraConfig(r=8) # Instead of r=16 or higher
+```
+
+5. **Target fewer modules**:
+```python
+target_modules=["q_proj", "v_proj"] # Instead of all-linear
+```
+
+### Loss Not Decreasing
+
+**Problem**: Training loss stays flat or increases.
+
+**Solutions**:
+
+1. **Check learning rate**:
+```python
+# Start lower
+TrainingArguments(learning_rate=1e-4) # Not 2e-4 or higher
+```
+
+2. **Verify adapter is active**:
+```python
+model.print_trainable_parameters()
+# Should show >0 trainable params
+
+# Check adapter applied
+print(model.peft_config)
+```
+
+3. **Check data formatting**:
+```python
+# Verify tokenization
+sample = dataset[0]
+decoded = tokenizer.decode(sample["input_ids"])
+print(decoded) # Should look correct
+```
+
+4. **Increase rank**:
+```python
+LoraConfig(r=32, lora_alpha=64) # More capacity
+```
+
+### NaN Loss
+
+**Error**: `Loss is NaN`
+
+**Fix**:
+```python
+# Use bf16 instead of fp16
+TrainingArguments(bf16=True, fp16=False)
+
+# Or enable loss scaling
+TrainingArguments(fp16=True, fp16_full_eval=True)
+
+# Lower learning rate
+TrainingArguments(learning_rate=5e-5)
+
+# Check for data issues
+for batch in dataloader:
+ if torch.isnan(batch["input_ids"].float()).any():
+ print("NaN in input!")
+```
+
+### Adapter Not Training
+
+**Problem**: `trainable params: 0` or model not updating.
+
+**Fix**:
+```python
+# Verify LoRA applied to correct modules
+for name, module in model.named_modules():
+ if "lora" in name.lower():
+ print(f"Found LoRA: {name}")
+
+# Check target_modules match model architecture
+from peft.utils import TRANSFORMERS_MODELS_TO_LORA_TARGET_MODULES_MAPPING
+print(TRANSFORMERS_MODELS_TO_LORA_TARGET_MODULES_MAPPING.get(model.config.model_type))
+
+# Ensure model in training mode
+model.train()
+
+# Check requires_grad
+for name, param in model.named_parameters():
+ if param.requires_grad:
+ print(f"Trainable: {name}")
+```
+
+## Loading Issues
+
+### Adapter Loading Fails
+
+**Error**: `ValueError: Can't find adapter weights`
+
+**Fix**:
+```python
+# Check adapter files exist
+import os
+print(os.listdir("./adapter-path"))
+# Should contain: adapter_config.json, adapter_model.safetensors
+
+# Load with correct structure
+from peft import PeftModel, PeftConfig
+
+# Check config
+config = PeftConfig.from_pretrained("./adapter-path")
+print(config)
+
+# Load base model first
+base_model = AutoModelForCausalLM.from_pretrained(config.base_model_name_or_path)
+model = PeftModel.from_pretrained(base_model, "./adapter-path")
+```
+
+### Base Model Mismatch
+
+**Error**: `RuntimeError: size mismatch`
+
+**Fix**:
+```python
+# Ensure base model matches adapter
+from peft import PeftConfig
+
+config = PeftConfig.from_pretrained("./adapter-path")
+print(f"Base model: {config.base_model_name_or_path}")
+
+# Load exact same base model
+base_model = AutoModelForCausalLM.from_pretrained(config.base_model_name_or_path)
+```
+
+### Safetensors vs PyTorch Format
+
+**Error**: `ValueError: We couldn't connect to 'https://huggingface.co'`
+
+**Fix**:
+```python
+# Force local loading
+model = PeftModel.from_pretrained(
+ base_model,
+ "./adapter-path",
+ local_files_only=True
+)
+
+# Or specify format
+model.save_pretrained("./adapter", safe_serialization=True) # safetensors
+model.save_pretrained("./adapter", safe_serialization=False) # pytorch
+```
+
+## Inference Issues
+
+### Slow Generation
+
+**Problem**: Inference much slower than expected.
+
+**Solutions**:
+
+1. **Merge adapter for deployment**:
+```python
+merged_model = model.merge_and_unload()
+# No adapter overhead during inference
+```
+
+2. **Use optimized inference engine**:
+```python
+from vllm import LLM
+llm = LLM(model="./merged-model", dtype="half")
+```
+
+3. **Enable Flash Attention**:
+```python
+model = AutoModelForCausalLM.from_pretrained(
+ model_name,
+ attn_implementation="flash_attention_2"
+)
+```
+
+### Output Quality Issues
+
+**Problem**: Fine-tuned model produces worse outputs.
+
+**Solutions**:
+
+1. **Check evaluation without adapter**:
+```python
+with model.disable_adapter():
+ base_output = model.generate(**inputs)
+# Compare with adapter output
+```
+
+2. **Lower temperature during eval**:
+```python
+model.generate(**inputs, temperature=0.1, do_sample=False)
+```
+
+3. **Retrain with more data**:
+```python
+# Increase training samples
+# Use higher quality data
+# Train for more epochs
+```
+
+### Wrong Adapter Active
+
+**Problem**: Model using wrong adapter or no adapter.
+
+**Fix**:
+```python
+# Check active adapters
+print(model.active_adapters)
+
+# Explicitly set adapter
+model.set_adapter("your-adapter-name")
+
+# List all adapters
+print(model.peft_config.keys())
+```
+
+## QLoRA Specific Issues
+
+### Quantization Errors
+
+**Error**: `RuntimeError: mat1 and mat2 shapes cannot be multiplied`
+
+**Fix**:
+```python
+# Ensure compute dtype matches
+bnb_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_compute_dtype=torch.bfloat16, # Match model dtype
+ bnb_4bit_quant_type="nf4"
+)
+
+# Load with correct dtype
+model = AutoModelForCausalLM.from_pretrained(
+ model_name,
+ quantization_config=bnb_config,
+ torch_dtype=torch.bfloat16
+)
+```
+
+### QLoRA OOM
+
+**Error**: OOM even with 4-bit quantization.
+
+**Fix**:
+```python
+# Enable double quantization
+bnb_config = BitsAndBytesConfig(
+ load_in_4bit=True,
+ bnb_4bit_use_double_quant=True # Further memory reduction
+)
+
+# Use offloading
+model = AutoModelForCausalLM.from_pretrained(
+ model_name,
+ quantization_config=bnb_config,
+ device_map="auto",
+ max_memory={0: "20GB", "cpu": "100GB"}
+)
+```
+
+### QLoRA Merge Fails
+
+**Error**: `RuntimeError: expected scalar type BFloat16 but found Float`
+
+**Fix**:
+```python
+# Dequantize before merging
+from peft import PeftModel
+
+# Load in higher precision for merging
+base_model = AutoModelForCausalLM.from_pretrained(
+ base_model_name,
+ torch_dtype=torch.float16, # Not quantized
+ device_map="auto"
+)
+
+# Load adapter
+model = PeftModel.from_pretrained(base_model, "./qlora-adapter")
+
+# Now merge
+merged = model.merge_and_unload()
+```
+
+## Multi-Adapter Issues
+
+### Adapter Conflict
+
+**Error**: `ValueError: Adapter with name 'default' already exists`
+
+**Fix**:
+```python
+# Use unique names
+model.load_adapter("./adapter1", adapter_name="task1")
+model.load_adapter("./adapter2", adapter_name="task2")
+
+# Or delete existing
+model.delete_adapter("default")
+```
+
+### Mixed Precision Adapters
+
+**Error**: Adapters trained with different dtypes.
+
+**Fix**:
+```python
+# Convert adapter precision
+model = PeftModel.from_pretrained(base_model, "./adapter")
+model = model.to(torch.bfloat16)
+
+# Or load with specific dtype
+model = PeftModel.from_pretrained(
+ base_model,
+ "./adapter",
+ torch_dtype=torch.bfloat16
+)
+```
+
+## Performance Optimization
+
+### Memory Profiling
+
+```python
+import torch
+
+def print_memory():
+ if torch.cuda.is_available():
+ allocated = torch.cuda.memory_allocated() / 1e9
+ reserved = torch.cuda.memory_reserved() / 1e9
+ print(f"Allocated: {allocated:.2f}GB, Reserved: {reserved:.2f}GB")
+
+# Profile during training
+print_memory() # Before
+model.train()
+loss = model(**batch).loss
+loss.backward()
+print_memory() # After
+```
+
+### Speed Profiling
+
+```python
+import time
+import torch
+
+def benchmark_generation(model, tokenizer, prompt, n_runs=5):
+ inputs = tokenizer(prompt, return_tensors="pt").to(model.device)
+
+ # Warmup
+ model.generate(**inputs, max_new_tokens=10)
+ torch.cuda.synchronize()
+
+ # Benchmark
+ times = []
+ for _ in range(n_runs):
+ start = time.perf_counter()
+ outputs = model.generate(**inputs, max_new_tokens=100)
+ torch.cuda.synchronize()
+ times.append(time.perf_counter() - start)
+
+ tokens = outputs.shape[1] - inputs.input_ids.shape[1]
+ avg_time = sum(times) / len(times)
+ print(f"Speed: {tokens/avg_time:.2f} tokens/sec")
+
+# Compare adapter vs merged
+benchmark_generation(adapter_model, tokenizer, "Hello")
+benchmark_generation(merged_model, tokenizer, "Hello")
+```
+
+## Getting Help
+
+1. **Check PEFT GitHub Issues**: https://github.com/huggingface/peft/issues
+2. **HuggingFace Forums**: https://discuss.huggingface.co/
+3. **PEFT Documentation**: https://huggingface.co/docs/peft
+
+### Debugging Template
+
+When reporting issues, include:
+
+```python
+# System info
+import peft
+import transformers
+import torch
+
+print(f"PEFT: {peft.__version__}")
+print(f"Transformers: {transformers.__version__}")
+print(f"PyTorch: {torch.__version__}")
+print(f"CUDA: {torch.version.cuda}")
+print(f"GPU: {torch.cuda.get_device_name(0) if torch.cuda.is_available() else 'N/A'}")
+
+# Config
+print(model.peft_config)
+model.print_trainable_parameters()
+```
diff --git a/skills/ray-data/SKILL.md b/skills/ray-data/SKILL.md
new file mode 100644
index 0000000..50fc4ab
--- /dev/null
+++ b/skills/ray-data/SKILL.md
@@ -0,0 +1,326 @@
+---
+name: ray-data
+description: Scalable data processing for ML workloads. Streaming execution across CPU/GPU, supports Parquet/CSV/JSON/images. Integrates with Ray Train, PyTorch, TensorFlow. Scales from single machine to 100s of nodes. Use for batch inference, data preprocessing, multi-modal data loading, or distributed ETL pipelines.
+version: 1.0.0
+author: Orchestra Research
+license: MIT
+tags: [Data Processing, Ray Data, Distributed Computing, ML Pipelines, Batch Inference, ETL, Scalable, Ray, PyTorch, TensorFlow]
+dependencies: ["ray[data]", pyarrow, pandas]
+---
+
+# Ray Data - Scalable ML Data Processing
+
+Distributed data processing library for ML and AI workloads.
+
+## When to use Ray Data
+
+**Use Ray Data when:**
+- Processing large datasets (>100GB) for ML training
+- Need distributed data preprocessing across cluster
+- Building batch inference pipelines
+- Loading multi-modal data (images, audio, video)
+- Scaling data processing from laptop to cluster
+
+**Key features**:
+- **Streaming execution**: Process data larger than memory
+- **GPU support**: Accelerate transforms with GPUs
+- **Framework integration**: PyTorch, TensorFlow, HuggingFace
+- **Multi-modal**: Images, Parquet, CSV, JSON, audio, video
+
+**Use alternatives instead**:
+- **Pandas**: Small data (<1GB) on single machine
+- **Dask**: Tabular data, SQL-like operations
+- **Spark**: Enterprise ETL, SQL queries
+
+## Quick start
+
+### Installation
+
+```bash
+pip install -U 'ray[data]'
+```
+
+### Load and transform data
+
+```python
+import ray
+
+# Read Parquet files
+ds = ray.data.read_parquet("s3://bucket/data/*.parquet")
+
+# Transform data (lazy execution)
+ds = ds.map_batches(lambda batch: {"processed": batch["text"].str.lower()})
+
+# Consume data
+for batch in ds.iter_batches(batch_size=100):
+ print(batch)
+```
+
+### Integration with Ray Train
+
+```python
+import ray
+from ray.train import ScalingConfig
+from ray.train.torch import TorchTrainer
+
+# Create dataset
+train_ds = ray.data.read_parquet("s3://bucket/train/*.parquet")
+
+def train_func(config):
+ # Access dataset in training
+ train_ds = ray.train.get_dataset_shard("train")
+
+ for epoch in range(10):
+ for batch in train_ds.iter_batches(batch_size=32):
+ # Train on batch
+ pass
+
+# Train with Ray
+trainer = TorchTrainer(
+ train_func,
+ datasets={"train": train_ds},
+ scaling_config=ScalingConfig(num_workers=4, use_gpu=True)
+)
+trainer.fit()
+```
+
+## Reading data
+
+### From cloud storage
+
+```python
+import ray
+
+# Parquet (recommended for ML)
+ds = ray.data.read_parquet("s3://bucket/data/*.parquet")
+
+# CSV
+ds = ray.data.read_csv("s3://bucket/data/*.csv")
+
+# JSON
+ds = ray.data.read_json("gs://bucket/data/*.json")
+
+# Images
+ds = ray.data.read_images("s3://bucket/images/")
+```
+
+### From Python objects
+
+```python
+# From list
+ds = ray.data.from_items([{"id": i, "value": i * 2} for i in range(1000)])
+
+# From range
+ds = ray.data.range(1000000) # Synthetic data
+
+# From pandas
+import pandas as pd
+df = pd.DataFrame({"col1": [1, 2, 3], "col2": [4, 5, 6]})
+ds = ray.data.from_pandas(df)
+```
+
+## Transformations
+
+### Map batches (vectorized)
+
+```python
+# Batch transformation (fast)
+def process_batch(batch):
+ batch["doubled"] = batch["value"] * 2
+ return batch
+
+ds = ds.map_batches(process_batch, batch_size=1000)
+```
+
+### Row transformations
+
+```python
+# Row-by-row (slower)
+def process_row(row):
+ row["squared"] = row["value"] ** 2
+ return row
+
+ds = ds.map(process_row)
+```
+
+### Filter
+
+```python
+# Filter rows
+ds = ds.filter(lambda row: row["value"] > 100)
+```
+
+### Group by and aggregate
+
+```python
+# Group by column
+ds = ds.groupby("category").count()
+
+# Custom aggregation
+ds = ds.groupby("category").map_groups(lambda group: {"sum": group["value"].sum()})
+```
+
+## GPU-accelerated transforms
+
+```python
+# Use GPU for preprocessing
+def preprocess_images_gpu(batch):
+ import torch
+ images = torch.tensor(batch["image"]).cuda()
+ # GPU preprocessing
+ processed = images * 255
+ return {"processed": processed.cpu().numpy()}
+
+ds = ds.map_batches(
+ preprocess_images_gpu,
+ batch_size=64,
+ num_gpus=1 # Request GPU
+)
+```
+
+## Writing data
+
+```python
+# Write to Parquet
+ds.write_parquet("s3://bucket/output/")
+
+# Write to CSV
+ds.write_csv("output/")
+
+# Write to JSON
+ds.write_json("output/")
+```
+
+## Performance optimization
+
+### Repartition
+
+```python
+# Control parallelism
+ds = ds.repartition(100) # 100 blocks for 100-core cluster
+```
+
+### Batch size tuning
+
+```python
+# Larger batches = faster vectorized ops
+ds.map_batches(process_fn, batch_size=10000) # vs batch_size=100
+```
+
+### Streaming execution
+
+```python
+# Process data larger than memory
+ds = ray.data.read_parquet("s3://huge-dataset/")
+for batch in ds.iter_batches(batch_size=1000):
+ process(batch) # Streamed, not loaded to memory
+```
+
+## Common patterns
+
+### Batch inference
+
+```python
+import ray
+
+# Load model
+def load_model():
+ # Load once per worker
+ return MyModel()
+
+# Inference function
+class BatchInference:
+ def __init__(self):
+ self.model = load_model()
+
+ def __call__(self, batch):
+ predictions = self.model(batch["input"])
+ return {"prediction": predictions}
+
+# Run distributed inference
+ds = ray.data.read_parquet("s3://data/")
+predictions = ds.map_batches(BatchInference, batch_size=32, num_gpus=1)
+predictions.write_parquet("s3://output/")
+```
+
+### Data preprocessing pipeline
+
+```python
+# Multi-step pipeline
+ds = (
+ ray.data.read_parquet("s3://raw/")
+ .map_batches(clean_data)
+ .map_batches(tokenize)
+ .map_batches(augment)
+ .write_parquet("s3://processed/")
+)
+```
+
+## Integration with ML frameworks
+
+### PyTorch
+
+```python
+# Convert to PyTorch
+torch_ds = ds.to_torch(label_column="label", batch_size=32)
+
+for batch in torch_ds:
+ # batch is dict with tensors
+ inputs, labels = batch["features"], batch["label"]
+```
+
+### TensorFlow
+
+```python
+# Convert to TensorFlow
+tf_ds = ds.to_tf(feature_columns=["image"], label_column="label", batch_size=32)
+
+for features, labels in tf_ds:
+ # Train model
+ pass
+```
+
+## Supported data formats
+
+| Format | Read | Write | Use Case |
+|--------|------|-------|----------|
+| Parquet | ✅ | ✅ | ML data (recommended) |
+| CSV | ✅ | ✅ | Tabular data |
+| JSON | ✅ | ✅ | Semi-structured |
+| Images | ✅ | ❌ | Computer vision |
+| NumPy | ✅ | ✅ | Arrays |
+| Pandas | ✅ | ❌ | DataFrames |
+
+## Performance benchmarks
+
+**Scaling** (processing 100GB data):
+- 1 node (16 cores): ~30 minutes
+- 4 nodes (64 cores): ~8 minutes
+- 16 nodes (256 cores): ~2 minutes
+
+**GPU acceleration** (image preprocessing):
+- CPU only: 1,000 images/sec
+- 1 GPU: 5,000 images/sec
+- 4 GPUs: 18,000 images/sec
+
+## Use cases
+
+**Production deployments**:
+- **Pinterest**: Last-mile data processing for model training
+- **ByteDance**: Scaling offline inference with multi-modal LLMs
+- **Spotify**: ML platform for batch inference
+
+## References
+
+- **[Transformations Guide](references/transformations.md)** - Map, filter, groupby operations
+- **[Integration Guide](references/integration.md)** - Ray Train, PyTorch, TensorFlow
+
+## Resources
+
+- **Docs**: https://docs.ray.io/en/latest/data/data.html
+- **GitHub**: https://github.com/ray-project/ray ⭐ 36,000+
+- **Version**: Ray 2.40.0+
+- **Examples**: https://docs.ray.io/en/latest/data/examples/overview.html
+
+
+
diff --git a/skills/ray-data/references/integration.md b/skills/ray-data/references/integration.md
new file mode 100644
index 0000000..adcaa78
--- /dev/null
+++ b/skills/ray-data/references/integration.md
@@ -0,0 +1,82 @@
+# Ray Data Integration Guide
+
+Integration with Ray Train and ML frameworks.
+
+## Ray Train integration
+
+### Basic training with datasets
+
+```python
+import ray
+from ray.train import ScalingConfig
+from ray.train.torch import TorchTrainer
+
+# Create datasets
+train_ds = ray.data.read_parquet("s3://data/train/")
+val_ds = ray.data.read_parquet("s3://data/val/")
+
+def train_func(config):
+ # Get dataset shards
+ train_ds = ray.train.get_dataset_shard("train")
+ val_ds = ray.train.get_dataset_shard("val")
+
+ for epoch in range(config["epochs"]):
+ # Iterate over batches
+ for batch in train_ds.iter_batches(batch_size=32):
+ # Train on batch
+ pass
+
+# Launch training
+trainer = TorchTrainer(
+ train_func,
+ train_loop_config={"epochs": 10},
+ datasets={"train": train_ds, "val": val_ds},
+ scaling_config=ScalingConfig(num_workers=4, use_gpu=True)
+)
+
+result = trainer.fit()
+```
+
+## PyTorch integration
+
+### Convert to PyTorch Dataset
+
+```python
+# Option 1: to_torch (recommended)
+torch_ds = ds.to_torch(
+ label_column="label",
+ batch_size=32,
+ drop_last=True
+)
+
+for batch in torch_ds:
+ inputs = batch["features"]
+ labels = batch["label"]
+ # Train model
+
+# Option 2: iter_torch_batches
+for batch in ds.iter_torch_batches(batch_size=32):
+ # batch is dict of tensors
+ pass
+```
+
+## TensorFlow integration
+
+```python
+tf_ds = ds.to_tf(
+ feature_columns=["image", "text"],
+ label_column="label",
+ batch_size=32
+)
+
+for features, labels in tf_ds:
+ # Train TensorFlow model
+ pass
+```
+
+## Best practices
+
+1. **Shard datasets in Ray Train** - Automatic with `get_dataset_shard()`
+2. **Use streaming** - Don't load entire dataset to memory
+3. **Preprocess in Ray Data** - Distribute preprocessing across cluster
+4. **Cache preprocessed data** - Write to Parquet, read in training
diff --git a/skills/ray-data/references/transformations.md b/skills/ray-data/references/transformations.md
new file mode 100644
index 0000000..6c8bc44
--- /dev/null
+++ b/skills/ray-data/references/transformations.md
@@ -0,0 +1,83 @@
+# Ray Data Transformations
+
+Complete guide to data transformations in Ray Data.
+
+## Core operations
+
+### Map batches (vectorized)
+
+```python
+# Recommended for performance
+def process_batch(batch):
+ # batch is dict of numpy arrays or pandas Series
+ batch["doubled"] = batch["value"] * 2
+ return batch
+
+ds = ds.map_batches(process_batch, batch_size=1000)
+```
+
+**Performance**: 10-100× faster than row-by-row
+
+### Map (row-by-row)
+
+```python
+# Use only when vectorization not possible
+def process_row(row):
+ row["squared"] = row["value"] ** 2
+ return row
+
+ds = ds.map(process_row)
+```
+
+### Filter
+
+```python
+# Remove rows
+ds = ds.filter(lambda row: row["score"] > 0.5)
+```
+
+### Flat map
+
+```python
+# One row → multiple rows
+def expand_row(row):
+ return [{"value": row["value"] + i} for i in range(3)]
+
+ds = ds.flat_map(expand_row)
+```
+
+## GPU-accelerated transforms
+
+```python
+def gpu_transform(batch):
+ import torch
+ data = torch.tensor(batch["data"]).cuda()
+ # GPU processing
+ result = data * 2
+ return {"processed": result.cpu().numpy()}
+
+ds = ds.map_batches(gpu_transform, num_gpus=1, batch_size=64)
+```
+
+## Groupby operations
+
+```python
+# Group by column
+grouped = ds.groupby("category")
+
+# Aggregate
+result = grouped.count()
+
+# Custom aggregation
+result = grouped.map_groups(lambda group: {
+ "sum": group["value"].sum(),
+ "mean": group["value"].mean()
+})
+```
+
+## Best practices
+
+1. **Use map_batches over map** - 10-100× faster
+2. **Tune batch_size** - Larger = faster (balance with memory)
+3. **Use GPUs for heavy compute** - Image/audio preprocessing
+4. **Stream large datasets** - Use iter_batches for >memory data
diff --git a/skills/skill-creator/LICENSE.txt b/skills/skill-creator/LICENSE.txt
new file mode 100644
index 0000000..7a4a3ea
--- /dev/null
+++ b/skills/skill-creator/LICENSE.txt
@@ -0,0 +1,202 @@
+
+ Apache License
+ Version 2.0, January 2004
+ http://www.apache.org/licenses/
+
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
+
+ 1. Definitions.
+
+ "License" shall mean the terms and conditions for use, reproduction,
+ and distribution as defined by Sections 1 through 9 of this document.
+
+ "Licensor" shall mean the copyright owner or entity authorized by
+ the copyright owner that is granting the License.
+
+ "Legal Entity" shall mean the union of the acting entity and all
+ other entities that control, are controlled by, or are under common
+ control with that entity. For the purposes of this definition,
+ "control" means (i) the power, direct or indirect, to cause the
+ direction or management of such entity, whether by contract or
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
+ outstanding shares, or (iii) beneficial ownership of such entity.
+
+ "You" (or "Your") shall mean an individual or Legal Entity
+ exercising permissions granted by this License.
+
+ "Source" form shall mean the preferred form for making modifications,
+ including but not limited to software source code, documentation
+ source, and configuration files.
+
+ "Object" form shall mean any form resulting from mechanical
+ transformation or translation of a Source form, including but
+ not limited to compiled object code, generated documentation,
+ and conversions to other media types.
+
+ "Work" shall mean the work of authorship, whether in Source or
+ Object form, made available under the License, as indicated by a
+ copyright notice that is included in or attached to the work
+ (an example is provided in the Appendix below).
+
+ "Derivative Works" shall mean any work, whether in Source or Object
+ form, that is based on (or derived from) the Work and for which the
+ editorial revisions, annotations, elaborations, or other modifications
+ represent, as a whole, an original work of authorship. For the purposes
+ of this License, Derivative Works shall not include works that remain
+ separable from, or merely link (or bind by name) to the interfaces of,
+ the Work and Derivative Works thereof.
+
+ "Contribution" shall mean any work of authorship, including
+ the original version of the Work and any modifications or additions
+ to that Work or Derivative Works thereof, that is intentionally
+ submitted to Licensor for inclusion in the Work by the copyright owner
+ or by an individual or Legal Entity authorized to submit on behalf of
+ the copyright owner. For the purposes of this definition, "submitted"
+ means any form of electronic, verbal, or written communication sent
+ to the Licensor or its representatives, including but not limited to
+ communication on electronic mailing lists, source code control systems,
+ and issue tracking systems that are managed by, or on behalf of, the
+ Licensor for the purpose of discussing and improving the Work, but
+ excluding communication that is conspicuously marked or otherwise
+ designated in writing by the copyright owner as "Not a Contribution."
+
+ "Contributor" shall mean Licensor and any individual or Legal Entity
+ on behalf of whom a Contribution has been received by Licensor and
+ subsequently incorporated within the Work.
+
+ 2. Grant of Copyright License. Subject to the terms and conditions of
+ this License, each Contributor hereby grants to You a perpetual,
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+ copyright license to reproduce, prepare Derivative Works of,
+ publicly display, publicly perform, sublicense, and distribute the
+ Work and such Derivative Works in Source or Object form.
+
+ 3. Grant of Patent License. Subject to the terms and conditions of
+ this License, each Contributor hereby grants to You a perpetual,
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+ (except as stated in this section) patent license to make, have made,
+ use, offer to sell, sell, import, and otherwise transfer the Work,
+ where such license applies only to those patent claims licensable
+ by such Contributor that are necessarily infringed by their
+ Contribution(s) alone or by combination of their Contribution(s)
+ with the Work to which such Contribution(s) was submitted. If You
+ institute patent litigation against any entity (including a
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
+ or a Contribution incorporated within the Work constitutes direct
+ or contributory patent infringement, then any patent licenses
+ granted to You under this License for that Work shall terminate
+ as of the date such litigation is filed.
+
+ 4. Redistribution. You may reproduce and distribute copies of the
+ Work or Derivative Works thereof in any medium, with or without
+ modifications, and in Source or Object form, provided that You
+ meet the following conditions:
+
+ (a) You must give any other recipients of the Work or
+ Derivative Works a copy of this License; and
+
+ (b) You must cause any modified files to carry prominent notices
+ stating that You changed the files; and
+
+ (c) You must retain, in the Source form of any Derivative Works
+ that You distribute, all copyright, patent, trademark, and
+ attribution notices from the Source form of the Work,
+ excluding those notices that do not pertain to any part of
+ the Derivative Works; and
+
+ (d) If the Work includes a "NOTICE" text file as part of its
+ distribution, then any Derivative Works that You distribute must
+ include a readable copy of the attribution notices contained
+ within such NOTICE file, excluding those notices that do not
+ pertain to any part of the Derivative Works, in at least one
+ of the following places: within a NOTICE text file distributed
+ as part of the Derivative Works; within the Source form or
+ documentation, if provided along with the Derivative Works; or,
+ within a display generated by the Derivative Works, if and
+ wherever such third-party notices normally appear. The contents
+ of the NOTICE file are for informational purposes only and
+ do not modify the License. You may add Your own attribution
+ notices within Derivative Works that You distribute, alongside
+ or as an addendum to the NOTICE text from the Work, provided
+ that such additional attribution notices cannot be construed
+ as modifying the License.
+
+ You may add Your own copyright statement to Your modifications and
+ may provide additional or different license terms and conditions
+ for use, reproduction, or distribution of Your modifications, or
+ for any such Derivative Works as a whole, provided Your use,
+ reproduction, and distribution of the Work otherwise complies with
+ the conditions stated in this License.
+
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
+ any Contribution intentionally submitted for inclusion in the Work
+ by You to the Licensor shall be under the terms and conditions of
+ this License, without any additional terms or conditions.
+ Notwithstanding the above, nothing herein shall supersede or modify
+ the terms of any separate license agreement you may have executed
+ with Licensor regarding such Contributions.
+
+ 6. Trademarks. This License does not grant permission to use the trade
+ names, trademarks, service marks, or product names of the Licensor,
+ except as required for reasonable and customary use in describing the
+ origin of the Work and reproducing the content of the NOTICE file.
+
+ 7. Disclaimer of Warranty. Unless required by applicable law or
+ agreed to in writing, Licensor provides the Work (and each
+ Contributor provides its Contributions) on an "AS IS" BASIS,
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
+ implied, including, without limitation, any warranties or conditions
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
+ PARTICULAR PURPOSE. You are solely responsible for determining the
+ appropriateness of using or redistributing the Work and assume any
+ risks associated with Your exercise of permissions under this License.
+
+ 8. Limitation of Liability. In no event and under no legal theory,
+ whether in tort (including negligence), contract, or otherwise,
+ unless required by applicable law (such as deliberate and grossly
+ negligent acts) or agreed to in writing, shall any Contributor be
+ liable to You for damages, including any direct, indirect, special,
+ incidental, or consequential damages of any character arising as a
+ result of this License or out of the use or inability to use the
+ Work (including but not limited to damages for loss of goodwill,
+ work stoppage, computer failure or malfunction, or any and all
+ other commercial damages or losses), even if such Contributor
+ has been advised of the possibility of such damages.
+
+ 9. Accepting Warranty or Additional Liability. While redistributing
+ the Work or Derivative Works thereof, You may choose to offer,
+ and charge a fee for, acceptance of support, warranty, indemnity,
+ or other liability obligations and/or rights consistent with this
+ License. However, in accepting such obligations, You may act only
+ on Your own behalf and on Your sole responsibility, not on behalf
+ of any other Contributor, and only if You agree to indemnify,
+ defend, and hold each Contributor harmless for any liability
+ incurred by, or claims asserted against, such Contributor by reason
+ of your accepting any such warranty or additional liability.
+
+ END OF TERMS AND CONDITIONS
+
+ APPENDIX: How to apply the Apache License to your work.
+
+ To apply the Apache License to your work, attach the following
+ boilerplate notice, with the fields enclosed by brackets "[]"
+ replaced with your own identifying information. (Don't include
+ the brackets!) The text should be enclosed in the appropriate
+ comment syntax for the file format. We also recommend that a
+ file or class name and description of purpose be included on the
+ same "printed page" as the copyright notice for easier
+ identification within third-party archives.
+
+ Copyright [yyyy] [name of copyright owner]
+
+ Licensed under the Apache License, Version 2.0 (the "License");
+ you may not use this file except in compliance with the License.
+ You may obtain a copy of the License at
+
+ http://www.apache.org/licenses/LICENSE-2.0
+
+ Unless required by applicable law or agreed to in writing, software
+ distributed under the License is distributed on an "AS IS" BASIS,
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ See the License for the specific language governing permissions and
+ limitations under the License.
\ No newline at end of file
diff --git a/skills/skill-creator/SKILL.md b/skills/skill-creator/SKILL.md
new file mode 100644
index 0000000..b7f8659
--- /dev/null
+++ b/skills/skill-creator/SKILL.md
@@ -0,0 +1,356 @@
+---
+name: skill-creator
+description: Guide for creating effective skills. This skill should be used when users want to create a new skill (or update an existing skill) that extends Claude's capabilities with specialized knowledge, workflows, or tool integrations.
+license: Complete terms in LICENSE.txt
+---
+
+# Skill Creator
+
+This skill provides guidance for creating effective skills.
+
+## About Skills
+
+Skills are modular, self-contained packages that extend Claude's capabilities by providing
+specialized knowledge, workflows, and tools. Think of them as "onboarding guides" for specific
+domains or tasks—they transform Claude from a general-purpose agent into a specialized agent
+equipped with procedural knowledge that no model can fully possess.
+
+### What Skills Provide
+
+1. Specialized workflows - Multi-step procedures for specific domains
+2. Tool integrations - Instructions for working with specific file formats or APIs
+3. Domain expertise - Company-specific knowledge, schemas, business logic
+4. Bundled resources - Scripts, references, and assets for complex and repetitive tasks
+
+## Core Principles
+
+### Concise is Key
+
+The context window is a public good. Skills share the context window with everything else Claude needs: system prompt, conversation history, other Skills' metadata, and the actual user request.
+
+**Default assumption: Claude is already very smart.** Only add context Claude doesn't already have. Challenge each piece of information: "Does Claude really need this explanation?" and "Does this paragraph justify its token cost?"
+
+Prefer concise examples over verbose explanations.
+
+### Set Appropriate Degrees of Freedom
+
+Match the level of specificity to the task's fragility and variability:
+
+**High freedom (text-based instructions)**: Use when multiple approaches are valid, decisions depend on context, or heuristics guide the approach.
+
+**Medium freedom (pseudocode or scripts with parameters)**: Use when a preferred pattern exists, some variation is acceptable, or configuration affects behavior.
+
+**Low freedom (specific scripts, few parameters)**: Use when operations are fragile and error-prone, consistency is critical, or a specific sequence must be followed.
+
+Think of Claude as exploring a path: a narrow bridge with cliffs needs specific guardrails (low freedom), while an open field allows many routes (high freedom).
+
+### Anatomy of a Skill
+
+Every skill consists of a required SKILL.md file and optional bundled resources:
+
+```
+skill-name/
+├── SKILL.md (required)
+│ ├── YAML frontmatter metadata (required)
+│ │ ├── name: (required)
+│ │ └── description: (required)
+│ └── Markdown instructions (required)
+└── Bundled Resources (optional)
+ ├── scripts/ - Executable code (Python/Bash/etc.)
+ ├── references/ - Documentation intended to be loaded into context as needed
+ └── assets/ - Files used in output (templates, icons, fonts, etc.)
+```
+
+#### SKILL.md (required)
+
+Every SKILL.md consists of:
+
+- **Frontmatter** (YAML): Contains `name` and `description` fields. These are the only fields that Claude reads to determine when the skill gets used, thus it is very important to be clear and comprehensive in describing what the skill is, and when it should be used.
+- **Body** (Markdown): Instructions and guidance for using the skill. Only loaded AFTER the skill triggers (if at all).
+
+#### Bundled Resources (optional)
+
+##### Scripts (`scripts/`)
+
+Executable code (Python/Bash/etc.) for tasks that require deterministic reliability or are repeatedly rewritten.
+
+- **When to include**: When the same code is being rewritten repeatedly or deterministic reliability is needed
+- **Example**: `scripts/rotate_pdf.py` for PDF rotation tasks
+- **Benefits**: Token efficient, deterministic, may be executed without loading into context
+- **Note**: Scripts may still need to be read by Claude for patching or environment-specific adjustments
+
+##### References (`references/`)
+
+Documentation and reference material intended to be loaded as needed into context to inform Claude's process and thinking.
+
+- **When to include**: For documentation that Claude should reference while working
+- **Examples**: `references/finance.md` for financial schemas, `references/mnda.md` for company NDA template, `references/policies.md` for company policies, `references/api_docs.md` for API specifications
+- **Use cases**: Database schemas, API documentation, domain knowledge, company policies, detailed workflow guides
+- **Benefits**: Keeps SKILL.md lean, loaded only when Claude determines it's needed
+- **Best practice**: If files are large (>10k words), include grep search patterns in SKILL.md
+- **Avoid duplication**: Information should live in either SKILL.md or references files, not both. Prefer references files for detailed information unless it's truly core to the skill—this keeps SKILL.md lean while making information discoverable without hogging the context window. Keep only essential procedural instructions and workflow guidance in SKILL.md; move detailed reference material, schemas, and examples to references files.
+
+##### Assets (`assets/`)
+
+Files not intended to be loaded into context, but rather used within the output Claude produces.
+
+- **When to include**: When the skill needs files that will be used in the final output
+- **Examples**: `assets/logo.png` for brand assets, `assets/slides.pptx` for PowerPoint templates, `assets/frontend-template/` for HTML/React boilerplate, `assets/font.ttf` for typography
+- **Use cases**: Templates, images, icons, boilerplate code, fonts, sample documents that get copied or modified
+- **Benefits**: Separates output resources from documentation, enables Claude to use files without loading them into context
+
+#### What to Not Include in a Skill
+
+A skill should only contain essential files that directly support its functionality. Do NOT create extraneous documentation or auxiliary files, including:
+
+- README.md
+- INSTALLATION_GUIDE.md
+- QUICK_REFERENCE.md
+- CHANGELOG.md
+- etc.
+
+The skill should only contain the information needed for an AI agent to do the job at hand. It should not contain auxilary context about the process that went into creating it, setup and testing procedures, user-facing documentation, etc. Creating additional documentation files just adds clutter and confusion.
+
+### Progressive Disclosure Design Principle
+
+Skills use a three-level loading system to manage context efficiently:
+
+1. **Metadata (name + description)** - Always in context (~100 words)
+2. **SKILL.md body** - When skill triggers (<5k words)
+3. **Bundled resources** - As needed by Claude (Unlimited because scripts can be executed without reading into context window)
+
+#### Progressive Disclosure Patterns
+
+Keep SKILL.md body to the essentials and under 500 lines to minimize context bloat. Split content into separate files when approaching this limit. When splitting out content into other files, it is very important to reference them from SKILL.md and describe clearly when to read them, to ensure the reader of the skill knows they exist and when to use them.
+
+**Key principle:** When a skill supports multiple variations, frameworks, or options, keep only the core workflow and selection guidance in SKILL.md. Move variant-specific details (patterns, examples, configuration) into separate reference files.
+
+**Pattern 1: High-level guide with references**
+
+```markdown
+# PDF Processing
+
+## Quick start
+
+Extract text with pdfplumber:
+[code example]
+
+## Advanced features
+
+- **Form filling**: See [FORMS.md](FORMS.md) for complete guide
+- **API reference**: See [REFERENCE.md](REFERENCE.md) for all methods
+- **Examples**: See [EXAMPLES.md](EXAMPLES.md) for common patterns
+```
+
+Claude loads FORMS.md, REFERENCE.md, or EXAMPLES.md only when needed.
+
+**Pattern 2: Domain-specific organization**
+
+For Skills with multiple domains, organize content by domain to avoid loading irrelevant context:
+
+```
+bigquery-skill/
+├── SKILL.md (overview and navigation)
+└── reference/
+ ├── finance.md (revenue, billing metrics)
+ ├── sales.md (opportunities, pipeline)
+ ├── product.md (API usage, features)
+ └── marketing.md (campaigns, attribution)
+```
+
+When a user asks about sales metrics, Claude only reads sales.md.
+
+Similarly, for skills supporting multiple frameworks or variants, organize by variant:
+
+```
+cloud-deploy/
+├── SKILL.md (workflow + provider selection)
+└── references/
+ ├── aws.md (AWS deployment patterns)
+ ├── gcp.md (GCP deployment patterns)
+ └── azure.md (Azure deployment patterns)
+```
+
+When the user chooses AWS, Claude only reads aws.md.
+
+**Pattern 3: Conditional details**
+
+Show basic content, link to advanced content:
+
+```markdown
+# DOCX Processing
+
+## Creating documents
+
+Use docx-js for new documents. See [DOCX-JS.md](DOCX-JS.md).
+
+## Editing documents
+
+For simple edits, modify the XML directly.
+
+**For tracked changes**: See [REDLINING.md](REDLINING.md)
+**For OOXML details**: See [OOXML.md](OOXML.md)
+```
+
+Claude reads REDLINING.md or OOXML.md only when the user needs those features.
+
+**Important guidelines:**
+
+- **Avoid deeply nested references** - Keep references one level deep from SKILL.md. All reference files should link directly from SKILL.md.
+- **Structure longer reference files** - For files longer than 100 lines, include a table of contents at the top so Claude can see the full scope when previewing.
+
+## Skill Creation Process
+
+Skill creation involves these steps:
+
+1. Understand the skill with concrete examples
+2. Plan reusable skill contents (scripts, references, assets)
+3. Initialize the skill (run init_skill.py)
+4. Edit the skill (implement resources and write SKILL.md)
+5. Package the skill (run package_skill.py)
+6. Iterate based on real usage
+
+Follow these steps in order, skipping only if there is a clear reason why they are not applicable.
+
+### Step 1: Understanding the Skill with Concrete Examples
+
+Skip this step only when the skill's usage patterns are already clearly understood. It remains valuable even when working with an existing skill.
+
+To create an effective skill, clearly understand concrete examples of how the skill will be used. This understanding can come from either direct user examples or generated examples that are validated with user feedback.
+
+For example, when building an image-editor skill, relevant questions include:
+
+- "What functionality should the image-editor skill support? Editing, rotating, anything else?"
+- "Can you give some examples of how this skill would be used?"
+- "I can imagine users asking for things like 'Remove the red-eye from this image' or 'Rotate this image'. Are there other ways you imagine this skill being used?"
+- "What would a user say that should trigger this skill?"
+
+To avoid overwhelming users, avoid asking too many questions in a single message. Start with the most important questions and follow up as needed for better effectiveness.
+
+Conclude this step when there is a clear sense of the functionality the skill should support.
+
+### Step 2: Planning the Reusable Skill Contents
+
+To turn concrete examples into an effective skill, analyze each example by:
+
+1. Considering how to execute on the example from scratch
+2. Identifying what scripts, references, and assets would be helpful when executing these workflows repeatedly
+
+Example: When building a `pdf-editor` skill to handle queries like "Help me rotate this PDF," the analysis shows:
+
+1. Rotating a PDF requires re-writing the same code each time
+2. A `scripts/rotate_pdf.py` script would be helpful to store in the skill
+
+Example: When designing a `frontend-webapp-builder` skill for queries like "Build me a todo app" or "Build me a dashboard to track my steps," the analysis shows:
+
+1. Writing a frontend webapp requires the same boilerplate HTML/React each time
+2. An `assets/hello-world/` template containing the boilerplate HTML/React project files would be helpful to store in the skill
+
+Example: When building a `big-query` skill to handle queries like "How many users have logged in today?" the analysis shows:
+
+1. Querying BigQuery requires re-discovering the table schemas and relationships each time
+2. A `references/schema.md` file documenting the table schemas would be helpful to store in the skill
+
+To establish the skill's contents, analyze each concrete example to create a list of the reusable resources to include: scripts, references, and assets.
+
+### Step 3: Initializing the Skill
+
+At this point, it is time to actually create the skill.
+
+Skip this step only if the skill being developed already exists, and iteration or packaging is needed. In this case, continue to the next step.
+
+When creating a new skill from scratch, always run the `init_skill.py` script. The script conveniently generates a new template skill directory that automatically includes everything a skill requires, making the skill creation process much more efficient and reliable.
+
+Usage:
+
+```bash
+scripts/init_skill.py --path
+```
+
+The script:
+
+- Creates the skill directory at the specified path
+- Generates a SKILL.md template with proper frontmatter and TODO placeholders
+- Creates example resource directories: `scripts/`, `references/`, and `assets/`
+- Adds example files in each directory that can be customized or deleted
+
+After initialization, customize or remove the generated SKILL.md and example files as needed.
+
+### Step 4: Edit the Skill
+
+When editing the (newly-generated or existing) skill, remember that the skill is being created for another instance of Claude to use. Include information that would be beneficial and non-obvious to Claude. Consider what procedural knowledge, domain-specific details, or reusable assets would help another Claude instance execute these tasks more effectively.
+
+#### Learn Proven Design Patterns
+
+Consult these helpful guides based on your skill's needs:
+
+- **Multi-step processes**: See references/workflows.md for sequential workflows and conditional logic
+- **Specific output formats or quality standards**: See references/output-patterns.md for template and example patterns
+
+These files contain established best practices for effective skill design.
+
+#### Start with Reusable Skill Contents
+
+To begin implementation, start with the reusable resources identified above: `scripts/`, `references/`, and `assets/` files. Note that this step may require user input. For example, when implementing a `brand-guidelines` skill, the user may need to provide brand assets or templates to store in `assets/`, or documentation to store in `references/`.
+
+Added scripts must be tested by actually running them to ensure there are no bugs and that the output matches what is expected. If there are many similar scripts, only a representative sample needs to be tested to ensure confidence that they all work while balancing time to completion.
+
+Any example files and directories not needed for the skill should be deleted. The initialization script creates example files in `scripts/`, `references/`, and `assets/` to demonstrate structure, but most skills won't need all of them.
+
+#### Update SKILL.md
+
+**Writing Guidelines:** Always use imperative/infinitive form.
+
+##### Frontmatter
+
+Write the YAML frontmatter with `name` and `description`:
+
+- `name`: The skill name
+- `description`: This is the primary triggering mechanism for your skill, and helps Claude understand when to use the skill.
+ - Include both what the Skill does and specific triggers/contexts for when to use it.
+ - Include all "when to use" information here - Not in the body. The body is only loaded after triggering, so "When to Use This Skill" sections in the body are not helpful to Claude.
+ - Example description for a `docx` skill: "Comprehensive document creation, editing, and analysis with support for tracked changes, comments, formatting preservation, and text extraction. Use when Claude needs to work with professional documents (.docx files) for: (1) Creating new documents, (2) Modifying or editing content, (3) Working with tracked changes, (4) Adding comments, or any other document tasks"
+
+Do not include any other fields in YAML frontmatter.
+
+##### Body
+
+Write instructions for using the skill and its bundled resources.
+
+### Step 5: Packaging a Skill
+
+Once development of the skill is complete, it must be packaged into a distributable .skill file that gets shared with the user. The packaging process automatically validates the skill first to ensure it meets all requirements:
+
+```bash
+scripts/package_skill.py
+```
+
+Optional output directory specification:
+
+```bash
+scripts/package_skill.py ./dist
+```
+
+The packaging script will:
+
+1. **Validate** the skill automatically, checking:
+
+ - YAML frontmatter format and required fields
+ - Skill naming conventions and directory structure
+ - Description completeness and quality
+ - File organization and resource references
+
+2. **Package** the skill if validation passes, creating a .skill file named after the skill (e.g., `my-skill.skill`) that includes all files and maintains the proper directory structure for distribution. The .skill file is a zip file with a .skill extension.
+
+If validation fails, the script will report the errors and exit without creating a package. Fix any validation errors and run the packaging command again.
+
+### Step 6: Iterate
+
+After testing the skill, users may request improvements. Often this happens right after using the skill, with fresh context of how the skill performed.
+
+**Iteration workflow:**
+
+1. Use the skill on real tasks
+2. Notice struggles or inefficiencies
+3. Identify how SKILL.md or bundled resources should be updated
+4. Implement changes and test again
diff --git a/skills/skill-creator/references/output-patterns.md b/skills/skill-creator/references/output-patterns.md
new file mode 100644
index 0000000..073ddda
--- /dev/null
+++ b/skills/skill-creator/references/output-patterns.md
@@ -0,0 +1,82 @@
+# Output Patterns
+
+Use these patterns when skills need to produce consistent, high-quality output.
+
+## Template Pattern
+
+Provide templates for output format. Match the level of strictness to your needs.
+
+**For strict requirements (like API responses or data formats):**
+
+```markdown
+## Report structure
+
+ALWAYS use this exact template structure:
+
+# [Analysis Title]
+
+## Executive summary
+[One-paragraph overview of key findings]
+
+## Key findings
+- Finding 1 with supporting data
+- Finding 2 with supporting data
+- Finding 3 with supporting data
+
+## Recommendations
+1. Specific actionable recommendation
+2. Specific actionable recommendation
+```
+
+**For flexible guidance (when adaptation is useful):**
+
+```markdown
+## Report structure
+
+Here is a sensible default format, but use your best judgment:
+
+# [Analysis Title]
+
+## Executive summary
+[Overview]
+
+## Key findings
+[Adapt sections based on what you discover]
+
+## Recommendations
+[Tailor to the specific context]
+
+Adjust sections as needed for the specific analysis type.
+```
+
+## Examples Pattern
+
+For skills where output quality depends on seeing examples, provide input/output pairs:
+
+```markdown
+## Commit message format
+
+Generate commit messages following these examples:
+
+**Example 1:**
+Input: Added user authentication with JWT tokens
+Output:
+```
+feat(auth): implement JWT-based authentication
+
+Add login endpoint and token validation middleware
+```
+
+**Example 2:**
+Input: Fixed bug where dates displayed incorrectly in reports
+Output:
+```
+fix(reports): correct date formatting in timezone conversion
+
+Use UTC timestamps consistently across report generation
+```
+
+Follow this style: type(scope): brief description, then detailed explanation.
+```
+
+Examples help Claude understand the desired style and level of detail more clearly than descriptions alone.
diff --git a/skills/skill-creator/references/workflows.md b/skills/skill-creator/references/workflows.md
new file mode 100644
index 0000000..a350c3c
--- /dev/null
+++ b/skills/skill-creator/references/workflows.md
@@ -0,0 +1,28 @@
+# Workflow Patterns
+
+## Sequential Workflows
+
+For complex tasks, break operations into clear, sequential steps. It is often helpful to give Claude an overview of the process towards the beginning of SKILL.md:
+
+```markdown
+Filling a PDF form involves these steps:
+
+1. Analyze the form (run analyze_form.py)
+2. Create field mapping (edit fields.json)
+3. Validate mapping (run validate_fields.py)
+4. Fill the form (run fill_form.py)
+5. Verify output (run verify_output.py)
+```
+
+## Conditional Workflows
+
+For tasks with branching logic, guide Claude through decision points:
+
+```markdown
+1. Determine the modification type:
+ **Creating new content?** → Follow "Creation workflow" below
+ **Editing existing content?** → Follow "Editing workflow" below
+
+2. Creation workflow: [steps]
+3. Editing workflow: [steps]
+```
\ No newline at end of file
diff --git a/skills/skill-creator/scripts/init_skill.py b/skills/skill-creator/scripts/init_skill.py
new file mode 100755
index 0000000..329ad4e
--- /dev/null
+++ b/skills/skill-creator/scripts/init_skill.py
@@ -0,0 +1,303 @@
+#!/usr/bin/env python3
+"""
+Skill Initializer - Creates a new skill from template
+
+Usage:
+ init_skill.py --path
+
+Examples:
+ init_skill.py my-new-skill --path skills/public
+ init_skill.py my-api-helper --path skills/private
+ init_skill.py custom-skill --path /custom/location
+"""
+
+import sys
+from pathlib import Path
+
+
+SKILL_TEMPLATE = """---
+name: {skill_name}
+description: [TODO: Complete and informative explanation of what the skill does and when to use it. Include WHEN to use this skill - specific scenarios, file types, or tasks that trigger it.]
+---
+
+# {skill_title}
+
+## Overview
+
+[TODO: 1-2 sentences explaining what this skill enables]
+
+## Structuring This Skill
+
+[TODO: Choose the structure that best fits this skill's purpose. Common patterns:
+
+**1. Workflow-Based** (best for sequential processes)
+- Works well when there are clear step-by-step procedures
+- Example: DOCX skill with "Workflow Decision Tree" → "Reading" → "Creating" → "Editing"
+- Structure: ## Overview → ## Workflow Decision Tree → ## Step 1 → ## Step 2...
+
+**2. Task-Based** (best for tool collections)
+- Works well when the skill offers different operations/capabilities
+- Example: PDF skill with "Quick Start" → "Merge PDFs" → "Split PDFs" → "Extract Text"
+- Structure: ## Overview → ## Quick Start → ## Task Category 1 → ## Task Category 2...
+
+**3. Reference/Guidelines** (best for standards or specifications)
+- Works well for brand guidelines, coding standards, or requirements
+- Example: Brand styling with "Brand Guidelines" → "Colors" → "Typography" → "Features"
+- Structure: ## Overview → ## Guidelines → ## Specifications → ## Usage...
+
+**4. Capabilities-Based** (best for integrated systems)
+- Works well when the skill provides multiple interrelated features
+- Example: Product Management with "Core Capabilities" → numbered capability list
+- Structure: ## Overview → ## Core Capabilities → ### 1. Feature → ### 2. Feature...
+
+Patterns can be mixed and matched as needed. Most skills combine patterns (e.g., start with task-based, add workflow for complex operations).
+
+Delete this entire "Structuring This Skill" section when done - it's just guidance.]
+
+## [TODO: Replace with the first main section based on chosen structure]
+
+[TODO: Add content here. See examples in existing skills:
+- Code samples for technical skills
+- Decision trees for complex workflows
+- Concrete examples with realistic user requests
+- References to scripts/templates/references as needed]
+
+## Resources
+
+This skill includes example resource directories that demonstrate how to organize different types of bundled resources:
+
+### scripts/
+Executable code (Python/Bash/etc.) that can be run directly to perform specific operations.
+
+**Examples from other skills:**
+- PDF skill: `fill_fillable_fields.py`, `extract_form_field_info.py` - utilities for PDF manipulation
+- DOCX skill: `document.py`, `utilities.py` - Python modules for document processing
+
+**Appropriate for:** Python scripts, shell scripts, or any executable code that performs automation, data processing, or specific operations.
+
+**Note:** Scripts may be executed without loading into context, but can still be read by Claude for patching or environment adjustments.
+
+### references/
+Documentation and reference material intended to be loaded into context to inform Claude's process and thinking.
+
+**Examples from other skills:**
+- Product management: `communication.md`, `context_building.md` - detailed workflow guides
+- BigQuery: API reference documentation and query examples
+- Finance: Schema documentation, company policies
+
+**Appropriate for:** In-depth documentation, API references, database schemas, comprehensive guides, or any detailed information that Claude should reference while working.
+
+### assets/
+Files not intended to be loaded into context, but rather used within the output Claude produces.
+
+**Examples from other skills:**
+- Brand styling: PowerPoint template files (.pptx), logo files
+- Frontend builder: HTML/React boilerplate project directories
+- Typography: Font files (.ttf, .woff2)
+
+**Appropriate for:** Templates, boilerplate code, document templates, images, icons, fonts, or any files meant to be copied or used in the final output.
+
+---
+
+**Any unneeded directories can be deleted.** Not every skill requires all three types of resources.
+"""
+
+EXAMPLE_SCRIPT = '''#!/usr/bin/env python3
+"""
+Example helper script for {skill_name}
+
+This is a placeholder script that can be executed directly.
+Replace with actual implementation or delete if not needed.
+
+Example real scripts from other skills:
+- pdf/scripts/fill_fillable_fields.py - Fills PDF form fields
+- pdf/scripts/convert_pdf_to_images.py - Converts PDF pages to images
+"""
+
+def main():
+ print("This is an example script for {skill_name}")
+ # TODO: Add actual script logic here
+ # This could be data processing, file conversion, API calls, etc.
+
+if __name__ == "__main__":
+ main()
+'''
+
+EXAMPLE_REFERENCE = """# Reference Documentation for {skill_title}
+
+This is a placeholder for detailed reference documentation.
+Replace with actual reference content or delete if not needed.
+
+Example real reference docs from other skills:
+- product-management/references/communication.md - Comprehensive guide for status updates
+- product-management/references/context_building.md - Deep-dive on gathering context
+- bigquery/references/ - API references and query examples
+
+## When Reference Docs Are Useful
+
+Reference docs are ideal for:
+- Comprehensive API documentation
+- Detailed workflow guides
+- Complex multi-step processes
+- Information too lengthy for main SKILL.md
+- Content that's only needed for specific use cases
+
+## Structure Suggestions
+
+### API Reference Example
+- Overview
+- Authentication
+- Endpoints with examples
+- Error codes
+- Rate limits
+
+### Workflow Guide Example
+- Prerequisites
+- Step-by-step instructions
+- Common patterns
+- Troubleshooting
+- Best practices
+"""
+
+EXAMPLE_ASSET = """# Example Asset File
+
+This placeholder represents where asset files would be stored.
+Replace with actual asset files (templates, images, fonts, etc.) or delete if not needed.
+
+Asset files are NOT intended to be loaded into context, but rather used within
+the output Claude produces.
+
+Example asset files from other skills:
+- Brand guidelines: logo.png, slides_template.pptx
+- Frontend builder: hello-world/ directory with HTML/React boilerplate
+- Typography: custom-font.ttf, font-family.woff2
+- Data: sample_data.csv, test_dataset.json
+
+## Common Asset Types
+
+- Templates: .pptx, .docx, boilerplate directories
+- Images: .png, .jpg, .svg, .gif
+- Fonts: .ttf, .otf, .woff, .woff2
+- Boilerplate code: Project directories, starter files
+- Icons: .ico, .svg
+- Data files: .csv, .json, .xml, .yaml
+
+Note: This is a text placeholder. Actual assets can be any file type.
+"""
+
+
+def title_case_skill_name(skill_name):
+ """Convert hyphenated skill name to Title Case for display."""
+ return ' '.join(word.capitalize() for word in skill_name.split('-'))
+
+
+def init_skill(skill_name, path):
+ """
+ Initialize a new skill directory with template SKILL.md.
+
+ Args:
+ skill_name: Name of the skill
+ path: Path where the skill directory should be created
+
+ Returns:
+ Path to created skill directory, or None if error
+ """
+ # Determine skill directory path
+ skill_dir = Path(path).resolve() / skill_name
+
+ # Check if directory already exists
+ if skill_dir.exists():
+ print(f"❌ Error: Skill directory already exists: {skill_dir}")
+ return None
+
+ # Create skill directory
+ try:
+ skill_dir.mkdir(parents=True, exist_ok=False)
+ print(f"✅ Created skill directory: {skill_dir}")
+ except Exception as e:
+ print(f"❌ Error creating directory: {e}")
+ return None
+
+ # Create SKILL.md from template
+ skill_title = title_case_skill_name(skill_name)
+ skill_content = SKILL_TEMPLATE.format(
+ skill_name=skill_name,
+ skill_title=skill_title
+ )
+
+ skill_md_path = skill_dir / 'SKILL.md'
+ try:
+ skill_md_path.write_text(skill_content)
+ print("✅ Created SKILL.md")
+ except Exception as e:
+ print(f"❌ Error creating SKILL.md: {e}")
+ return None
+
+ # Create resource directories with example files
+ try:
+ # Create scripts/ directory with example script
+ scripts_dir = skill_dir / 'scripts'
+ scripts_dir.mkdir(exist_ok=True)
+ example_script = scripts_dir / 'example.py'
+ example_script.write_text(EXAMPLE_SCRIPT.format(skill_name=skill_name))
+ example_script.chmod(0o755)
+ print("✅ Created scripts/example.py")
+
+ # Create references/ directory with example reference doc
+ references_dir = skill_dir / 'references'
+ references_dir.mkdir(exist_ok=True)
+ example_reference = references_dir / 'api_reference.md'
+ example_reference.write_text(EXAMPLE_REFERENCE.format(skill_title=skill_title))
+ print("✅ Created references/api_reference.md")
+
+ # Create assets/ directory with example asset placeholder
+ assets_dir = skill_dir / 'assets'
+ assets_dir.mkdir(exist_ok=True)
+ example_asset = assets_dir / 'example_asset.txt'
+ example_asset.write_text(EXAMPLE_ASSET)
+ print("✅ Created assets/example_asset.txt")
+ except Exception as e:
+ print(f"❌ Error creating resource directories: {e}")
+ return None
+
+ # Print next steps
+ print(f"\n✅ Skill '{skill_name}' initialized successfully at {skill_dir}")
+ print("\nNext steps:")
+ print("1. Edit SKILL.md to complete the TODO items and update the description")
+ print("2. Customize or delete the example files in scripts/, references/, and assets/")
+ print("3. Run the validator when ready to check the skill structure")
+
+ return skill_dir
+
+
+def main():
+ if len(sys.argv) < 4 or sys.argv[2] != '--path':
+ print("Usage: init_skill.py --path ")
+ print("\nSkill name requirements:")
+ print(" - Hyphen-case identifier (e.g., 'data-analyzer')")
+ print(" - Lowercase letters, digits, and hyphens only")
+ print(" - Max 40 characters")
+ print(" - Must match directory name exactly")
+ print("\nExamples:")
+ print(" init_skill.py my-new-skill --path skills/public")
+ print(" init_skill.py my-api-helper --path skills/private")
+ print(" init_skill.py custom-skill --path /custom/location")
+ sys.exit(1)
+
+ skill_name = sys.argv[1]
+ path = sys.argv[3]
+
+ print(f"🚀 Initializing skill: {skill_name}")
+ print(f" Location: {path}")
+ print()
+
+ result = init_skill(skill_name, path)
+
+ if result:
+ sys.exit(0)
+ else:
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/skills/skill-creator/scripts/package_skill.py b/skills/skill-creator/scripts/package_skill.py
new file mode 100755
index 0000000..5cd36cb
--- /dev/null
+++ b/skills/skill-creator/scripts/package_skill.py
@@ -0,0 +1,110 @@
+#!/usr/bin/env python3
+"""
+Skill Packager - Creates a distributable .skill file of a skill folder
+
+Usage:
+ python utils/package_skill.py [output-directory]
+
+Example:
+ python utils/package_skill.py skills/public/my-skill
+ python utils/package_skill.py skills/public/my-skill ./dist
+"""
+
+import sys
+import zipfile
+from pathlib import Path
+from quick_validate import validate_skill
+
+
+def package_skill(skill_path, output_dir=None):
+ """
+ Package a skill folder into a .skill file.
+
+ Args:
+ skill_path: Path to the skill folder
+ output_dir: Optional output directory for the .skill file (defaults to current directory)
+
+ Returns:
+ Path to the created .skill file, or None if error
+ """
+ skill_path = Path(skill_path).resolve()
+
+ # Validate skill folder exists
+ if not skill_path.exists():
+ print(f"❌ Error: Skill folder not found: {skill_path}")
+ return None
+
+ if not skill_path.is_dir():
+ print(f"❌ Error: Path is not a directory: {skill_path}")
+ return None
+
+ # Validate SKILL.md exists
+ skill_md = skill_path / "SKILL.md"
+ if not skill_md.exists():
+ print(f"❌ Error: SKILL.md not found in {skill_path}")
+ return None
+
+ # Run validation before packaging
+ print("🔍 Validating skill...")
+ valid, message = validate_skill(skill_path)
+ if not valid:
+ print(f"❌ Validation failed: {message}")
+ print(" Please fix the validation errors before packaging.")
+ return None
+ print(f"✅ {message}\n")
+
+ # Determine output location
+ skill_name = skill_path.name
+ if output_dir:
+ output_path = Path(output_dir).resolve()
+ output_path.mkdir(parents=True, exist_ok=True)
+ else:
+ output_path = Path.cwd()
+
+ skill_filename = output_path / f"{skill_name}.skill"
+
+ # Create the .skill file (zip format)
+ try:
+ with zipfile.ZipFile(skill_filename, 'w', zipfile.ZIP_DEFLATED) as zipf:
+ # Walk through the skill directory
+ for file_path in skill_path.rglob('*'):
+ if file_path.is_file():
+ # Calculate the relative path within the zip
+ arcname = file_path.relative_to(skill_path.parent)
+ zipf.write(file_path, arcname)
+ print(f" Added: {arcname}")
+
+ print(f"\n✅ Successfully packaged skill to: {skill_filename}")
+ return skill_filename
+
+ except Exception as e:
+ print(f"❌ Error creating .skill file: {e}")
+ return None
+
+
+def main():
+ if len(sys.argv) < 2:
+ print("Usage: python utils/package_skill.py [output-directory]")
+ print("\nExample:")
+ print(" python utils/package_skill.py skills/public/my-skill")
+ print(" python utils/package_skill.py skills/public/my-skill ./dist")
+ sys.exit(1)
+
+ skill_path = sys.argv[1]
+ output_dir = sys.argv[2] if len(sys.argv) > 2 else None
+
+ print(f"📦 Packaging skill: {skill_path}")
+ if output_dir:
+ print(f" Output directory: {output_dir}")
+ print()
+
+ result = package_skill(skill_path, output_dir)
+
+ if result:
+ sys.exit(0)
+ else:
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/skills/skill-creator/scripts/quick_validate.py b/skills/skill-creator/scripts/quick_validate.py
new file mode 100755
index 0000000..d9fbeb7
--- /dev/null
+++ b/skills/skill-creator/scripts/quick_validate.py
@@ -0,0 +1,95 @@
+#!/usr/bin/env python3
+"""
+Quick validation script for skills - minimal version
+"""
+
+import sys
+import os
+import re
+import yaml
+from pathlib import Path
+
+def validate_skill(skill_path):
+ """Basic validation of a skill"""
+ skill_path = Path(skill_path)
+
+ # Check SKILL.md exists
+ skill_md = skill_path / 'SKILL.md'
+ if not skill_md.exists():
+ return False, "SKILL.md not found"
+
+ # Read and validate frontmatter
+ content = skill_md.read_text()
+ if not content.startswith('---'):
+ return False, "No YAML frontmatter found"
+
+ # Extract frontmatter
+ match = re.match(r'^---\n(.*?)\n---', content, re.DOTALL)
+ if not match:
+ return False, "Invalid frontmatter format"
+
+ frontmatter_text = match.group(1)
+
+ # Parse YAML frontmatter
+ try:
+ frontmatter = yaml.safe_load(frontmatter_text)
+ if not isinstance(frontmatter, dict):
+ return False, "Frontmatter must be a YAML dictionary"
+ except yaml.YAMLError as e:
+ return False, f"Invalid YAML in frontmatter: {e}"
+
+ # Define allowed properties
+ ALLOWED_PROPERTIES = {'name', 'description', 'license', 'allowed-tools', 'metadata'}
+
+ # Check for unexpected properties (excluding nested keys under metadata)
+ unexpected_keys = set(frontmatter.keys()) - ALLOWED_PROPERTIES
+ if unexpected_keys:
+ return False, (
+ f"Unexpected key(s) in SKILL.md frontmatter: {', '.join(sorted(unexpected_keys))}. "
+ f"Allowed properties are: {', '.join(sorted(ALLOWED_PROPERTIES))}"
+ )
+
+ # Check required fields
+ if 'name' not in frontmatter:
+ return False, "Missing 'name' in frontmatter"
+ if 'description' not in frontmatter:
+ return False, "Missing 'description' in frontmatter"
+
+ # Extract name for validation
+ name = frontmatter.get('name', '')
+ if not isinstance(name, str):
+ return False, f"Name must be a string, got {type(name).__name__}"
+ name = name.strip()
+ if name:
+ # Check naming convention (hyphen-case: lowercase with hyphens)
+ if not re.match(r'^[a-z0-9-]+$', name):
+ return False, f"Name '{name}' should be hyphen-case (lowercase letters, digits, and hyphens only)"
+ if name.startswith('-') or name.endswith('-') or '--' in name:
+ return False, f"Name '{name}' cannot start/end with hyphen or contain consecutive hyphens"
+ # Check name length (max 64 characters per spec)
+ if len(name) > 64:
+ return False, f"Name is too long ({len(name)} characters). Maximum is 64 characters."
+
+ # Extract and validate description
+ description = frontmatter.get('description', '')
+ if not isinstance(description, str):
+ return False, f"Description must be a string, got {type(description).__name__}"
+ description = description.strip()
+ if description:
+ # Check for angle brackets
+ if '<' in description or '>' in description:
+ return False, "Description cannot contain angle brackets (< or >)"
+ # Check description length (max 1024 characters per spec)
+ if len(description) > 1024:
+ return False, f"Description is too long ({len(description)} characters). Maximum is 1024 characters."
+
+ return True, "Skill is valid!"
+
+if __name__ == "__main__":
+ if len(sys.argv) != 2:
+ print("Usage: python quick_validate.py ")
+ sys.exit(1)
+
+ valid, message = validate_skill(sys.argv[1])
+ print(message)
+ sys.exit(0 if valid else 1)
\ No newline at end of file
diff --git a/skills/tensorboard/SKILL.md b/skills/tensorboard/SKILL.md
new file mode 100644
index 0000000..2764d52
--- /dev/null
+++ b/skills/tensorboard/SKILL.md
@@ -0,0 +1,629 @@
+---
+name: tensorboard
+description: Visualize training metrics, debug models with histograms, compare experiments, visualize model graphs, and profile performance with TensorBoard - Google's ML visualization toolkit
+version: 1.0.0
+author: Orchestra Research
+license: MIT
+tags: [MLOps, TensorBoard, Visualization, Training Metrics, Model Debugging, PyTorch, TensorFlow, Experiment Tracking, Performance Profiling]
+dependencies: [tensorboard, torch, tensorflow]
+---
+
+# TensorBoard: Visualization Toolkit for ML
+
+## When to Use This Skill
+
+Use TensorBoard when you need to:
+- **Visualize training metrics** like loss and accuracy over time
+- **Debug models** with histograms and distributions
+- **Compare experiments** across multiple runs
+- **Visualize model graphs** and architecture
+- **Project embeddings** to lower dimensions (t-SNE, PCA)
+- **Track hyperparameter** experiments
+- **Profile performance** and identify bottlenecks
+- **Visualize images and text** during training
+
+**Users**: 20M+ downloads/year | **GitHub Stars**: 27k+ | **License**: Apache 2.0
+
+## Installation
+
+```bash
+# Install TensorBoard
+pip install tensorboard
+
+# PyTorch integration
+pip install torch torchvision tensorboard
+
+# TensorFlow integration (TensorBoard included)
+pip install tensorflow
+
+# Launch TensorBoard
+tensorboard --logdir=runs
+# Access at http://localhost:6006
+```
+
+## Quick Start
+
+### PyTorch
+
+```python
+from torch.utils.tensorboard import SummaryWriter
+
+# Create writer
+writer = SummaryWriter('runs/experiment_1')
+
+# Training loop
+for epoch in range(10):
+ train_loss = train_epoch()
+ val_acc = validate()
+
+ # Log metrics
+ writer.add_scalar('Loss/train', train_loss, epoch)
+ writer.add_scalar('Accuracy/val', val_acc, epoch)
+
+# Close writer
+writer.close()
+
+# Launch: tensorboard --logdir=runs
+```
+
+### TensorFlow/Keras
+
+```python
+import tensorflow as tf
+
+# Create callback
+tensorboard_callback = tf.keras.callbacks.TensorBoard(
+ log_dir='logs/fit',
+ histogram_freq=1
+)
+
+# Train model
+model.fit(
+ x_train, y_train,
+ epochs=10,
+ validation_data=(x_val, y_val),
+ callbacks=[tensorboard_callback]
+)
+
+# Launch: tensorboard --logdir=logs
+```
+
+## Core Concepts
+
+### 1. SummaryWriter (PyTorch)
+
+```python
+from torch.utils.tensorboard import SummaryWriter
+
+# Default directory: runs/CURRENT_DATETIME
+writer = SummaryWriter()
+
+# Custom directory
+writer = SummaryWriter('runs/experiment_1')
+
+# Custom comment (appended to default directory)
+writer = SummaryWriter(comment='baseline')
+
+# Log data
+writer.add_scalar('Loss/train', 0.5, step=0)
+writer.add_scalar('Loss/train', 0.3, step=1)
+
+# Flush and close
+writer.flush()
+writer.close()
+```
+
+### 2. Logging Scalars
+
+```python
+# PyTorch
+from torch.utils.tensorboard import SummaryWriter
+writer = SummaryWriter()
+
+for epoch in range(100):
+ train_loss = train()
+ val_loss = validate()
+
+ # Log individual metrics
+ writer.add_scalar('Loss/train', train_loss, epoch)
+ writer.add_scalar('Loss/val', val_loss, epoch)
+ writer.add_scalar('Accuracy/train', train_acc, epoch)
+ writer.add_scalar('Accuracy/val', val_acc, epoch)
+
+ # Learning rate
+ lr = optimizer.param_groups[0]['lr']
+ writer.add_scalar('Learning_rate', lr, epoch)
+
+writer.close()
+```
+
+```python
+# TensorFlow
+import tensorflow as tf
+
+train_summary_writer = tf.summary.create_file_writer('logs/train')
+val_summary_writer = tf.summary.create_file_writer('logs/val')
+
+for epoch in range(100):
+ with train_summary_writer.as_default():
+ tf.summary.scalar('loss', train_loss, step=epoch)
+ tf.summary.scalar('accuracy', train_acc, step=epoch)
+
+ with val_summary_writer.as_default():
+ tf.summary.scalar('loss', val_loss, step=epoch)
+ tf.summary.scalar('accuracy', val_acc, step=epoch)
+```
+
+### 3. Logging Multiple Scalars
+
+```python
+# PyTorch: Group related metrics
+writer.add_scalars('Loss', {
+ 'train': train_loss,
+ 'validation': val_loss,
+ 'test': test_loss
+}, epoch)
+
+writer.add_scalars('Metrics', {
+ 'accuracy': accuracy,
+ 'precision': precision,
+ 'recall': recall,
+ 'f1': f1_score
+}, epoch)
+```
+
+### 4. Logging Images
+
+```python
+# PyTorch
+import torch
+from torchvision.utils import make_grid
+
+# Single image
+writer.add_image('Input/sample', img_tensor, epoch)
+
+# Multiple images as grid
+img_grid = make_grid(images[:64], nrow=8)
+writer.add_image('Batch/inputs', img_grid, epoch)
+
+# Predictions visualization
+pred_grid = make_grid(predictions[:16], nrow=4)
+writer.add_image('Predictions', pred_grid, epoch)
+```
+
+```python
+# TensorFlow
+import tensorflow as tf
+
+with file_writer.as_default():
+ # Encode images as PNG
+ tf.summary.image('Training samples', images, step=epoch, max_outputs=25)
+```
+
+### 5. Logging Histograms
+
+```python
+# PyTorch: Track weight distributions
+for name, param in model.named_parameters():
+ writer.add_histogram(name, param, epoch)
+
+ # Track gradients
+ if param.grad is not None:
+ writer.add_histogram(f'{name}.grad', param.grad, epoch)
+
+# Track activations
+writer.add_histogram('Activations/relu1', activations, epoch)
+```
+
+```python
+# TensorFlow
+with file_writer.as_default():
+ tf.summary.histogram('weights/layer1', layer1.kernel, step=epoch)
+ tf.summary.histogram('activations/relu1', activations, step=epoch)
+```
+
+### 6. Logging Model Graph
+
+```python
+# PyTorch
+import torch
+
+model = MyModel()
+dummy_input = torch.randn(1, 3, 224, 224)
+
+writer.add_graph(model, dummy_input)
+writer.close()
+```
+
+```python
+# TensorFlow (automatic with Keras)
+tensorboard_callback = tf.keras.callbacks.TensorBoard(
+ log_dir='logs',
+ write_graph=True
+)
+
+model.fit(x, y, callbacks=[tensorboard_callback])
+```
+
+## Advanced Features
+
+### Embedding Projector
+
+Visualize high-dimensional data (embeddings, features) in 2D/3D.
+
+```python
+import torch
+from torch.utils.tensorboard import SummaryWriter
+
+# Get embeddings (e.g., word embeddings, image features)
+embeddings = model.get_embeddings(data) # Shape: (N, embedding_dim)
+
+# Metadata (labels for each point)
+metadata = ['class_1', 'class_2', 'class_1', ...]
+
+# Images (optional, for image embeddings)
+label_images = torch.stack([img1, img2, img3, ...])
+
+# Log to TensorBoard
+writer.add_embedding(
+ embeddings,
+ metadata=metadata,
+ label_img=label_images,
+ global_step=epoch
+)
+```
+
+**In TensorBoard:**
+- Navigate to "Projector" tab
+- Choose PCA, t-SNE, or UMAP visualization
+- Search, filter, and explore clusters
+
+### Hyperparameter Tuning
+
+```python
+from torch.utils.tensorboard import SummaryWriter
+
+# Try different hyperparameters
+for lr in [0.001, 0.01, 0.1]:
+ for batch_size in [16, 32, 64]:
+ # Create unique run directory
+ writer = SummaryWriter(f'runs/lr{lr}_bs{batch_size}')
+
+ # Log hyperparameters
+ writer.add_hparams(
+ {'lr': lr, 'batch_size': batch_size},
+ {'hparam/accuracy': final_acc, 'hparam/loss': final_loss}
+ )
+
+ # Train and log
+ for epoch in range(10):
+ loss = train(lr, batch_size)
+ writer.add_scalar('Loss/train', loss, epoch)
+
+ writer.close()
+
+# Compare in TensorBoard's "HParams" tab
+```
+
+### Text Logging
+
+```python
+# PyTorch: Log text (e.g., model predictions, summaries)
+writer.add_text('Predictions', f'Epoch {epoch}: {predictions}', epoch)
+writer.add_text('Config', str(config), 0)
+
+# Log markdown tables
+markdown_table = """
+| Metric | Value |
+|--------|-------|
+| Accuracy | 0.95 |
+| F1 Score | 0.93 |
+"""
+writer.add_text('Results', markdown_table, epoch)
+```
+
+### PR Curves
+
+Precision-Recall curves for classification.
+
+```python
+from torch.utils.tensorboard import SummaryWriter
+
+# Get predictions and labels
+predictions = model(test_data) # Shape: (N, num_classes)
+labels = test_labels # Shape: (N,)
+
+# Log PR curve for each class
+for i in range(num_classes):
+ writer.add_pr_curve(
+ f'PR_curve/class_{i}',
+ labels == i,
+ predictions[:, i],
+ global_step=epoch
+ )
+```
+
+## Integration Examples
+
+### PyTorch Training Loop
+
+```python
+import torch
+import torch.nn as nn
+from torch.utils.tensorboard import SummaryWriter
+
+# Setup
+writer = SummaryWriter('runs/resnet_experiment')
+model = ResNet50()
+optimizer = torch.optim.Adam(model.parameters(), lr=0.001)
+criterion = nn.CrossEntropyLoss()
+
+# Log model graph
+dummy_input = torch.randn(1, 3, 224, 224)
+writer.add_graph(model, dummy_input)
+
+# Training loop
+for epoch in range(50):
+ model.train()
+ train_loss = 0.0
+ train_correct = 0
+
+ for batch_idx, (data, target) in enumerate(train_loader):
+ optimizer.zero_grad()
+ output = model(data)
+ loss = criterion(output, target)
+ loss.backward()
+ optimizer.step()
+
+ train_loss += loss.item()
+ pred = output.argmax(dim=1)
+ train_correct += pred.eq(target).sum().item()
+
+ # Log batch metrics (every 100 batches)
+ if batch_idx % 100 == 0:
+ global_step = epoch * len(train_loader) + batch_idx
+ writer.add_scalar('Loss/train_batch', loss.item(), global_step)
+
+ # Epoch metrics
+ train_loss /= len(train_loader)
+ train_acc = train_correct / len(train_loader.dataset)
+
+ # Validation
+ model.eval()
+ val_loss = 0.0
+ val_correct = 0
+
+ with torch.no_grad():
+ for data, target in val_loader:
+ output = model(data)
+ val_loss += criterion(output, target).item()
+ pred = output.argmax(dim=1)
+ val_correct += pred.eq(target).sum().item()
+
+ val_loss /= len(val_loader)
+ val_acc = val_correct / len(val_loader.dataset)
+
+ # Log epoch metrics
+ writer.add_scalars('Loss', {'train': train_loss, 'val': val_loss}, epoch)
+ writer.add_scalars('Accuracy', {'train': train_acc, 'val': val_acc}, epoch)
+
+ # Log learning rate
+ writer.add_scalar('Learning_rate', optimizer.param_groups[0]['lr'], epoch)
+
+ # Log histograms (every 5 epochs)
+ if epoch % 5 == 0:
+ for name, param in model.named_parameters():
+ writer.add_histogram(name, param, epoch)
+
+ # Log sample predictions
+ if epoch % 10 == 0:
+ sample_images = data[:8]
+ writer.add_image('Sample_inputs', make_grid(sample_images), epoch)
+
+writer.close()
+```
+
+### TensorFlow/Keras Training
+
+```python
+import tensorflow as tf
+
+# Define model
+model = tf.keras.models.Sequential([
+ tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(28, 28, 1)),
+ tf.keras.layers.MaxPooling2D(),
+ tf.keras.layers.Flatten(),
+ tf.keras.layers.Dense(128, activation='relu'),
+ tf.keras.layers.Dense(10, activation='softmax')
+])
+
+model.compile(
+ optimizer='adam',
+ loss='sparse_categorical_crossentropy',
+ metrics=['accuracy']
+)
+
+# TensorBoard callback
+tensorboard_callback = tf.keras.callbacks.TensorBoard(
+ log_dir='logs/fit',
+ histogram_freq=1, # Log histograms every epoch
+ write_graph=True, # Visualize model graph
+ write_images=True, # Visualize weights as images
+ update_freq='epoch', # Log metrics every epoch
+ profile_batch='500,520', # Profile batches 500-520
+ embeddings_freq=1 # Log embeddings every epoch
+)
+
+# Train
+model.fit(
+ x_train, y_train,
+ epochs=10,
+ validation_data=(x_val, y_val),
+ callbacks=[tensorboard_callback]
+)
+```
+
+## Comparing Experiments
+
+### Multiple Runs
+
+```bash
+# Run experiments with different configs
+python train.py --lr 0.001 --logdir runs/exp1
+python train.py --lr 0.01 --logdir runs/exp2
+python train.py --lr 0.1 --logdir runs/exp3
+
+# View all runs together
+tensorboard --logdir=runs
+```
+
+**In TensorBoard:**
+- All runs appear in the same dashboard
+- Toggle runs on/off for comparison
+- Use regex to filter run names
+- Overlay charts to compare metrics
+
+### Organizing Experiments
+
+```python
+# Hierarchical organization
+runs/
+├── baseline/
+│ ├── run_1/
+│ └── run_2/
+├── improved/
+│ ├── run_1/
+│ └── run_2/
+└── final/
+ └── run_1/
+
+# Log with hierarchy
+writer = SummaryWriter('runs/baseline/run_1')
+```
+
+## Best Practices
+
+### 1. Use Descriptive Run Names
+
+```python
+# ✅ Good: Descriptive names
+from datetime import datetime
+timestamp = datetime.now().strftime('%Y%m%d_%H%M%S')
+writer = SummaryWriter(f'runs/resnet50_lr0.001_bs32_{timestamp}')
+
+# ❌ Bad: Auto-generated names
+writer = SummaryWriter() # Creates runs/Jan01_12-34-56_hostname
+```
+
+### 2. Group Related Metrics
+
+```python
+# ✅ Good: Grouped metrics
+writer.add_scalar('Loss/train', train_loss, step)
+writer.add_scalar('Loss/val', val_loss, step)
+writer.add_scalar('Accuracy/train', train_acc, step)
+writer.add_scalar('Accuracy/val', val_acc, step)
+
+# ❌ Bad: Flat namespace
+writer.add_scalar('train_loss', train_loss, step)
+writer.add_scalar('val_loss', val_loss, step)
+```
+
+### 3. Log Regularly but Not Too Often
+
+```python
+# ✅ Good: Log epoch metrics always, batch metrics occasionally
+for epoch in range(100):
+ for batch_idx, (data, target) in enumerate(train_loader):
+ loss = train_step(data, target)
+
+ # Log every 100 batches
+ if batch_idx % 100 == 0:
+ writer.add_scalar('Loss/batch', loss, global_step)
+
+ # Always log epoch metrics
+ writer.add_scalar('Loss/epoch', epoch_loss, epoch)
+
+# ❌ Bad: Log every batch (creates huge log files)
+for batch in train_loader:
+ writer.add_scalar('Loss', loss, step) # Too frequent
+```
+
+### 4. Close Writer When Done
+
+```python
+# ✅ Good: Use context manager
+with SummaryWriter('runs/exp1') as writer:
+ for epoch in range(10):
+ writer.add_scalar('Loss', loss, epoch)
+# Automatically closes
+
+# Or manually
+writer = SummaryWriter('runs/exp1')
+# ... logging ...
+writer.close()
+```
+
+### 5. Use Separate Writers for Train/Val
+
+```python
+# ✅ Good: Separate log directories
+train_writer = SummaryWriter('runs/exp1/train')
+val_writer = SummaryWriter('runs/exp1/val')
+
+train_writer.add_scalar('loss', train_loss, epoch)
+val_writer.add_scalar('loss', val_loss, epoch)
+```
+
+## Performance Profiling
+
+### TensorFlow Profiler
+
+```python
+# Enable profiling
+tensorboard_callback = tf.keras.callbacks.TensorBoard(
+ log_dir='logs',
+ profile_batch='10,20' # Profile batches 10-20
+)
+
+model.fit(x, y, callbacks=[tensorboard_callback])
+
+# View in TensorBoard Profile tab
+# Shows: GPU utilization, kernel stats, memory usage, bottlenecks
+```
+
+### PyTorch Profiler
+
+```python
+import torch.profiler as profiler
+
+with profiler.profile(
+ activities=[
+ profiler.ProfilerActivity.CPU,
+ profiler.ProfilerActivity.CUDA
+ ],
+ on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/profiler'),
+ record_shapes=True,
+ with_stack=True
+) as prof:
+ for batch in train_loader:
+ loss = train_step(batch)
+ prof.step()
+
+# View in TensorBoard Profile tab
+```
+
+## Resources
+
+- **Documentation**: https://www.tensorflow.org/tensorboard
+- **PyTorch Integration**: https://pytorch.org/docs/stable/tensorboard.html
+- **GitHub**: https://github.com/tensorflow/tensorboard (27k+ stars)
+- **TensorBoard.dev**: https://tensorboard.dev (share experiments publicly)
+
+## See Also
+
+- `references/visualization.md` - Comprehensive visualization guide
+- `references/profiling.md` - Performance profiling patterns
+- `references/integrations.md` - Framework-specific integration examples
+
+
diff --git a/skills/tensorboard/references/integrations.md b/skills/tensorboard/references/integrations.md
new file mode 100644
index 0000000..5865350
--- /dev/null
+++ b/skills/tensorboard/references/integrations.md
@@ -0,0 +1,638 @@
+# Framework Integration Guide
+
+Complete guide to integrating TensorBoard with popular ML frameworks.
+
+## Table of Contents
+- PyTorch
+- TensorFlow/Keras
+- PyTorch Lightning
+- HuggingFace Transformers
+- Fast.ai
+- JAX
+- scikit-learn
+
+## PyTorch
+
+### Basic Integration
+
+```python
+import torch
+import torch.nn as nn
+from torch.utils.tensorboard import SummaryWriter
+
+# Create writer
+writer = SummaryWriter('runs/pytorch_experiment')
+
+# Model and optimizer
+model = ResNet50()
+optimizer = torch.optim.Adam(model.parameters(), lr=0.001)
+criterion = nn.CrossEntropyLoss()
+
+# Log model graph
+dummy_input = torch.randn(1, 3, 224, 224)
+writer.add_graph(model, dummy_input)
+
+# Training loop
+for epoch in range(100):
+ model.train()
+ train_loss = 0.0
+
+ for batch_idx, (data, target) in enumerate(train_loader):
+ optimizer.zero_grad()
+ output = model(data)
+ loss = criterion(output, target)
+ loss.backward()
+ optimizer.step()
+
+ train_loss += loss.item()
+
+ # Log batch metrics
+ if batch_idx % 100 == 0:
+ global_step = epoch * len(train_loader) + batch_idx
+ writer.add_scalar('Loss/train_batch', loss.item(), global_step)
+
+ # Epoch metrics
+ train_loss /= len(train_loader)
+ writer.add_scalar('Loss/train_epoch', train_loss, epoch)
+
+ # Log histograms
+ for name, param in model.named_parameters():
+ writer.add_histogram(name, param, epoch)
+
+writer.close()
+```
+
+### torchvision Integration
+
+```python
+from torchvision.utils import make_grid
+
+# Log image batch
+for batch_idx, (images, labels) in enumerate(train_loader):
+ if batch_idx == 0: # First batch
+ img_grid = make_grid(images[:64], nrow=8)
+ writer.add_image('Training_batch', img_grid, epoch)
+ break
+```
+
+### Distributed Training
+
+```python
+import torch.distributed as dist
+from torch.nn.parallel import DistributedDataParallel as DDP
+
+# Setup
+dist.init_process_group(backend='nccl')
+rank = dist.get_rank()
+
+# Only log from rank 0
+if rank == 0:
+ writer = SummaryWriter('runs/distributed_experiment')
+
+model = DDP(model, device_ids=[rank])
+
+for epoch in range(100):
+ train_loss = train_epoch()
+
+ # Log only from rank 0
+ if rank == 0:
+ writer.add_scalar('Loss/train', train_loss, epoch)
+```
+
+## TensorFlow/Keras
+
+### Keras Callback
+
+```python
+import tensorflow as tf
+
+# TensorBoard callback
+tensorboard_callback = tf.keras.callbacks.TensorBoard(
+ log_dir='logs/keras_experiment',
+ histogram_freq=1, # Log histograms every epoch
+ write_graph=True, # Visualize model graph
+ write_images=True, # Visualize layer weights as images
+ update_freq='epoch', # Log metrics per epoch (or 'batch', or integer)
+ profile_batch='10,20', # Profile batches 10-20
+ embeddings_freq=1 # Log embeddings every epoch
+)
+
+# Compile model
+model.compile(
+ optimizer='adam',
+ loss='sparse_categorical_crossentropy',
+ metrics=['accuracy']
+)
+
+# Train with callback
+history = model.fit(
+ x_train, y_train,
+ epochs=10,
+ validation_data=(x_val, y_val),
+ callbacks=[tensorboard_callback]
+)
+```
+
+### Custom Training Loop
+
+```python
+import tensorflow as tf
+
+# Create summary writers
+train_summary_writer = tf.summary.create_file_writer('logs/train')
+val_summary_writer = tf.summary.create_file_writer('logs/val')
+
+# Training loop
+for epoch in range(100):
+ # Training
+ for step, (x_batch, y_batch) in enumerate(train_dataset):
+ with tf.GradientTape() as tape:
+ predictions = model(x_batch, training=True)
+ loss = loss_fn(y_batch, predictions)
+
+ gradients = tape.gradient(loss, model.trainable_variables)
+ optimizer.apply_gradients(zip(gradients, model.trainable_variables))
+
+ # Log training metrics
+ with train_summary_writer.as_default():
+ tf.summary.scalar('loss', loss, step=epoch * len(train_dataset) + step)
+
+ # Validation
+ for x_batch, y_batch in val_dataset:
+ predictions = model(x_batch, training=False)
+ val_loss = loss_fn(y_batch, predictions)
+ val_acc = accuracy_fn(y_batch, predictions)
+
+ # Log validation metrics
+ with val_summary_writer.as_default():
+ tf.summary.scalar('loss', val_loss, step=epoch)
+ tf.summary.scalar('accuracy', val_acc, step=epoch)
+
+ # Log histograms
+ with train_summary_writer.as_default():
+ for layer in model.layers:
+ for weight in layer.weights:
+ tf.summary.histogram(weight.name, weight, step=epoch)
+```
+
+### tf.data Integration
+
+```python
+# Log dataset samples
+for images, labels in train_dataset.take(1):
+ with file_writer.as_default():
+ tf.summary.image('Training samples', images, step=0, max_outputs=25)
+```
+
+## PyTorch Lightning
+
+### Built-in Logger
+
+```python
+import pytorch_lightning as pl
+from pytorch_lightning.loggers import TensorBoardLogger
+
+# Create logger
+logger = TensorBoardLogger('logs', name='lightning_experiment')
+
+# Lightning module
+class LitModel(pl.LightningModule):
+ def __init__(self):
+ super().__init__()
+ self.model = ResNet50()
+
+ def training_step(self, batch, batch_idx):
+ x, y = batch
+ y_hat = self.model(x)
+ loss = F.cross_entropy(y_hat, y)
+
+ # Log metrics
+ self.log('train_loss', loss, on_step=True, on_epoch=True)
+
+ return loss
+
+ def validation_step(self, batch, batch_idx):
+ x, y = batch
+ y_hat = self.model(x)
+ loss = F.cross_entropy(y_hat, y)
+ acc = (y_hat.argmax(dim=1) == y).float().mean()
+
+ # Log metrics
+ self.log('val_loss', loss, on_epoch=True)
+ self.log('val_acc', acc, on_epoch=True)
+
+ return loss
+
+ def configure_optimizers(self):
+ return torch.optim.Adam(self.parameters(), lr=0.001)
+
+# Trainer
+trainer = pl.Trainer(
+ max_epochs=100,
+ logger=logger,
+ log_every_n_steps=50
+)
+
+# Train
+model = LitModel()
+trainer.fit(model, train_loader, val_loader)
+```
+
+### Custom Logging
+
+```python
+class LitModel(pl.LightningModule):
+ def training_step(self, batch, batch_idx):
+ x, y = batch
+ y_hat = self.model(x)
+ loss = F.cross_entropy(y_hat, y)
+
+ # Log scalar
+ self.log('train_loss', loss)
+
+ # Log images (every 100 batches)
+ if batch_idx % 100 == 0:
+ from torchvision.utils import make_grid
+ img_grid = make_grid(x[:8])
+ self.logger.experiment.add_image('train_images', img_grid, self.global_step)
+
+ # Log histogram
+ self.logger.experiment.add_histogram('predictions', y_hat, self.global_step)
+
+ return loss
+```
+
+## HuggingFace Transformers
+
+### TrainingArguments Integration
+
+```python
+from transformers import Trainer, TrainingArguments
+
+training_args = TrainingArguments(
+ output_dir='./results',
+ num_train_epochs=3,
+ per_device_train_batch_size=16,
+ per_device_eval_batch_size=64,
+ logging_dir='./logs', # TensorBoard log directory
+ logging_steps=100, # Log every 100 steps
+ evaluation_strategy='epoch',
+ save_strategy='epoch',
+ load_best_model_at_end=True,
+ report_to='tensorboard' # Enable TensorBoard
+)
+
+trainer = Trainer(
+ model=model,
+ args=training_args,
+ train_dataset=train_dataset,
+ eval_dataset=eval_dataset,
+ tokenizer=tokenizer
+)
+
+# Train (automatically logs to TensorBoard)
+trainer.train()
+```
+
+### Custom Metrics
+
+```python
+from transformers import Trainer, TrainingArguments
+import numpy as np
+
+def compute_metrics(eval_pred):
+ """Custom metrics for evaluation."""
+ predictions, labels = eval_pred
+ predictions = np.argmax(predictions, axis=1)
+
+ accuracy = (predictions == labels).mean()
+ f1 = f1_score(labels, predictions, average='weighted')
+
+ return {
+ 'accuracy': accuracy,
+ 'f1': f1
+ }
+
+trainer = Trainer(
+ model=model,
+ args=training_args,
+ train_dataset=train_dataset,
+ eval_dataset=eval_dataset,
+ compute_metrics=compute_metrics # Custom metrics logged to TensorBoard
+)
+```
+
+### Manual Logging
+
+```python
+from transformers import TrainerCallback
+from torch.utils.tensorboard import SummaryWriter
+
+class TensorBoardCallback(TrainerCallback):
+ """Custom TensorBoard logging."""
+
+ def __init__(self, log_dir='logs'):
+ self.writer = SummaryWriter(log_dir)
+
+ def on_log(self, args, state, control, logs=None, **kwargs):
+ """Called when logging."""
+ if logs:
+ for key, value in logs.items():
+ self.writer.add_scalar(key, value, state.global_step)
+
+ def on_train_end(self, args, state, control, **kwargs):
+ """Close writer."""
+ self.writer.close()
+
+# Use callback
+trainer = Trainer(
+ model=model,
+ args=training_args,
+ train_dataset=train_dataset,
+ callbacks=[TensorBoardCallback()]
+)
+```
+
+## Fast.ai
+
+### Learner Integration
+
+```python
+from fastai.vision.all import *
+from fastai.callback.tensorboard import TensorBoardCallback
+
+# Create data loaders
+dls = ImageDataLoaders.from_folder(path, train='train', valid='valid')
+
+# Create learner
+learn = cnn_learner(dls, resnet50, metrics=accuracy)
+
+# Train with TensorBoard logging
+learn.fit_one_cycle(
+ 10,
+ cbs=TensorBoardCallback('logs/fastai', trace_model=True)
+)
+
+# View logs
+# tensorboard --logdir=logs/fastai
+```
+
+### Custom Callbacks
+
+```python
+from fastai.callback.core import Callback
+from torch.utils.tensorboard import SummaryWriter
+
+class CustomTensorBoardCallback(Callback):
+ """Custom TensorBoard callback."""
+
+ def __init__(self, log_dir='logs'):
+ self.writer = SummaryWriter(log_dir)
+
+ def after_batch(self):
+ """Log after each batch."""
+ if self.train_iter % 100 == 0:
+ self.writer.add_scalar('Loss/train', self.loss, self.train_iter)
+
+ def after_epoch(self):
+ """Log after each epoch."""
+ self.writer.add_scalar('Loss/train_epoch', self.recorder.train_loss, self.epoch)
+ self.writer.add_scalar('Loss/val_epoch', self.recorder.valid_loss, self.epoch)
+
+ # Log metrics
+ for i, metric in enumerate(self.recorder.metrics):
+ metric_name = self.recorder.metric_names[i+1]
+ self.writer.add_scalar(f'Metrics/{metric_name}', metric, self.epoch)
+
+# Use callback
+learn.fit_one_cycle(10, cbs=[CustomTensorBoardCallback()])
+```
+
+## JAX
+
+### Basic Integration
+
+```python
+import jax
+import jax.numpy as jnp
+from torch.utils.tensorboard import SummaryWriter
+
+writer = SummaryWriter('logs/jax_experiment')
+
+# Training loop
+for epoch in range(100):
+ for batch in train_batches:
+ # JAX training step
+ state, loss = train_step(state, batch)
+
+ # Log to TensorBoard (convert JAX array to numpy)
+ writer.add_scalar('Loss/train', float(loss), epoch)
+
+ # Validation
+ val_loss = evaluate(state, val_batches)
+ writer.add_scalar('Loss/val', float(val_loss), epoch)
+
+writer.close()
+```
+
+### Flax Integration
+
+```python
+from flax.training import train_state
+import optax
+from torch.utils.tensorboard import SummaryWriter
+
+writer = SummaryWriter('logs/flax_experiment')
+
+# Create train state
+state = train_state.TrainState.create(
+ apply_fn=model.apply,
+ params=params,
+ tx=optax.adam(0.001)
+)
+
+# Training loop
+for epoch in range(100):
+ for batch in train_loader:
+ state, loss = train_step(state, batch)
+
+ # Log metrics
+ writer.add_scalar('Loss/train', loss.item(), epoch)
+
+ # Log parameters
+ for name, param in state.params.items():
+ writer.add_histogram(f'Params/{name}', jnp.array(param), epoch)
+
+writer.close()
+```
+
+## scikit-learn
+
+### Manual Logging
+
+```python
+from sklearn.ensemble import RandomForestClassifier
+from sklearn.model_selection import cross_val_score
+from torch.utils.tensorboard import SummaryWriter
+
+writer = SummaryWriter('logs/sklearn_experiment')
+
+# Hyperparameter search
+for n_estimators in [10, 50, 100, 200]:
+ for max_depth in [3, 5, 10, None]:
+ # Train model
+ model = RandomForestClassifier(
+ n_estimators=n_estimators,
+ max_depth=max_depth,
+ random_state=42
+ )
+
+ # Cross-validation
+ scores = cross_val_score(model, X_train, y_train, cv=5)
+
+ # Log results
+ run_name = f'n{n_estimators}_d{max_depth}'
+ writer.add_scalar(f'{run_name}/cv_mean', scores.mean(), 0)
+ writer.add_scalar(f'{run_name}/cv_std', scores.std(), 0)
+
+ # Log hyperparameters
+ writer.add_hparams(
+ {'n_estimators': n_estimators, 'max_depth': max_depth or -1},
+ {'cv_accuracy': scores.mean()}
+ )
+
+writer.close()
+```
+
+### GridSearchCV Logging
+
+```python
+from sklearn.model_selection import GridSearchCV
+from torch.utils.tensorboard import SummaryWriter
+
+writer = SummaryWriter('logs/gridsearch')
+
+# Grid search
+param_grid = {
+ 'n_estimators': [10, 50, 100],
+ 'max_depth': [3, 5, 10]
+}
+
+grid_search = GridSearchCV(
+ RandomForestClassifier(),
+ param_grid,
+ cv=5,
+ return_train_score=True
+)
+
+grid_search.fit(X_train, y_train)
+
+# Log all results
+for i, params in enumerate(grid_search.cv_results_['params']):
+ mean_train_score = grid_search.cv_results_['mean_train_score'][i]
+ mean_test_score = grid_search.cv_results_['mean_test_score'][i]
+
+ param_str = '_'.join([f'{k}{v}' for k, v in params.items()])
+
+ writer.add_scalar(f'{param_str}/train', mean_train_score, 0)
+ writer.add_scalar(f'{param_str}/test', mean_test_score, 0)
+
+# Log best params
+writer.add_text('Best_params', str(grid_search.best_params_), 0)
+writer.add_scalar('Best_score', grid_search.best_score_, 0)
+
+writer.close()
+```
+
+## Best Practices
+
+### 1. Consistent Naming Conventions
+
+```python
+# ✅ Good: Hierarchical names across frameworks
+writer.add_scalar('Loss/train', train_loss, step)
+writer.add_scalar('Loss/val', val_loss, step)
+writer.add_scalar('Metrics/accuracy', accuracy, step)
+
+# Works the same in PyTorch, TensorFlow, Lightning
+```
+
+### 2. Use Framework-Specific Features
+
+```python
+# PyTorch: Use SummaryWriter
+from torch.utils.tensorboard import SummaryWriter
+
+# TensorFlow: Use tf.summary
+import tensorflow as tf
+tf.summary.scalar('loss', loss, step=step)
+
+# Lightning: Use self.log()
+self.log('train_loss', loss)
+
+# Transformers: Use report_to='tensorboard'
+training_args = TrainingArguments(report_to='tensorboard')
+```
+
+### 3. Centralize Logging Logic
+
+```python
+class MetricLogger:
+ """Universal metric logger."""
+
+ def __init__(self, log_dir='logs'):
+ self.writer = SummaryWriter(log_dir)
+
+ def log_scalar(self, name, value, step):
+ self.writer.add_scalar(name, value, step)
+
+ def log_image(self, name, image, step):
+ self.writer.add_image(name, image, step)
+
+ def log_histogram(self, name, values, step):
+ self.writer.add_histogram(name, values, step)
+
+ def close(self):
+ self.writer.close()
+
+# Use across frameworks
+logger = MetricLogger('logs/universal')
+logger.log_scalar('Loss/train', train_loss, epoch)
+```
+
+### 4. Framework Detection
+
+```python
+def get_tensorboard_writer(framework='auto', log_dir='logs'):
+ """Get TensorBoard writer for any framework."""
+ if framework == 'auto':
+ # Auto-detect framework
+ try:
+ import torch
+ framework = 'pytorch'
+ except ImportError:
+ try:
+ import tensorflow as tf
+ framework = 'tensorflow'
+ except ImportError:
+ raise ValueError("No supported framework found")
+
+ if framework == 'pytorch':
+ from torch.utils.tensorboard import SummaryWriter
+ return SummaryWriter(log_dir)
+
+ elif framework == 'tensorflow':
+ import tensorflow as tf
+ return tf.summary.create_file_writer(log_dir)
+
+# Use it
+writer = get_tensorboard_writer(log_dir='logs/auto')
+```
+
+## Resources
+
+- **PyTorch**: https://pytorch.org/docs/stable/tensorboard.html
+- **TensorFlow**: https://www.tensorflow.org/tensorboard
+- **Lightning**: https://pytorch-lightning.readthedocs.io/en/stable/extensions/logging.html
+- **Transformers**: https://huggingface.co/docs/transformers/main_classes/trainer
+- **Fast.ai**: https://docs.fast.ai/callback.tensorboard.html
diff --git a/skills/tensorboard/references/profiling.md b/skills/tensorboard/references/profiling.md
new file mode 100644
index 0000000..5b1da6b
--- /dev/null
+++ b/skills/tensorboard/references/profiling.md
@@ -0,0 +1,545 @@
+# Performance Profiling Guide
+
+Complete guide to profiling and optimizing ML models with TensorBoard.
+
+## Table of Contents
+- PyTorch Profiler
+- TensorFlow Profiler
+- GPU Utilization
+- Memory Profiling
+- Bottleneck Detection
+- Optimization Strategies
+
+## PyTorch Profiler
+
+### Basic Profiling
+
+```python
+import torch
+import torch.profiler as profiler
+
+model = MyModel().cuda()
+optimizer = torch.optim.Adam(model.parameters())
+
+# Profile training loop
+with profiler.profile(
+ activities=[
+ profiler.ProfilerActivity.CPU,
+ profiler.ProfilerActivity.CUDA,
+ ],
+ on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/profiler'),
+ record_shapes=True,
+ with_stack=True
+) as prof:
+ for step, (data, target) in enumerate(train_loader):
+ optimizer.zero_grad()
+ output = model(data.cuda())
+ loss = F.cross_entropy(output, target.cuda())
+ loss.backward()
+ optimizer.step()
+
+ # Mark step for profiler
+ prof.step()
+
+ if step >= 10: # Profile first 10 steps
+ break
+```
+
+### Profiler Configuration
+
+```python
+with profiler.profile(
+ activities=[
+ profiler.ProfilerActivity.CPU, # Profile CPU ops
+ profiler.ProfilerActivity.CUDA, # Profile GPU ops
+ ],
+ schedule=profiler.schedule(
+ wait=1, # Warmup steps (skip profiling)
+ warmup=1, # Steps to warmup profiler
+ active=3, # Steps to actively profile
+ repeat=2 # Repeat cycle 2 times
+ ),
+ on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/profiler'),
+ record_shapes=True, # Record tensor shapes
+ profile_memory=True, # Track memory allocation
+ with_stack=True, # Record source code stack traces
+ with_flops=True # Estimate FLOPS
+) as prof:
+ for step, batch in enumerate(train_loader):
+ train_step(batch)
+ prof.step()
+```
+
+### Profile Inference
+
+```python
+model.eval()
+
+with profiler.profile(
+ activities=[profiler.ProfilerActivity.CPU, profiler.ProfilerActivity.CUDA],
+ on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/inference_profiler')
+) as prof:
+ with torch.no_grad():
+ for i in range(100):
+ data = torch.randn(1, 3, 224, 224).cuda()
+ output = model(data)
+ prof.step()
+```
+
+### Analyze Profile Data
+
+```python
+# Print profiler summary
+print(prof.key_averages().table(sort_by="cuda_time_total", row_limit=10))
+
+# Export Chrome trace (for chrome://tracing)
+prof.export_chrome_trace("trace.json")
+
+# View in TensorBoard
+# tensorboard --logdir=runs/profiler
+```
+
+**TensorBoard Profile Tab shows:**
+- Overview: GPU utilization, step time breakdown
+- Operator view: Time spent in each operation
+- Kernel view: GPU kernel execution
+- Trace view: Timeline of operations
+- Memory view: Memory allocation over time
+
+## TensorFlow Profiler
+
+### Profile with Callback
+
+```python
+import tensorflow as tf
+
+# Create profiler callback
+tensorboard_callback = tf.keras.callbacks.TensorBoard(
+ log_dir='logs/profiler',
+ profile_batch='10,20' # Profile batches 10-20
+)
+
+# Train with profiling
+model.fit(
+ x_train, y_train,
+ epochs=5,
+ callbacks=[tensorboard_callback]
+)
+
+# Launch TensorBoard
+# tensorboard --logdir=logs/profiler
+```
+
+### Programmatic Profiling
+
+```python
+import tensorflow as tf
+
+# Start profiler
+tf.profiler.experimental.start('logs/profiler')
+
+# Training code
+for epoch in range(5):
+ for step, (x, y) in enumerate(train_dataset):
+ with tf.GradientTape() as tape:
+ predictions = model(x, training=True)
+ loss = loss_fn(y, predictions)
+
+ gradients = tape.gradient(loss, model.trainable_variables)
+ optimizer.apply_gradients(zip(gradients, model.trainable_variables))
+
+ # Profile specific steps
+ if epoch == 2 and step == 10:
+ tf.profiler.experimental.start('logs/profiler_step10')
+
+ if epoch == 2 and step == 20:
+ tf.profiler.experimental.stop()
+
+# Stop profiler
+tf.profiler.experimental.stop()
+```
+
+### Profile Custom Training Loop
+
+```python
+# Profile with context manager
+with tf.profiler.experimental.Profile('logs/profiler'):
+ for epoch in range(3):
+ for step, (x, y) in enumerate(train_dataset):
+ train_step(x, y)
+```
+
+## GPU Utilization
+
+### Monitor GPU Usage
+
+```python
+import torch
+import torch.profiler as profiler
+
+with profiler.profile(
+ activities=[profiler.ProfilerActivity.CUDA],
+ on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/gpu_profile'),
+ with_stack=True
+) as prof:
+ for step, batch in enumerate(train_loader):
+ # Your training step
+ output = model(batch.cuda())
+ loss = criterion(output, target.cuda())
+ loss.backward()
+ optimizer.step()
+
+ prof.step()
+
+# View in TensorBoard > Profile > Overview
+# Shows: GPU utilization %, kernel efficiency, memory bandwidth
+```
+
+### Optimize GPU Utilization
+
+```python
+# ✅ Good: Keep GPU busy
+def train_step(batch):
+ # Overlap data transfer with computation
+ data = batch.cuda(non_blocking=True) # Async transfer
+
+ # Mixed precision for faster computation
+ with torch.cuda.amp.autocast():
+ output = model(data)
+ loss = criterion(output, target)
+
+ return loss
+
+# ❌ Bad: GPU idle during data transfer
+def train_step_slow(batch):
+ data = batch.cuda() # Blocking transfer
+ output = model(data)
+ return loss
+```
+
+### Reduce CPU-GPU Synchronization
+
+```python
+# ✅ Good: Minimize synchronization
+for epoch in range(100):
+ for batch in train_loader:
+ loss = train_step(batch)
+
+ # Accumulate losses (no sync)
+ total_loss += loss.item()
+
+ # Synchronize once per epoch
+ avg_loss = total_loss / len(train_loader)
+
+# ❌ Bad: Frequent synchronization
+for batch in train_loader:
+ loss = train_step(batch)
+ print(f"Loss: {loss.item()}") # Syncs every batch!
+```
+
+## Memory Profiling
+
+### Track Memory Allocation
+
+```python
+import torch
+import torch.profiler as profiler
+
+with profiler.profile(
+ activities=[profiler.ProfilerActivity.CUDA],
+ profile_memory=True,
+ record_shapes=True,
+ on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/memory_profile')
+) as prof:
+ for step, batch in enumerate(train_loader):
+ train_step(batch)
+ prof.step()
+
+# View in TensorBoard > Profile > Memory View
+# Shows: Memory allocation over time, peak memory, allocation stack traces
+```
+
+### Find Memory Leaks
+
+```python
+import torch
+
+# Record memory snapshots
+torch.cuda.memory._record_memory_history(
+ enabled=True,
+ max_entries=100000
+)
+
+# Training
+for batch in train_loader:
+ train_step(batch)
+
+# Save memory snapshot
+snapshot = torch.cuda.memory._snapshot()
+torch.cuda.memory._dump_snapshot("memory_snapshot.pickle")
+
+# Analyze with:
+# python -m torch.cuda.memory_viz trace_plot memory_snapshot.pickle -o memory_trace.html
+```
+
+### Optimize Memory Usage
+
+```python
+# ✅ Good: Gradient accumulation for large batches
+accumulation_steps = 4
+
+for i, batch in enumerate(train_loader):
+ # Forward
+ output = model(batch)
+ loss = criterion(output, target) / accumulation_steps
+
+ # Backward
+ loss.backward()
+
+ # Step optimizer every accumulation_steps
+ if (i + 1) % accumulation_steps == 0:
+ optimizer.step()
+ optimizer.zero_grad()
+
+# ✅ Good: Release memory explicitly
+del intermediate_tensor
+torch.cuda.empty_cache()
+
+# ✅ Good: Use gradient checkpointing
+from torch.utils.checkpoint import checkpoint
+
+def custom_forward(module, input):
+ return checkpoint(module, input)
+```
+
+## Bottleneck Detection
+
+### Identify Slow Operations
+
+```python
+with profiler.profile(
+ activities=[profiler.ProfilerActivity.CPU, profiler.ProfilerActivity.CUDA],
+ on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/bottleneck_profile'),
+ with_stack=True
+) as prof:
+ for step, batch in enumerate(train_loader):
+ train_step(batch)
+ prof.step()
+
+# Print slowest operations
+print(prof.key_averages().table(
+ sort_by="cuda_time_total",
+ row_limit=20
+))
+
+# Expected output:
+# Name | CPU time | CUDA time | Calls
+# aten::conv2d | 5.2 ms | 45.3 ms | 32
+# aten::batch_norm | 1.1 ms | 8.7 ms | 32
+# aten::relu | 0.3 ms | 2.1 ms | 32
+```
+
+### Optimize Data Loading
+
+```python
+# ✅ Good: Efficient data loading
+train_loader = torch.utils.data.DataLoader(
+ dataset,
+ batch_size=32,
+ num_workers=4, # Parallel data loading
+ pin_memory=True, # Faster GPU transfer
+ prefetch_factor=2, # Prefetch batches
+ persistent_workers=True # Reuse workers
+)
+
+# Profile data loading
+import time
+
+start = time.time()
+for batch in train_loader:
+ pass
+print(f"Data loading time: {time.time() - start:.2f}s")
+
+# ❌ Bad: Single worker, no pinning
+train_loader = torch.utils.data.DataLoader(
+ dataset,
+ batch_size=32,
+ num_workers=0 # Slow!
+)
+```
+
+### Profile Specific Operations
+
+```python
+# Context manager for specific code blocks
+with profiler.record_function("data_preprocessing"):
+ data = preprocess(batch)
+
+with profiler.record_function("forward_pass"):
+ output = model(data)
+
+with profiler.record_function("loss_computation"):
+ loss = criterion(output, target)
+
+# View in TensorBoard > Profile > Trace View
+```
+
+## Optimization Strategies
+
+### Mixed Precision Training
+
+```python
+import torch
+from torch.cuda.amp import autocast, GradScaler
+
+scaler = GradScaler()
+
+for batch in train_loader:
+ optimizer.zero_grad()
+
+ # Mixed precision forward pass
+ with autocast():
+ output = model(batch.cuda())
+ loss = criterion(output, target.cuda())
+
+ # Scaled backward pass
+ scaler.scale(loss).backward()
+ scaler.step(optimizer)
+ scaler.update()
+
+# Profile to verify speedup
+with profiler.profile(
+ activities=[profiler.ProfilerActivity.CUDA],
+ on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/mixed_precision')
+) as prof:
+ train_with_mixed_precision()
+ prof.step()
+```
+
+### Kernel Fusion
+
+```python
+# ✅ Good: Fused operations
+# torch.nn.functional.gelu() is fused
+output = F.gelu(x)
+
+# ❌ Bad: Separate operations
+# Manual GELU (slower due to multiple kernels)
+output = 0.5 * x * (1 + torch.tanh(math.sqrt(2 / math.pi) * (x + 0.044715 * x**3)))
+
+# Use torch.jit to fuse custom operations
+@torch.jit.script
+def fused_gelu(x):
+ return 0.5 * x * (1 + torch.tanh(math.sqrt(2 / math.pi) * (x + 0.044715 * x**3)))
+```
+
+### Reduce Host-Device Transfers
+
+```python
+# ✅ Good: Keep data on GPU
+data = data.cuda() # Transfer once
+for epoch in range(100):
+ output = model(data) # No transfer
+ loss = criterion(output, target)
+
+# ❌ Bad: Frequent transfers
+for epoch in range(100):
+ output = model(data.cuda()) # Transfer every epoch!
+ loss = criterion(output.cpu(), target.cpu()) # Transfer back!
+```
+
+### Batch Size Optimization
+
+```python
+# Find optimal batch size with profiling
+for batch_size in [16, 32, 64, 128, 256]:
+ train_loader = DataLoader(dataset, batch_size=batch_size)
+
+ with profiler.profile(
+ activities=[profiler.ProfilerActivity.CUDA],
+ profile_memory=True,
+ on_trace_ready=torch.profiler.tensorboard_trace_handler(f'./runs/bs{batch_size}')
+ ) as prof:
+ for step, batch in enumerate(train_loader):
+ train_step(batch)
+ prof.step()
+
+ if step >= 10:
+ break
+
+# Compare in TensorBoard:
+# - GPU utilization
+# - Memory usage
+# - Throughput (samples/sec)
+```
+
+## Best Practices
+
+### 1. Profile Representative Workloads
+
+```python
+# ✅ Good: Profile realistic training scenario
+with profiler.profile(...) as prof:
+ for epoch in range(3): # Profile multiple epochs
+ for step, batch in enumerate(train_loader):
+ train_step(batch)
+ prof.step()
+
+# ❌ Bad: Profile single step
+with profiler.profile(...) as prof:
+ train_step(single_batch)
+```
+
+### 2. Profile Periodically
+
+```python
+# Profile every N epochs
+if epoch % 10 == 0:
+ with profiler.profile(
+ activities=[profiler.ProfilerActivity.CUDA],
+ on_trace_ready=torch.profiler.tensorboard_trace_handler(f'./runs/epoch{epoch}')
+ ) as prof:
+ train_epoch()
+```
+
+### 3. Compare Before/After Optimizations
+
+```python
+# Baseline
+with profiler.profile(...) as prof:
+ baseline_train()
+ prof.step()
+
+# After optimization
+with profiler.profile(...) as prof:
+ optimized_train()
+ prof.step()
+
+# Compare in TensorBoard
+```
+
+### 4. Profile Inference
+
+```python
+# Production inference profiling
+model.eval()
+
+with profiler.profile(
+ activities=[profiler.ProfilerActivity.CUDA],
+ on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/inference')
+) as prof:
+ with torch.no_grad():
+ for i in range(1000): # Realistic load
+ data = get_production_request()
+ output = model(data)
+ prof.step()
+
+# Analyze latency percentiles in TensorBoard
+```
+
+## Resources
+
+- **PyTorch Profiler**: https://pytorch.org/tutorials/recipes/recipes/profiler_recipe.html
+- **TensorFlow Profiler**: https://www.tensorflow.org/guide/profiler
+- **NVIDIA Nsight**: https://developer.nvidia.com/nsight-systems
+- **PyTorch Bottleneck**: https://pytorch.org/docs/stable/bottleneck.html
diff --git a/skills/tensorboard/references/visualization.md b/skills/tensorboard/references/visualization.md
new file mode 100644
index 0000000..be9e7d3
--- /dev/null
+++ b/skills/tensorboard/references/visualization.md
@@ -0,0 +1,620 @@
+# Comprehensive Visualization Guide
+
+Complete guide to visualizing ML experiments with TensorBoard.
+
+## Table of Contents
+- Scalars
+- Images
+- Histograms & Distributions
+- Graphs
+- Embeddings
+- Text
+- PR Curves
+- Custom Visualizations
+
+## Scalars
+
+### Basic Scalar Logging
+
+```python
+from torch.utils.tensorboard import SummaryWriter
+
+writer = SummaryWriter('runs/scalars_demo')
+
+# Log single metric
+for step in range(100):
+ loss = compute_loss()
+ writer.add_scalar('Loss', loss, step)
+
+writer.close()
+```
+
+### Multiple Scalars
+
+```python
+# Group related metrics
+writer.add_scalars('Loss', {
+ 'train': train_loss,
+ 'validation': val_loss,
+ 'test': test_loss
+}, epoch)
+
+writer.add_scalars('Metrics/Classification', {
+ 'accuracy': accuracy,
+ 'precision': precision,
+ 'recall': recall,
+ 'f1_score': f1
+}, epoch)
+```
+
+### Time-Series Metrics
+
+```python
+# Track metrics over training
+for epoch in range(100):
+ # Training metrics
+ train_loss = 0.0
+ for batch in train_loader:
+ loss = train_batch(batch)
+ train_loss += loss
+
+ train_loss /= len(train_loader)
+
+ # Validation metrics
+ val_loss, val_acc = validate()
+
+ # Log
+ writer.add_scalar('Loss/train', train_loss, epoch)
+ writer.add_scalar('Loss/val', val_loss, epoch)
+ writer.add_scalar('Accuracy/val', val_acc, epoch)
+
+ # Log learning rate
+ current_lr = optimizer.param_groups[0]['lr']
+ writer.add_scalar('Learning_rate', current_lr, epoch)
+```
+
+### Custom Smoothing
+
+TensorBoard UI allows smoothing scalars:
+- Slider from 0 (no smoothing) to 1 (maximum smoothing)
+- Exponential moving average
+- Useful for noisy metrics
+
+## Images
+
+### Single Image
+
+```python
+import torch
+from torch.utils.tensorboard import SummaryWriter
+
+writer = SummaryWriter('runs/images_demo')
+
+# Log single image (C, H, W)
+img = torch.rand(3, 224, 224)
+writer.add_image('Sample_image', img, 0)
+```
+
+### Image Grid
+
+```python
+from torchvision.utils import make_grid
+
+# Create grid from batch
+images = torch.rand(64, 3, 224, 224) # Batch of 64 images
+img_grid = make_grid(images, nrow=8) # 8 images per row
+
+writer.add_image('Image_grid', img_grid, epoch)
+```
+
+### Training Visualizations
+
+```python
+# Visualize inputs, predictions, and ground truth
+for epoch in range(10):
+ # Get batch
+ images, labels = next(iter(val_loader))
+
+ # Predict
+ with torch.no_grad():
+ predictions = model(images)
+
+ # Visualize inputs
+ input_grid = make_grid(images[:16], nrow=4)
+ writer.add_image('Inputs', input_grid, epoch)
+
+ # Visualize predictions (if images)
+ if isinstance(predictions, torch.Tensor) and predictions.dim() == 4:
+ pred_grid = make_grid(predictions[:16], nrow=4)
+ writer.add_image('Predictions', pred_grid, epoch)
+```
+
+### Attention Maps
+
+```python
+# Visualize attention weights
+attention_maps = model.get_attention(images) # (B, H, W)
+
+# Normalize to [0, 1]
+attention_maps = (attention_maps - attention_maps.min()) / (attention_maps.max() - attention_maps.min())
+
+# Add channel dimension
+attention_maps = attention_maps.unsqueeze(1) # (B, 1, H, W)
+
+# Create grid
+attention_grid = make_grid(attention_maps[:16], nrow=4)
+writer.add_image('Attention_maps', attention_grid, epoch)
+```
+
+### TensorFlow Images
+
+```python
+import tensorflow as tf
+
+file_writer = tf.summary.create_file_writer('logs/images')
+
+with file_writer.as_default():
+ # Log image batch
+ tf.summary.image('Training_samples', images, step=epoch, max_outputs=25)
+
+ # Log single image
+ tf.summary.image('Sample', img[tf.newaxis, ...], step=epoch)
+```
+
+## Histograms & Distributions
+
+### Weight Histograms
+
+```python
+# PyTorch: Track weight distributions over time
+for epoch in range(100):
+ train_epoch()
+
+ # Log all model parameters
+ for name, param in model.named_parameters():
+ writer.add_histogram(f'Weights/{name}', param, epoch)
+
+ # Log gradients
+ for name, param in model.named_parameters():
+ if param.grad is not None:
+ writer.add_histogram(f'Gradients/{name}', param.grad, epoch)
+```
+
+### Activation Histograms
+
+```python
+# Hook to capture activations
+activations = {}
+
+def get_activation(name):
+ def hook(model, input, output):
+ activations[name] = output.detach()
+ return hook
+
+# Register hooks
+model.conv1.register_forward_hook(get_activation('conv1'))
+model.conv2.register_forward_hook(get_activation('conv2'))
+model.fc.register_forward_hook(get_activation('fc'))
+
+# Forward pass
+output = model(input)
+
+# Log activations
+for name, activation in activations.items():
+ writer.add_histogram(f'Activations/{name}', activation, epoch)
+```
+
+### Custom Distributions
+
+```python
+# Log prediction distributions
+predictions = model(test_data)
+writer.add_histogram('Predictions', predictions, epoch)
+
+# Log loss distributions across batches
+losses = []
+for batch in val_loader:
+ loss = compute_loss(batch)
+ losses.append(loss)
+
+losses = torch.tensor(losses)
+writer.add_histogram('Loss_distribution', losses, epoch)
+```
+
+### TensorFlow Histograms
+
+```python
+import tensorflow as tf
+
+file_writer = tf.summary.create_file_writer('logs/histograms')
+
+with file_writer.as_default():
+ # Log weight distributions
+ for layer in model.layers:
+ for weight in layer.weights:
+ tf.summary.histogram(weight.name, weight, step=epoch)
+```
+
+## Graphs
+
+### Model Architecture
+
+```python
+import torch
+from torch.utils.tensorboard import SummaryWriter
+
+# PyTorch model
+model = ResNet50(num_classes=1000)
+
+# Create dummy input (same shape as real input)
+dummy_input = torch.randn(1, 3, 224, 224)
+
+# Log graph
+writer = SummaryWriter('runs/graph_demo')
+writer.add_graph(model, dummy_input)
+writer.close()
+
+# View in TensorBoard "Graphs" tab
+```
+
+### TensorFlow Graph
+
+```python
+# TensorFlow automatically logs graph with Keras
+tensorboard_callback = tf.keras.callbacks.TensorBoard(
+ log_dir='logs',
+ write_graph=True # Enable graph logging
+)
+
+model.fit(x, y, callbacks=[tensorboard_callback])
+```
+
+## Embeddings
+
+### Projecting Embeddings
+
+```python
+import torch
+from torch.utils.tensorboard import SummaryWriter
+
+writer = SummaryWriter('runs/embeddings_demo')
+
+# Get embeddings (e.g., word embeddings, image features)
+# Shape: (num_samples, embedding_dim)
+embeddings = model.get_embeddings(data)
+
+# Metadata (labels for each embedding)
+metadata = ['cat', 'dog', 'bird', 'cat', 'dog', ...]
+
+# Optional: Images for each embedding
+label_img = torch.stack([img1, img2, img3, ...]) # (num_samples, C, H, W)
+
+# Log embeddings
+writer.add_embedding(
+ embeddings,
+ metadata=metadata,
+ label_img=label_img,
+ global_step=epoch,
+ tag='Word_embeddings'
+)
+
+writer.close()
+```
+
+**In TensorBoard Projector:**
+- Choose PCA, t-SNE, or UMAP
+- Color by metadata labels
+- Search and filter points
+- Explore nearest neighbors
+
+### Image Embeddings
+
+```python
+# Extract features from CNN
+features = []
+labels = []
+images = []
+
+model.eval()
+with torch.no_grad():
+ for data, target in test_loader:
+ # Get features from penultimate layer
+ feature = model.get_features(data) # (B, feature_dim)
+ features.append(feature)
+ labels.extend(target.cpu().numpy())
+ images.append(data)
+
+# Concatenate
+features = torch.cat(features)
+images = torch.cat(images)
+
+# Metadata (class names)
+class_names = ['airplane', 'car', 'bird', 'cat', 'deer', 'dog', 'frog', 'horse', 'ship', 'truck']
+metadata = [class_names[label] for label in labels]
+
+# Log to TensorBoard
+writer.add_embedding(
+ features,
+ metadata=metadata,
+ label_img=images,
+ tag='CIFAR10_features'
+)
+```
+
+### Text Embeddings
+
+```python
+# Word2Vec or BERT embeddings
+word_embeddings = model.word_embeddings.weight.data # (vocab_size, embedding_dim)
+vocabulary = ['the', 'cat', 'dog', 'run', 'jump', ...]
+
+writer.add_embedding(
+ word_embeddings,
+ metadata=vocabulary,
+ tag='Word2Vec_embeddings'
+)
+```
+
+## Text
+
+### Basic Text Logging
+
+```python
+from torch.utils.tensorboard import SummaryWriter
+
+writer = SummaryWriter('runs/text_demo')
+
+# Log plain text
+writer.add_text('Config', str(config), 0)
+writer.add_text('Hyperparameters', f'lr={lr}, batch_size={batch_size}', 0)
+
+# Log predictions
+predictions_text = f"Epoch {epoch}:\n"
+for i, pred in enumerate(predictions[:5]):
+ predictions_text += f"Sample {i}: {pred}\n"
+
+writer.add_text('Predictions', predictions_text, epoch)
+```
+
+### Markdown Tables
+
+```python
+# Log results as markdown table
+results = f"""
+| Metric | Train | Validation | Test |
+|--------|-------|------------|------|
+| Accuracy | {train_acc:.4f} | {val_acc:.4f} | {test_acc:.4f} |
+| Loss | {train_loss:.4f} | {val_loss:.4f} | {test_loss:.4f} |
+| F1 Score | {train_f1:.4f} | {val_f1:.4f} | {test_f1:.4f} |
+"""
+
+writer.add_text('Results/Summary', results, epoch)
+```
+
+### Model Summaries
+
+```python
+# Log model architecture as text
+from torchinfo import summary
+
+model_summary = str(summary(model, input_size=(1, 3, 224, 224), verbose=0))
+writer.add_text('Model/Architecture', f'```\n{model_summary}\n```', 0)
+```
+
+## PR Curves
+
+### Precision-Recall Curves
+
+```python
+from torch.utils.tensorboard import SummaryWriter
+from sklearn.metrics import precision_recall_curve
+
+writer = SummaryWriter('runs/pr_curves')
+
+# Get predictions and ground truth
+y_true = []
+y_scores = []
+
+model.eval()
+with torch.no_grad():
+ for data, target in test_loader:
+ output = model(data)
+ probs = torch.softmax(output, dim=1)
+
+ y_true.extend(target.cpu().numpy())
+ y_scores.extend(probs.cpu().numpy())
+
+y_true = np.array(y_true)
+y_scores = np.array(y_scores)
+
+# Log PR curve for each class
+num_classes = y_scores.shape[1]
+for class_idx in range(num_classes):
+ # Binary classification: class vs rest
+ labels = (y_true == class_idx).astype(int)
+ scores = y_scores[:, class_idx]
+
+ # Add PR curve
+ writer.add_pr_curve(
+ f'PR_curve/class_{class_idx}',
+ labels,
+ scores,
+ global_step=epoch
+ )
+
+writer.close()
+```
+
+### ROC Curves
+
+```python
+# TensorBoard doesn't have built-in ROC, but we can log as image
+from sklearn.metrics import roc_curve, auc
+import matplotlib.pyplot as plt
+
+fig, ax = plt.subplots()
+
+for class_idx in range(num_classes):
+ labels = (y_true == class_idx).astype(int)
+ scores = y_scores[:, class_idx]
+
+ fpr, tpr, _ = roc_curve(labels, scores)
+ roc_auc = auc(fpr, tpr)
+
+ ax.plot(fpr, tpr, label=f'Class {class_idx} (AUC = {roc_auc:.2f})')
+
+ax.plot([0, 1], [0, 1], 'k--')
+ax.set_xlabel('False Positive Rate')
+ax.set_ylabel('True Positive Rate')
+ax.set_title('ROC Curves')
+ax.legend()
+
+# Convert to tensor and log
+fig.canvas.draw()
+img = np.frombuffer(fig.canvas.tostring_rgb(), dtype=np.uint8)
+img = img.reshape(fig.canvas.get_width_height()[::-1] + (3,))
+img = torch.from_numpy(img).permute(2, 0, 1)
+
+writer.add_image('ROC_curves', img, epoch)
+plt.close(fig)
+```
+
+## Custom Visualizations
+
+### Confusion Matrix
+
+```python
+import matplotlib.pyplot as plt
+import seaborn as sns
+from sklearn.metrics import confusion_matrix
+
+# Compute confusion matrix
+cm = confusion_matrix(y_true, y_pred)
+
+# Plot
+fig, ax = plt.subplots(figsize=(10, 10))
+sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', ax=ax)
+ax.set_xlabel('Predicted')
+ax.set_ylabel('True')
+ax.set_title('Confusion Matrix')
+
+# Convert to tensor and log
+fig.canvas.draw()
+img = np.frombuffer(fig.canvas.tostring_rgb(), dtype=np.uint8)
+img = img.reshape(fig.canvas.get_width_height()[::-1] + (3,))
+img = torch.from_numpy(img).permute(2, 0, 1)
+
+writer.add_image('Confusion_matrix', img, epoch)
+plt.close(fig)
+```
+
+### Loss Landscape
+
+```python
+# Visualize loss surface around current parameters
+import numpy as np
+
+def compute_loss_landscape(model, data, target, param1, param2):
+ """Compute loss for a grid of parameter values."""
+ # Save original params
+ original_params = {name: param.clone() for name, param in model.named_parameters()}
+
+ # Grid
+ param1_range = np.linspace(-1, 1, 50)
+ param2_range = np.linspace(-1, 1, 50)
+ losses = np.zeros((50, 50))
+
+ for i, p1 in enumerate(param1_range):
+ for j, p2 in enumerate(param2_range):
+ # Perturb parameters
+ model.state_dict()[param1].add_(p1)
+ model.state_dict()[param2].add_(p2)
+
+ # Compute loss
+ with torch.no_grad():
+ output = model(data)
+ loss = F.cross_entropy(output, target)
+ losses[i, j] = loss.item()
+
+ # Restore parameters
+ model.load_state_dict(original_params)
+
+ return losses
+
+# Plot
+fig = plt.figure()
+ax = fig.add_subplot(111, projection='3d')
+X, Y = np.meshgrid(np.linspace(-1, 1, 50), np.linspace(-1, 1, 50))
+ax.plot_surface(X, Y, losses, cmap='viridis')
+ax.set_title('Loss Landscape')
+
+# Log
+fig.canvas.draw()
+img = np.frombuffer(fig.canvas.tostring_rgb(), dtype=np.uint8)
+img = img.reshape(fig.canvas.get_width_height()[::-1] + (3,))
+img = torch.from_numpy(img).permute(2, 0, 1)
+writer.add_image('Loss_landscape', img, epoch)
+plt.close(fig)
+```
+
+## Best Practices
+
+### 1. Use Hierarchical Tags
+
+```python
+# ✅ Good: Organized with hierarchy
+writer.add_scalar('Loss/train', train_loss, step)
+writer.add_scalar('Loss/val', val_loss, step)
+writer.add_scalar('Metrics/accuracy', accuracy, step)
+writer.add_scalar('Metrics/f1_score', f1, step)
+
+# ❌ Bad: Flat namespace
+writer.add_scalar('train_loss', train_loss, step)
+writer.add_scalar('val_loss', val_loss, step)
+```
+
+### 2. Log Regularly but Not Excessively
+
+```python
+# ✅ Good: Epoch-level + periodic batch-level
+for epoch in range(100):
+ for batch_idx, batch in enumerate(train_loader):
+ loss = train_step(batch)
+
+ # Log every 100 batches
+ if batch_idx % 100 == 0:
+ global_step = epoch * len(train_loader) + batch_idx
+ writer.add_scalar('Loss/train_batch', loss, global_step)
+
+ # Always log epoch metrics
+ writer.add_scalar('Loss/train_epoch', epoch_loss, epoch)
+
+# ❌ Bad: Every batch (creates huge logs)
+for batch in train_loader:
+ writer.add_scalar('Loss', loss, step)
+```
+
+### 3. Visualize Sample Predictions
+
+```python
+# Log predictions periodically
+if epoch % 5 == 0:
+ model.eval()
+ with torch.no_grad():
+ sample_images, sample_labels = next(iter(val_loader))
+ predictions = model(sample_images)
+
+ # Visualize
+ img_grid = make_grid(sample_images[:16], nrow=4)
+ writer.add_image('Samples/inputs', img_grid, epoch)
+
+ # Add predictions as text
+ pred_text = '\n'.join([f'{i}: {pred.argmax()}' for i, pred in enumerate(predictions[:16])])
+ writer.add_text('Samples/predictions', pred_text, epoch)
+```
+
+## Resources
+
+- **TensorBoard Documentation**: https://www.tensorflow.org/tensorboard
+- **PyTorch TensorBoard**: https://pytorch.org/docs/stable/tensorboard.html
+- **Projector Guide**: https://www.tensorflow.org/tensorboard/tensorboard_projector_plugin
diff --git a/skills/vllm/SKILL.md b/skills/vllm/SKILL.md
new file mode 100644
index 0000000..164a99b
--- /dev/null
+++ b/skills/vllm/SKILL.md
@@ -0,0 +1,364 @@
+---
+name: vllm
+description: Serves LLMs with high throughput using vLLM's PagedAttention and continuous batching. Use when deploying production LLM APIs, optimizing inference latency/throughput, or serving models with limited GPU memory. Supports OpenAI-compatible endpoints, quantization (GPTQ/AWQ/FP8), and tensor parallelism.
+version: 1.0.0
+author: Orchestra Research
+license: MIT
+tags: [vLLM, Inference Serving, PagedAttention, Continuous Batching, High Throughput, Production, OpenAI API, Quantization, Tensor Parallelism]
+dependencies: [vllm, torch, transformers]
+---
+
+# vLLM - High-Performance LLM Serving
+
+## Quick start
+
+vLLM achieves 24x higher throughput than standard transformers through PagedAttention (block-based KV cache) and continuous batching (mixing prefill/decode requests).
+
+**Installation**:
+```bash
+pip install vllm
+```
+
+**Basic offline inference**:
+```python
+from vllm import LLM, SamplingParams
+
+llm = LLM(model="meta-llama/Llama-3-8B-Instruct")
+sampling = SamplingParams(temperature=0.7, max_tokens=256)
+
+outputs = llm.generate(["Explain quantum computing"], sampling)
+print(outputs[0].outputs[0].text)
+```
+
+**OpenAI-compatible server**:
+```bash
+vllm serve meta-llama/Llama-3-8B-Instruct
+
+# Query with OpenAI SDK
+python -c "
+from openai import OpenAI
+client = OpenAI(base_url='http://localhost:8000/v1', api_key='EMPTY')
+print(client.chat.completions.create(
+ model='meta-llama/Llama-3-8B-Instruct',
+ messages=[{'role': 'user', 'content': 'Hello!'}]
+).choices[0].message.content)
+"
+```
+
+## Common workflows
+
+### Workflow 1: Production API deployment
+
+Copy this checklist and track progress:
+
+```
+Deployment Progress:
+- [ ] Step 1: Configure server settings
+- [ ] Step 2: Test with limited traffic
+- [ ] Step 3: Enable monitoring
+- [ ] Step 4: Deploy to production
+- [ ] Step 5: Verify performance metrics
+```
+
+**Step 1: Configure server settings**
+
+Choose configuration based on your model size:
+
+```bash
+# For 7B-13B models on single GPU
+vllm serve meta-llama/Llama-3-8B-Instruct \
+ --gpu-memory-utilization 0.9 \
+ --max-model-len 8192 \
+ --port 8000
+
+# For 30B-70B models with tensor parallelism
+vllm serve meta-llama/Llama-2-70b-hf \
+ --tensor-parallel-size 4 \
+ --gpu-memory-utilization 0.9 \
+ --quantization awq \
+ --port 8000
+
+# For production with caching and metrics
+vllm serve meta-llama/Llama-3-8B-Instruct \
+ --gpu-memory-utilization 0.9 \
+ --enable-prefix-caching \
+ --enable-metrics \
+ --metrics-port 9090 \
+ --port 8000 \
+ --host 0.0.0.0
+```
+
+**Step 2: Test with limited traffic**
+
+Run load test before production:
+
+```bash
+# Install load testing tool
+pip install locust
+
+# Create test_load.py with sample requests
+# Run: locust -f test_load.py --host http://localhost:8000
+```
+
+Verify TTFT (time to first token) < 500ms and throughput > 100 req/sec.
+
+**Step 3: Enable monitoring**
+
+vLLM exposes Prometheus metrics on port 9090:
+
+```bash
+curl http://localhost:9090/metrics | grep vllm
+```
+
+Key metrics to monitor:
+- `vllm:time_to_first_token_seconds` - Latency
+- `vllm:num_requests_running` - Active requests
+- `vllm:gpu_cache_usage_perc` - KV cache utilization
+
+**Step 4: Deploy to production**
+
+Use Docker for consistent deployment:
+
+```bash
+# Run vLLM in Docker
+docker run --gpus all -p 8000:8000 \
+ vllm/vllm-openai:latest \
+ --model meta-llama/Llama-3-8B-Instruct \
+ --gpu-memory-utilization 0.9 \
+ --enable-prefix-caching
+```
+
+**Step 5: Verify performance metrics**
+
+Check that deployment meets targets:
+- TTFT < 500ms (for short prompts)
+- Throughput > target req/sec
+- GPU utilization > 80%
+- No OOM errors in logs
+
+### Workflow 2: Offline batch inference
+
+For processing large datasets without server overhead.
+
+Copy this checklist:
+
+```
+Batch Processing:
+- [ ] Step 1: Prepare input data
+- [ ] Step 2: Configure LLM engine
+- [ ] Step 3: Run batch inference
+- [ ] Step 4: Process results
+```
+
+**Step 1: Prepare input data**
+
+```python
+# Load prompts from file
+prompts = []
+with open("prompts.txt") as f:
+ prompts = [line.strip() for line in f]
+
+print(f"Loaded {len(prompts)} prompts")
+```
+
+**Step 2: Configure LLM engine**
+
+```python
+from vllm import LLM, SamplingParams
+
+llm = LLM(
+ model="meta-llama/Llama-3-8B-Instruct",
+ tensor_parallel_size=2, # Use 2 GPUs
+ gpu_memory_utilization=0.9,
+ max_model_len=4096
+)
+
+sampling = SamplingParams(
+ temperature=0.7,
+ top_p=0.95,
+ max_tokens=512,
+ stop=["", "\n\n"]
+)
+```
+
+**Step 3: Run batch inference**
+
+vLLM automatically batches requests for efficiency:
+
+```python
+# Process all prompts in one call
+outputs = llm.generate(prompts, sampling)
+
+# vLLM handles batching internally
+# No need to manually chunk prompts
+```
+
+**Step 4: Process results**
+
+```python
+# Extract generated text
+results = []
+for output in outputs:
+ prompt = output.prompt
+ generated = output.outputs[0].text
+ results.append({
+ "prompt": prompt,
+ "generated": generated,
+ "tokens": len(output.outputs[0].token_ids)
+ })
+
+# Save to file
+import json
+with open("results.jsonl", "w") as f:
+ for result in results:
+ f.write(json.dumps(result) + "\n")
+
+print(f"Processed {len(results)} prompts")
+```
+
+### Workflow 3: Quantized model serving
+
+Fit large models in limited GPU memory.
+
+```
+Quantization Setup:
+- [ ] Step 1: Choose quantization method
+- [ ] Step 2: Find or create quantized model
+- [ ] Step 3: Launch with quantization flag
+- [ ] Step 4: Verify accuracy
+```
+
+**Step 1: Choose quantization method**
+
+- **AWQ**: Best for 70B models, minimal accuracy loss
+- **GPTQ**: Wide model support, good compression
+- **FP8**: Fastest on H100 GPUs
+
+**Step 2: Find or create quantized model**
+
+Use pre-quantized models from HuggingFace:
+
+```bash
+# Search for AWQ models
+# Example: TheBloke/Llama-2-70B-AWQ
+```
+
+**Step 3: Launch with quantization flag**
+
+```bash
+# Using pre-quantized model
+vllm serve TheBloke/Llama-2-70B-AWQ \
+ --quantization awq \
+ --tensor-parallel-size 1 \
+ --gpu-memory-utilization 0.95
+
+# Results: 70B model in ~40GB VRAM
+```
+
+**Step 4: Verify accuracy**
+
+Test outputs match expected quality:
+
+```python
+# Compare quantized vs non-quantized responses
+# Verify task-specific performance unchanged
+```
+
+## When to use vs alternatives
+
+**Use vLLM when:**
+- Deploying production LLM APIs (100+ req/sec)
+- Serving OpenAI-compatible endpoints
+- Limited GPU memory but need large models
+- Multi-user applications (chatbots, assistants)
+- Need low latency with high throughput
+
+**Use alternatives instead:**
+- **llama.cpp**: CPU/edge inference, single-user
+- **HuggingFace transformers**: Research, prototyping, one-off generation
+- **TensorRT-LLM**: NVIDIA-only, need absolute maximum performance
+- **Text-Generation-Inference**: Already in HuggingFace ecosystem
+
+## Common issues
+
+**Issue: Out of memory during model loading**
+
+Reduce memory usage:
+```bash
+vllm serve MODEL \
+ --gpu-memory-utilization 0.7 \
+ --max-model-len 4096
+```
+
+Or use quantization:
+```bash
+vllm serve MODEL --quantization awq
+```
+
+**Issue: Slow first token (TTFT > 1 second)**
+
+Enable prefix caching for repeated prompts:
+```bash
+vllm serve MODEL --enable-prefix-caching
+```
+
+For long prompts, enable chunked prefill:
+```bash
+vllm serve MODEL --enable-chunked-prefill
+```
+
+**Issue: Model not found error**
+
+Use `--trust-remote-code` for custom models:
+```bash
+vllm serve MODEL --trust-remote-code
+```
+
+**Issue: Low throughput (<50 req/sec)**
+
+Increase concurrent sequences:
+```bash
+vllm serve MODEL --max-num-seqs 512
+```
+
+Check GPU utilization with `nvidia-smi` - should be >80%.
+
+**Issue: Inference slower than expected**
+
+Verify tensor parallelism uses power of 2 GPUs:
+```bash
+vllm serve MODEL --tensor-parallel-size 4 # Not 3
+```
+
+Enable speculative decoding for faster generation:
+```bash
+vllm serve MODEL --speculative-model DRAFT_MODEL
+```
+
+## Advanced topics
+
+**Server deployment patterns**: See [references/server-deployment.md](references/server-deployment.md) for Docker, Kubernetes, and load balancing configurations.
+
+**Performance optimization**: See [references/optimization.md](references/optimization.md) for PagedAttention tuning, continuous batching details, and benchmark results.
+
+**Quantization guide**: See [references/quantization.md](references/quantization.md) for AWQ/GPTQ/FP8 setup, model preparation, and accuracy comparisons.
+
+**Troubleshooting**: See [references/troubleshooting.md](references/troubleshooting.md) for detailed error messages, debugging steps, and performance diagnostics.
+
+## Hardware requirements
+
+- **Small models (7B-13B)**: 1x A10 (24GB) or A100 (40GB)
+- **Medium models (30B-40B)**: 2x A100 (40GB) with tensor parallelism
+- **Large models (70B+)**: 4x A100 (40GB) or 2x A100 (80GB), use AWQ/GPTQ
+
+Supported platforms: NVIDIA (primary), AMD ROCm, Intel GPUs, TPUs
+
+## Resources
+
+- Official docs: https://docs.vllm.ai
+- GitHub: https://github.com/vllm-project/vllm
+- Paper: "Efficient Memory Management for Large Language Model Serving with PagedAttention" (SOSP 2023)
+- Community: https://discuss.vllm.ai
+
+
+
diff --git a/skills/vllm/references/optimization.md b/skills/vllm/references/optimization.md
new file mode 100644
index 0000000..3d0cac5
--- /dev/null
+++ b/skills/vllm/references/optimization.md
@@ -0,0 +1,226 @@
+# Performance Optimization
+
+## Contents
+- PagedAttention explained
+- Continuous batching mechanics
+- Prefix caching strategies
+- Speculative decoding setup
+- Benchmark results and comparisons
+- Performance tuning guide
+
+## PagedAttention explained
+
+**Traditional attention problem**:
+- KV cache stored in contiguous memory
+- Wastes ~50% GPU memory due to fragmentation
+- Cannot dynamically reallocate for varying sequence lengths
+
+**PagedAttention solution**:
+- Divides KV cache into fixed-size blocks (like OS virtual memory)
+- Dynamic allocation from free block queue
+- Shares blocks across sequences (for prefix caching)
+
+**Memory savings example**:
+```
+Traditional: 70B model needs 160GB KV cache → OOM on 8x A100
+PagedAttention: 70B model needs 80GB KV cache → Fits on 4x A100
+```
+
+**Configuration**:
+```bash
+# Block size (default: 16 tokens)
+vllm serve MODEL --block-size 16
+
+# Number of GPU blocks (auto-calculated)
+# Controlled by --gpu-memory-utilization
+vllm serve MODEL --gpu-memory-utilization 0.9
+```
+
+## Continuous batching mechanics
+
+**Traditional batching**:
+- Wait for all sequences in batch to finish
+- GPU idle while waiting for longest sequence
+- Low GPU utilization (~40-60%)
+
+**Continuous batching**:
+- Add new requests as slots become available
+- Mix prefill (new requests) and decode (ongoing) in same batch
+- High GPU utilization (>90%)
+
+**Throughput improvement**:
+```
+Traditional batching: 50 req/sec @ 50% GPU util
+Continuous batching: 200 req/sec @ 90% GPU util
+= 4x throughput improvement
+```
+
+**Tuning parameters**:
+```bash
+# Max concurrent sequences (higher = more batching)
+vllm serve MODEL --max-num-seqs 256
+
+# Prefill/decode schedule (auto-balanced by default)
+# No manual tuning needed
+```
+
+## Prefix caching strategies
+
+Reuse computed KV cache for common prompt prefixes.
+
+**Use cases**:
+- System prompts repeated across requests
+- Few-shot examples in every prompt
+- RAG contexts with overlapping chunks
+
+**Example savings**:
+```
+Prompt: [System: 500 tokens] + [User: 100 tokens]
+
+Without caching: Compute 600 tokens every request
+With caching: Compute 500 tokens once, then 100 tokens/request
+= 83% faster TTFT
+```
+
+**Enable prefix caching**:
+```bash
+vllm serve MODEL --enable-prefix-caching
+```
+
+**Automatic prefix detection**:
+- vLLM detects common prefixes automatically
+- No code changes required
+- Works with OpenAI-compatible API
+
+**Cache hit rate monitoring**:
+```bash
+curl http://localhost:9090/metrics | grep cache_hit
+# vllm_cache_hit_rate: 0.75 (75% hit rate)
+```
+
+## Speculative decoding setup
+
+Use smaller "draft" model to propose tokens, larger model to verify.
+
+**Speed improvement**:
+```
+Standard: Generate 1 token per forward pass
+Speculative: Generate 3-5 tokens per forward pass
+= 2-3x faster generation
+```
+
+**How it works**:
+1. Draft model proposes K tokens (fast)
+2. Target model verifies all K tokens in parallel (one pass)
+3. Accept verified tokens, restart from first rejection
+
+**Setup with separate draft model**:
+```bash
+vllm serve meta-llama/Llama-3-70B-Instruct \
+ --speculative-model TinyLlama/TinyLlama-1.1B-Chat-v1.0 \
+ --num-speculative-tokens 5
+```
+
+**Setup with n-gram draft** (no separate model):
+```bash
+vllm serve MODEL \
+ --speculative-method ngram \
+ --num-speculative-tokens 3
+```
+
+**When to use**:
+- Output length > 100 tokens
+- Draft model 5-10x smaller than target
+- Acceptable 2-3% accuracy trade-off
+
+## Benchmark results
+
+**vLLM vs HuggingFace Transformers** (Llama 3 8B, A100):
+```
+Metric | HF Transformers | vLLM | Improvement
+------------------------|-----------------|--------|------------
+Throughput (req/sec) | 12 | 280 | 23x
+TTFT (ms) | 850 | 120 | 7x
+Tokens/sec | 45 | 2,100 | 47x
+GPU Memory (GB) | 28 | 16 | 1.75x less
+```
+
+**vLLM vs TensorRT-LLM** (Llama 2 70B, 4x A100):
+```
+Metric | TensorRT-LLM | vLLM | Notes
+------------------------|--------------|--------|------------------
+Throughput (req/sec) | 320 | 285 | TRT 12% faster
+Setup complexity | High | Low | vLLM much easier
+NVIDIA-only | Yes | No | vLLM multi-platform
+Quantization support | FP8, INT8 | AWQ/GPTQ/FP8 | vLLM more options
+```
+
+## Performance tuning guide
+
+**Step 1: Measure baseline**
+
+```bash
+# Install benchmarking tool
+pip install locust
+
+# Run baseline benchmark
+vllm bench throughput \
+ --model MODEL \
+ --input-tokens 128 \
+ --output-tokens 256 \
+ --num-prompts 1000
+
+# Record: throughput, TTFT, tokens/sec
+```
+
+**Step 2: Tune memory utilization**
+
+```bash
+# Try different values: 0.7, 0.85, 0.9, 0.95
+vllm serve MODEL --gpu-memory-utilization 0.9
+```
+
+Higher = more batch capacity = higher throughput, but risk OOM.
+
+**Step 3: Tune concurrency**
+
+```bash
+# Try values: 128, 256, 512, 1024
+vllm serve MODEL --max-num-seqs 256
+```
+
+Higher = more batching opportunity, but may increase latency.
+
+**Step 4: Enable optimizations**
+
+```bash
+vllm serve MODEL \
+ --enable-prefix-caching \ # For repeated prompts
+ --enable-chunked-prefill \ # For long prompts
+ --gpu-memory-utilization 0.9 \
+ --max-num-seqs 512
+```
+
+**Step 5: Re-benchmark and compare**
+
+Target improvements:
+- Throughput: +30-100%
+- TTFT: -20-50%
+- GPU utilization: >85%
+
+**Common performance issues**:
+
+**Low throughput (<50 req/sec)**:
+- Increase `--max-num-seqs`
+- Enable `--enable-prefix-caching`
+- Check GPU utilization (should be >80%)
+
+**High TTFT (>1 second)**:
+- Enable `--enable-chunked-prefill`
+- Reduce `--max-model-len` if possible
+- Check if model is too large for GPU
+
+**OOM errors**:
+- Reduce `--gpu-memory-utilization` to 0.7
+- Reduce `--max-model-len`
+- Use quantization (`--quantization awq`)
diff --git a/skills/vllm/references/quantization.md b/skills/vllm/references/quantization.md
new file mode 100644
index 0000000..44901a2
--- /dev/null
+++ b/skills/vllm/references/quantization.md
@@ -0,0 +1,284 @@
+# Quantization Guide
+
+## Contents
+- Quantization methods comparison
+- AWQ setup and usage
+- GPTQ setup and usage
+- FP8 quantization (H100)
+- Model preparation
+- Accuracy vs compression trade-offs
+
+## Quantization methods comparison
+
+| Method | Compression | Accuracy Loss | Speed | Best For |
+|--------|-------------|---------------|-------|----------|
+| **AWQ** | 4-bit (75%) | <1% | Fast | 70B models, production |
+| **GPTQ** | 4-bit (75%) | 1-2% | Fast | Wide model support |
+| **FP8** | 8-bit (50%) | <0.5% | Fastest | H100 GPUs only |
+| **SqueezeLLM** | 3-4 bit (75-80%) | 2-3% | Medium | Extreme compression |
+
+**Recommendation**:
+- **Production**: Use AWQ for 70B models
+- **H100 GPUs**: Use FP8 for best speed
+- **Maximum compatibility**: Use GPTQ
+- **Extreme compression**: Use SqueezeLLM
+
+## AWQ setup and usage
+
+**AWQ** (Activation-aware Weight Quantization) achieves best accuracy at 4-bit.
+
+**Step 1: Find pre-quantized model**
+
+Search HuggingFace for AWQ models:
+```bash
+# Example: TheBloke/Llama-2-70B-AWQ
+# Example: TheBloke/Mixtral-8x7B-Instruct-v0.1-AWQ
+```
+
+**Step 2: Launch with AWQ**
+
+```bash
+vllm serve TheBloke/Llama-2-70B-AWQ \
+ --quantization awq \
+ --tensor-parallel-size 1 \
+ --gpu-memory-utilization 0.95
+```
+
+**Memory savings**:
+```
+Llama 2 70B fp16: 140GB VRAM (4x A100 needed)
+Llama 2 70B AWQ: 35GB VRAM (1x A100 40GB)
+= 4x memory reduction
+```
+
+**Step 3: Verify performance**
+
+Test that outputs are acceptable:
+```python
+from openai import OpenAI
+
+client = OpenAI(base_url="http://localhost:8000/v1", api_key="EMPTY")
+
+# Test complex reasoning
+response = client.chat.completions.create(
+ model="TheBloke/Llama-2-70B-AWQ",
+ messages=[{"role": "user", "content": "Explain quantum entanglement"}]
+)
+
+print(response.choices[0].message.content)
+# Verify quality matches your requirements
+```
+
+**Quantize your own model** (requires GPU with 80GB+ VRAM):
+
+```python
+from awq import AutoAWQForCausalLM
+from transformers import AutoTokenizer
+
+model_path = "meta-llama/Llama-2-70b-hf"
+quant_path = "llama-2-70b-awq"
+
+# Load model
+model = AutoAWQForCausalLM.from_pretrained(model_path)
+tokenizer = AutoTokenizer.from_pretrained(model_path)
+
+# Quantize
+quant_config = {"zero_point": True, "q_group_size": 128, "w_bit": 4}
+model.quantize(tokenizer, quant_config=quant_config)
+
+# Save
+model.save_quantized(quant_path)
+tokenizer.save_pretrained(quant_path)
+```
+
+## GPTQ setup and usage
+
+**GPTQ** has widest model support and good compression.
+
+**Step 1: Find GPTQ model**
+
+```bash
+# Example: TheBloke/Llama-2-13B-GPTQ
+# Example: TheBloke/CodeLlama-34B-GPTQ
+```
+
+**Step 2: Launch with GPTQ**
+
+```bash
+vllm serve TheBloke/Llama-2-13B-GPTQ \
+ --quantization gptq \
+ --dtype float16
+```
+
+**GPTQ configuration options**:
+```bash
+# Specify GPTQ parameters if needed
+vllm serve MODEL \
+ --quantization gptq \
+ --gptq-act-order \ # Activation ordering
+ --dtype float16
+```
+
+**Quantize your own model**:
+
+```python
+from auto_gptq import AutoGPTQForCausalLM, BaseQuantizeConfig
+from transformers import AutoTokenizer
+
+model_name = "meta-llama/Llama-2-13b-hf"
+quantized_name = "llama-2-13b-gptq"
+
+# Load model
+tokenizer = AutoTokenizer.from_pretrained(model_name)
+model = AutoGPTQForCausalLM.from_pretrained(model_name, quantize_config)
+
+# Prepare calibration data
+calib_data = [...] # List of sample texts
+
+# Quantize
+quantize_config = BaseQuantizeConfig(
+ bits=4,
+ group_size=128,
+ desc_act=True
+)
+model.quantize(calib_data)
+
+# Save
+model.save_quantized(quantized_name)
+```
+
+## FP8 quantization (H100)
+
+**FP8** (8-bit floating point) offers best speed on H100 GPUs with minimal accuracy loss.
+
+**Requirements**:
+- H100 or H800 GPU
+- CUDA 12.3+ (12.8 recommended)
+- Hopper architecture support
+
+**Step 1: Enable FP8**
+
+```bash
+vllm serve meta-llama/Llama-3-70B-Instruct \
+ --quantization fp8 \
+ --tensor-parallel-size 2
+```
+
+**Performance gains on H100**:
+```
+fp16: 180 tokens/sec
+FP8: 320 tokens/sec
+= 1.8x speedup
+```
+
+**Step 2: Verify accuracy**
+
+FP8 typically has <0.5% accuracy degradation:
+```python
+# Run evaluation suite
+# Compare FP8 vs FP16 on your tasks
+# Verify acceptable accuracy
+```
+
+**Dynamic FP8 quantization** (no pre-quantized model needed):
+
+```bash
+# vLLM automatically quantizes at runtime
+vllm serve MODEL --quantization fp8
+# No model preparation required
+```
+
+## Model preparation
+
+**Pre-quantized models (easiest)**:
+
+1. Search HuggingFace: `[model name] AWQ` or `[model name] GPTQ`
+2. Download or use directly: `TheBloke/[Model]-AWQ`
+3. Launch with appropriate `--quantization` flag
+
+**Quantize your own model**:
+
+**AWQ**:
+```bash
+# Install AutoAWQ
+pip install autoawq
+
+# Run quantization script
+python quantize_awq.py --model MODEL --output OUTPUT
+```
+
+**GPTQ**:
+```bash
+# Install AutoGPTQ
+pip install auto-gptq
+
+# Run quantization script
+python quantize_gptq.py --model MODEL --output OUTPUT
+```
+
+**Calibration data**:
+- Use 128-512 diverse examples from target domain
+- Representative of production inputs
+- Higher quality calibration = better accuracy
+
+## Accuracy vs compression trade-offs
+
+**Empirical results** (Llama 2 70B on MMLU benchmark):
+
+| Quantization | Accuracy | Memory | Speed | Production-Ready |
+|--------------|----------|--------|-------|------------------|
+| FP16 (baseline) | 100% | 140GB | 1.0x | ✅ (if memory available) |
+| FP8 | 99.5% | 70GB | 1.8x | ✅ (H100 only) |
+| AWQ 4-bit | 99.0% | 35GB | 1.5x | ✅ (best for 70B) |
+| GPTQ 4-bit | 98.5% | 35GB | 1.5x | ✅ (good compatibility) |
+| SqueezeLLM 3-bit | 96.0% | 26GB | 1.3x | ⚠️ (check accuracy) |
+
+**When to use each**:
+
+**No quantization (FP16)**:
+- Have sufficient GPU memory
+- Need absolute best accuracy
+- Model <13B parameters
+
+**FP8**:
+- Using H100/H800 GPUs
+- Need best speed with minimal accuracy loss
+- Production deployment
+
+**AWQ 4-bit**:
+- Need to fit 70B model in 40GB GPU
+- Production deployment
+- <1% accuracy loss acceptable
+
+**GPTQ 4-bit**:
+- Wide model support needed
+- Not on H100 (use FP8 instead)
+- 1-2% accuracy loss acceptable
+
+**Testing strategy**:
+
+1. **Baseline**: Measure FP16 accuracy on your evaluation set
+2. **Quantize**: Create quantized version
+3. **Evaluate**: Compare quantized vs baseline on same tasks
+4. **Decide**: Accept if degradation < threshold (typically 1-2%)
+
+**Example evaluation**:
+```python
+from evaluate import load_evaluation_suite
+
+# Run on FP16 baseline
+baseline_score = evaluate(model_fp16, eval_suite)
+
+# Run on quantized
+quant_score = evaluate(model_awq, eval_suite)
+
+# Compare
+degradation = (baseline_score - quant_score) / baseline_score * 100
+print(f"Accuracy degradation: {degradation:.2f}%")
+
+# Decision
+if degradation < 1.0:
+ print("✅ Quantization acceptable for production")
+else:
+ print("⚠️ Review accuracy loss")
+```
diff --git a/skills/vllm/references/server-deployment.md b/skills/vllm/references/server-deployment.md
new file mode 100644
index 0000000..da5b837
--- /dev/null
+++ b/skills/vllm/references/server-deployment.md
@@ -0,0 +1,255 @@
+# Server Deployment Patterns
+
+## Contents
+- Docker deployment
+- Kubernetes deployment
+- Load balancing with Nginx
+- Multi-node distributed serving
+- Production configuration examples
+- Health checks and monitoring
+
+## Docker deployment
+
+**Basic Dockerfile**:
+```dockerfile
+FROM nvidia/cuda:12.1.0-devel-ubuntu22.04
+
+RUN apt-get update && apt-get install -y python3-pip
+RUN pip install vllm
+
+EXPOSE 8000
+
+CMD ["vllm", "serve", "meta-llama/Llama-3-8B-Instruct", \
+ "--host", "0.0.0.0", "--port", "8000", \
+ "--gpu-memory-utilization", "0.9"]
+```
+
+**Build and run**:
+```bash
+docker build -t vllm-server .
+docker run --gpus all -p 8000:8000 vllm-server
+```
+
+**Docker Compose** (with metrics):
+```yaml
+version: '3.8'
+services:
+ vllm:
+ image: vllm/vllm-openai:latest
+ command: >
+ --model meta-llama/Llama-3-8B-Instruct
+ --gpu-memory-utilization 0.9
+ --enable-metrics
+ --metrics-port 9090
+ ports:
+ - "8000:8000"
+ - "9090:9090"
+ deploy:
+ resources:
+ reservations:
+ devices:
+ - driver: nvidia
+ count: all
+ capabilities: [gpu]
+```
+
+## Kubernetes deployment
+
+**Deployment manifest**:
+```yaml
+apiVersion: apps/v1
+kind: Deployment
+metadata:
+ name: vllm-server
+spec:
+ replicas: 2
+ selector:
+ matchLabels:
+ app: vllm
+ template:
+ metadata:
+ labels:
+ app: vllm
+ spec:
+ containers:
+ - name: vllm
+ image: vllm/vllm-openai:latest
+ args:
+ - "--model=meta-llama/Llama-3-8B-Instruct"
+ - "--gpu-memory-utilization=0.9"
+ - "--enable-prefix-caching"
+ resources:
+ limits:
+ nvidia.com/gpu: 1
+ ports:
+ - containerPort: 8000
+ name: http
+ - containerPort: 9090
+ name: metrics
+ readinessProbe:
+ httpGet:
+ path: /health
+ port: 8000
+ initialDelaySeconds: 30
+ periodSeconds: 10
+ livenessProbe:
+ httpGet:
+ path: /health
+ port: 8000
+ initialDelaySeconds: 60
+ periodSeconds: 30
+---
+apiVersion: v1
+kind: Service
+metadata:
+ name: vllm-service
+spec:
+ selector:
+ app: vllm
+ ports:
+ - port: 8000
+ targetPort: 8000
+ name: http
+ - port: 9090
+ targetPort: 9090
+ name: metrics
+ type: LoadBalancer
+```
+
+## Load balancing with Nginx
+
+**Nginx configuration**:
+```nginx
+upstream vllm_backend {
+ least_conn; # Route to least-loaded server
+ server localhost:8001;
+ server localhost:8002;
+ server localhost:8003;
+}
+
+server {
+ listen 80;
+
+ location / {
+ proxy_pass http://vllm_backend;
+ proxy_set_header Host $host;
+ proxy_set_header X-Real-IP $remote_addr;
+
+ # Timeouts for long-running inference
+ proxy_read_timeout 300s;
+ proxy_connect_timeout 75s;
+ }
+
+ # Metrics endpoint
+ location /metrics {
+ proxy_pass http://localhost:9090/metrics;
+ }
+}
+```
+
+**Start multiple vLLM instances**:
+```bash
+# Terminal 1
+vllm serve MODEL --port 8001 --tensor-parallel-size 1
+
+# Terminal 2
+vllm serve MODEL --port 8002 --tensor-parallel-size 1
+
+# Terminal 3
+vllm serve MODEL --port 8003 --tensor-parallel-size 1
+
+# Start Nginx
+nginx -c /path/to/nginx.conf
+```
+
+## Multi-node distributed serving
+
+For models too large for single node:
+
+**Node 1** (master):
+```bash
+export MASTER_ADDR=192.168.1.10
+export MASTER_PORT=29500
+export RANK=0
+export WORLD_SIZE=2
+
+vllm serve meta-llama/Llama-2-70b-hf \
+ --tensor-parallel-size 8 \
+ --pipeline-parallel-size 2
+```
+
+**Node 2** (worker):
+```bash
+export MASTER_ADDR=192.168.1.10
+export MASTER_PORT=29500
+export RANK=1
+export WORLD_SIZE=2
+
+vllm serve meta-llama/Llama-2-70b-hf \
+ --tensor-parallel-size 8 \
+ --pipeline-parallel-size 2
+```
+
+## Production configuration examples
+
+**High throughput** (batch-heavy workload):
+```bash
+vllm serve MODEL \
+ --max-num-seqs 512 \
+ --gpu-memory-utilization 0.95 \
+ --enable-prefix-caching \
+ --trust-remote-code
+```
+
+**Low latency** (interactive workload):
+```bash
+vllm serve MODEL \
+ --max-num-seqs 64 \
+ --gpu-memory-utilization 0.85 \
+ --enable-chunked-prefill
+```
+
+**Memory-constrained** (40GB GPU for 70B model):
+```bash
+vllm serve TheBloke/Llama-2-70B-AWQ \
+ --quantization awq \
+ --tensor-parallel-size 1 \
+ --gpu-memory-utilization 0.95 \
+ --max-model-len 4096
+```
+
+## Health checks and monitoring
+
+**Health check endpoint**:
+```bash
+curl http://localhost:8000/health
+# Returns: {"status": "ok"}
+```
+
+**Readiness check** (wait for model loaded):
+```bash
+#!/bin/bash
+until curl -f http://localhost:8000/health; do
+ echo "Waiting for vLLM to be ready..."
+ sleep 5
+done
+echo "vLLM is ready!"
+```
+
+**Prometheus scraping**:
+```yaml
+# prometheus.yml
+scrape_configs:
+ - job_name: 'vllm'
+ static_configs:
+ - targets: ['localhost:9090']
+ metrics_path: '/metrics'
+ scrape_interval: 15s
+```
+
+**Grafana dashboard** (key metrics):
+- Requests per second: `rate(vllm_request_success_total[5m])`
+- TTFT p50: `histogram_quantile(0.5, vllm_time_to_first_token_seconds_bucket)`
+- TTFT p99: `histogram_quantile(0.99, vllm_time_to_first_token_seconds_bucket)`
+- GPU cache usage: `vllm_gpu_cache_usage_perc`
+- Active requests: `vllm_num_requests_running`
diff --git a/skills/vllm/references/troubleshooting.md b/skills/vllm/references/troubleshooting.md
new file mode 100644
index 0000000..c00cc9a
--- /dev/null
+++ b/skills/vllm/references/troubleshooting.md
@@ -0,0 +1,447 @@
+# Troubleshooting Guide
+
+## Contents
+- Out of memory (OOM) errors
+- Performance issues
+- Model loading errors
+- Network and connection issues
+- Quantization problems
+- Distributed serving issues
+- Debugging tools and commands
+
+## Out of memory (OOM) errors
+
+### Symptom: `torch.cuda.OutOfMemoryError` during model loading
+
+**Cause**: Model + KV cache exceeds available VRAM
+
+**Solutions (try in order)**:
+
+1. **Reduce GPU memory utilization**:
+```bash
+vllm serve MODEL --gpu-memory-utilization 0.7 # Try 0.7, 0.75, 0.8
+```
+
+2. **Reduce max sequence length**:
+```bash
+vllm serve MODEL --max-model-len 4096 # Instead of 8192
+```
+
+3. **Enable quantization**:
+```bash
+vllm serve MODEL --quantization awq # 4x memory reduction
+```
+
+4. **Use tensor parallelism** (multiple GPUs):
+```bash
+vllm serve MODEL --tensor-parallel-size 2 # Split across 2 GPUs
+```
+
+5. **Reduce max concurrent sequences**:
+```bash
+vllm serve MODEL --max-num-seqs 128 # Default is 256
+```
+
+### Symptom: OOM during inference (not model loading)
+
+**Cause**: KV cache fills up during generation
+
+**Solutions**:
+
+```bash
+# Reduce KV cache allocation
+vllm serve MODEL --gpu-memory-utilization 0.85
+
+# Reduce batch size
+vllm serve MODEL --max-num-seqs 64
+
+# Reduce max tokens per request
+# Set in client request: max_tokens=512
+```
+
+### Symptom: OOM with quantized model
+
+**Cause**: Quantization overhead or incorrect configuration
+
+**Solution**:
+```bash
+# Ensure quantization flag matches model
+vllm serve TheBloke/Llama-2-70B-AWQ --quantization awq # Must specify
+
+# Try different dtype
+vllm serve MODEL --quantization awq --dtype float16
+```
+
+## Performance issues
+
+### Symptom: Low throughput (<50 req/sec expected >100)
+
+**Diagnostic steps**:
+
+1. **Check GPU utilization**:
+```bash
+watch -n 1 nvidia-smi
+# GPU utilization should be >80%
+```
+
+If <80%, increase concurrent requests:
+```bash
+vllm serve MODEL --max-num-seqs 512 # Increase from 256
+```
+
+2. **Check if memory-bound**:
+```bash
+# If memory at 100% but GPU <80%, reduce sequence length
+vllm serve MODEL --max-model-len 4096
+```
+
+3. **Enable optimizations**:
+```bash
+vllm serve MODEL \
+ --enable-prefix-caching \
+ --enable-chunked-prefill \
+ --max-num-seqs 512
+```
+
+4. **Check tensor parallelism settings**:
+```bash
+# Must use power-of-2 GPUs
+vllm serve MODEL --tensor-parallel-size 4 # Not 3 or 5
+```
+
+### Symptom: High TTFT (time to first token >1 second)
+
+**Causes and solutions**:
+
+**Long prompts**:
+```bash
+vllm serve MODEL --enable-chunked-prefill
+```
+
+**No prefix caching**:
+```bash
+vllm serve MODEL --enable-prefix-caching # For repeated prompts
+```
+
+**Too many concurrent requests**:
+```bash
+vllm serve MODEL --max-num-seqs 64 # Reduce to prioritize latency
+```
+
+**Model too large for single GPU**:
+```bash
+vllm serve MODEL --tensor-parallel-size 2 # Parallelize prefill
+```
+
+### Symptom: Slow token generation (low tokens/sec)
+
+**Diagnostic**:
+```bash
+# Check if model is correct size
+vllm serve MODEL # Should see model size in logs
+
+# Check speculative decoding
+vllm serve MODEL --speculative-model DRAFT_MODEL
+```
+
+**For H100 GPUs**, enable FP8:
+```bash
+vllm serve MODEL --quantization fp8
+```
+
+## Model loading errors
+
+### Symptom: `OSError: MODEL not found`
+
+**Causes**:
+
+1. **Model name typo**:
+```bash
+# Check exact model name on HuggingFace
+vllm serve meta-llama/Llama-3-8B-Instruct # Correct capitalization
+```
+
+2. **Private/gated model**:
+```bash
+# Login to HuggingFace first
+huggingface-cli login
+# Then run vLLM
+vllm serve meta-llama/Llama-3-70B-Instruct
+```
+
+3. **Custom model needs trust flag**:
+```bash
+vllm serve MODEL --trust-remote-code
+```
+
+### Symptom: `ValueError: Tokenizer not found`
+
+**Solution**:
+```bash
+# Download model manually first
+python -c "from transformers import AutoTokenizer; AutoTokenizer.from_pretrained('MODEL')"
+
+# Then launch vLLM
+vllm serve MODEL
+```
+
+### Symptom: `ImportError: No module named 'flash_attn'`
+
+**Solution**:
+```bash
+# Install flash attention
+pip install flash-attn --no-build-isolation
+
+# Or disable flash attention
+vllm serve MODEL --disable-flash-attn
+```
+
+## Network and connection issues
+
+### Symptom: `Connection refused` when querying server
+
+**Diagnostic**:
+
+1. **Check server is running**:
+```bash
+curl http://localhost:8000/health
+```
+
+2. **Check port binding**:
+```bash
+# Bind to all interfaces for remote access
+vllm serve MODEL --host 0.0.0.0 --port 8000
+
+# Check if port is in use
+lsof -i :8000
+```
+
+3. **Check firewall**:
+```bash
+# Allow port through firewall
+sudo ufw allow 8000
+```
+
+### Symptom: Slow response times over network
+
+**Solutions**:
+
+1. **Increase timeout**:
+```python
+from openai import OpenAI
+
+client = OpenAI(
+ base_url="http://localhost:8000/v1",
+ api_key="EMPTY",
+ timeout=300.0 # 5 minute timeout
+)
+```
+
+2. **Check network latency**:
+```bash
+ping SERVER_IP # Should be <10ms for local network
+```
+
+3. **Use connection pooling**:
+```python
+import requests
+from requests.adapters import HTTPAdapter
+from urllib3.util.retry import Retry
+
+session = requests.Session()
+retries = Retry(total=3, backoff_factor=1)
+session.mount('http://', HTTPAdapter(max_retries=retries))
+```
+
+## Quantization problems
+
+### Symptom: `RuntimeError: Quantization format not supported`
+
+**Solution**:
+```bash
+# Ensure correct quantization method
+vllm serve MODEL --quantization awq # For AWQ models
+vllm serve MODEL --quantization gptq # For GPTQ models
+
+# Check model card for quantization type
+```
+
+### Symptom: Poor quality outputs after quantization
+
+**Diagnostic**:
+
+1. **Verify model is correctly quantized**:
+```bash
+# Check model config.json for quantization_config
+cat ~/.cache/huggingface/hub/models--MODEL/config.json
+```
+
+2. **Try different quantization method**:
+```bash
+# If AWQ quality issues, try FP8 (H100 only)
+vllm serve MODEL --quantization fp8
+
+# Or use less aggressive quantization
+vllm serve MODEL # No quantization
+```
+
+3. **Increase temperature for better diversity**:
+```python
+sampling_params = SamplingParams(temperature=0.8, top_p=0.95)
+```
+
+## Distributed serving issues
+
+### Symptom: `RuntimeError: Distributed init failed`
+
+**Diagnostic**:
+
+1. **Check environment variables**:
+```bash
+# On all nodes
+echo $MASTER_ADDR # Should be same
+echo $MASTER_PORT # Should be same
+echo $RANK # Should be unique per node (0, 1, 2, ...)
+echo $WORLD_SIZE # Should be same (total nodes)
+```
+
+2. **Check network connectivity**:
+```bash
+# From node 1 to node 2
+ping NODE2_IP
+nc -zv NODE2_IP 29500 # Check port accessibility
+```
+
+3. **Check NCCL settings**:
+```bash
+export NCCL_DEBUG=INFO
+export NCCL_SOCKET_IFNAME=eth0 # Or your network interface
+vllm serve MODEL --tensor-parallel-size 8
+```
+
+### Symptom: `NCCL error: unhandled cuda error`
+
+**Solutions**:
+
+```bash
+# Set NCCL to use correct network interface
+export NCCL_SOCKET_IFNAME=eth0 # Replace with your interface
+
+# Increase timeout
+export NCCL_TIMEOUT=1800 # 30 minutes
+
+# Force P2P for debugging
+export NCCL_P2P_DISABLE=1
+```
+
+## Debugging tools and commands
+
+### Enable debug logging
+
+```bash
+export VLLM_LOGGING_LEVEL=DEBUG
+vllm serve MODEL
+```
+
+### Monitor GPU usage
+
+```bash
+# Real-time GPU monitoring
+watch -n 1 nvidia-smi
+
+# Memory breakdown
+nvidia-smi --query-gpu=memory.used,memory.free --format=csv -l 1
+```
+
+### Profile performance
+
+```bash
+# Built-in benchmarking
+vllm bench throughput \
+ --model MODEL \
+ --input-tokens 128 \
+ --output-tokens 256 \
+ --num-prompts 100
+
+vllm bench latency \
+ --model MODEL \
+ --input-tokens 128 \
+ --output-tokens 256 \
+ --batch-size 8
+```
+
+### Check metrics
+
+```bash
+# Prometheus metrics
+curl http://localhost:9090/metrics
+
+# Filter for specific metrics
+curl http://localhost:9090/metrics | grep vllm_time_to_first_token
+
+# Key metrics to monitor:
+# - vllm_time_to_first_token_seconds
+# - vllm_time_per_output_token_seconds
+# - vllm_num_requests_running
+# - vllm_gpu_cache_usage_perc
+# - vllm_request_success_total
+```
+
+### Test server health
+
+```bash
+# Health check
+curl http://localhost:8000/health
+
+# Model info
+curl http://localhost:8000/v1/models
+
+# Test completion
+curl http://localhost:8000/v1/completions \
+ -H "Content-Type: application/json" \
+ -d '{
+ "model": "MODEL",
+ "prompt": "Hello",
+ "max_tokens": 10
+ }'
+```
+
+### Common environment variables
+
+```bash
+# CUDA settings
+export CUDA_VISIBLE_DEVICES=0,1,2,3 # Limit to specific GPUs
+
+# vLLM settings
+export VLLM_LOGGING_LEVEL=DEBUG
+export VLLM_TRACE_FUNCTION=1 # Profile functions
+export VLLM_USE_V1=1 # Use v1.0 engine (faster)
+
+# NCCL settings (distributed)
+export NCCL_DEBUG=INFO
+export NCCL_SOCKET_IFNAME=eth0
+export NCCL_IB_DISABLE=0 # Enable InfiniBand
+```
+
+### Collect diagnostic info for bug reports
+
+```bash
+# System info
+nvidia-smi
+python --version
+pip show vllm
+
+# vLLM version and config
+vllm --version
+python -c "import vllm; print(vllm.__version__)"
+
+# Run with debug logging
+export VLLM_LOGGING_LEVEL=DEBUG
+vllm serve MODEL 2>&1 | tee vllm_debug.log
+
+# Include in bug report:
+# - vllm_debug.log
+# - nvidia-smi output
+# - Full command used
+# - Expected vs actual behavior
+```
diff --git a/tests/__init__.py b/tests/__init__.py
new file mode 100644
index 0000000..e69de29
diff --git a/tests/conftest.py b/tests/conftest.py
new file mode 100644
index 0000000..a172812
--- /dev/null
+++ b/tests/conftest.py
@@ -0,0 +1,42 @@
+"""Shared fixtures for EvoScientist tests."""
+
+import os
+import tempfile
+
+import pytest
+
+
+@pytest.fixture
+def sample_tool_call():
+ """A minimal tool call dict."""
+ return {"id": "tc_001", "name": "execute", "args": {"command": "ls -la"}}
+
+
+@pytest.fixture
+def sample_tool_result():
+ """A minimal tool result dict."""
+ return {"name": "execute", "content": "[OK] file1.py file2.py", "success": True}
+
+
+@pytest.fixture
+def sample_events():
+ """A sequence of stream event dicts covering common types."""
+ return [
+ {"type": "thinking", "content": "Let me think..."},
+ {"type": "text", "content": "Here is the answer."},
+ {"type": "tool_call", "id": "tc_001", "name": "execute", "args": {"command": "ls"}},
+ {"type": "tool_result", "name": "execute", "content": "[OK] done", "success": True},
+ {"type": "subagent_start", "name": "research-agent", "description": "Find papers"},
+ {"type": "subagent_tool_call", "subagent": "research-agent", "name": "tavily_search", "args": {"query": "test"}, "id": "tc_sa_001"},
+ {"type": "subagent_tool_result", "subagent": "research-agent", "name": "tavily_search", "content": "Results...", "success": True},
+ {"type": "subagent_end", "name": "research-agent"},
+ {"type": "done", "response": "Here is the answer."},
+ ]
+
+
+@pytest.fixture
+def tmp_workspace(tmp_path):
+ """Provide a temporary workspace directory path."""
+ ws = tmp_path / "workspace"
+ ws.mkdir()
+ return str(ws)
diff --git a/tests/test_backends.py b/tests/test_backends.py
new file mode 100644
index 0000000..4a8fda0
--- /dev/null
+++ b/tests/test_backends.py
@@ -0,0 +1,123 @@
+"""Tests for EvoScientist/backends.py — validate_command, path conversion, resolve_path."""
+
+import os
+import tempfile
+
+import pytest
+
+from EvoScientist.backends import (
+ validate_command,
+ convert_virtual_paths_in_command,
+ CustomSandboxBackend,
+)
+
+
+# === validate_command ===
+
+class TestValidateCommand:
+ def test_safe_ls(self):
+ assert validate_command("ls -la") is None
+
+ def test_safe_python(self):
+ assert validate_command("python script.py") is None
+
+ def test_safe_pip(self):
+ assert validate_command("pip install pandas") is None
+
+ def test_blocked_traversal(self):
+ result = validate_command("cat ../../../etc/passwd")
+ assert result is not None
+ assert "blocked" in result.lower()
+
+ def test_blocked_sudo(self):
+ result = validate_command("sudo rm -rf /")
+ assert result is not None
+ assert "blocked" in result.lower()
+
+ def test_blocked_chmod(self):
+ result = validate_command("chmod 777 file.py")
+ assert result is not None
+
+ def test_blocked_dd(self):
+ result = validate_command("dd if=/dev/zero of=file bs=1M count=100")
+ assert result is not None
+
+ def test_blocked_home_tilde(self):
+ result = validate_command("cat ~/secrets.txt")
+ assert result is not None
+
+ def test_blocked_rm_rf_absolute(self):
+ result = validate_command("rm -rf /important")
+ assert result is not None
+
+ def test_blocked_cd_absolute(self):
+ result = validate_command("cd /etc && cat passwd")
+ assert result is not None
+
+ def test_safe_echo(self):
+ assert validate_command("echo hello world") is None
+
+ def test_safe_grep(self):
+ assert validate_command("grep -r 'pattern' .") is None
+
+
+# === convert_virtual_paths_in_command ===
+
+class TestConvertVirtualPaths:
+ def test_absolute_to_relative(self):
+ result = convert_virtual_paths_in_command("python /main.py")
+ assert result == "python ./main.py"
+
+ def test_nested_path(self):
+ result = convert_virtual_paths_in_command("cat /data/file.txt")
+ assert result == "cat ./data/file.txt"
+
+ def test_root_only(self):
+ result = convert_virtual_paths_in_command("ls /")
+ assert result == "ls ."
+
+ def test_no_change_relative(self):
+ result = convert_virtual_paths_in_command("python main.py")
+ assert result == "python main.py"
+
+ def test_url_preserved(self):
+ result = convert_virtual_paths_in_command("curl https://example.com/path")
+ # URLs should not be converted
+ assert "https://example.com/path" in result
+
+ def test_no_op_no_paths(self):
+ result = convert_virtual_paths_in_command("echo hello")
+ assert result == "echo hello"
+
+
+# === CustomSandboxBackend._resolve_path ===
+
+class TestResolvePath:
+ def test_strip_workspace_prefix(self, tmp_workspace):
+ backend = CustomSandboxBackend(root_dir=tmp_workspace, virtual_mode=True)
+ # /workspace/main.py should resolve to root/main.py
+ resolved = backend._resolve_path("/workspace/main.py")
+ assert str(resolved).endswith("main.py")
+ assert "workspace/workspace" not in str(resolved)
+
+ def test_workspace_root(self, tmp_workspace):
+ backend = CustomSandboxBackend(root_dir=tmp_workspace, virtual_mode=True)
+ resolved = backend._resolve_path("/workspace")
+ # Should resolve to root dir
+ assert resolved == backend._resolve_path("/")
+
+ def test_system_path_with_workspace_marker(self, tmp_workspace):
+ backend = CustomSandboxBackend(root_dir=tmp_workspace, virtual_mode=True)
+ resolved = backend._resolve_path("/Users/someone/project/workspace/main.py")
+ assert str(resolved).endswith("main.py")
+
+ def test_system_path_without_workspace(self, tmp_workspace):
+ backend = CustomSandboxBackend(root_dir=tmp_workspace, virtual_mode=True)
+ resolved = backend._resolve_path("/Users/someone/file.py")
+ # Falls back to basename
+ assert str(resolved).endswith("file.py")
+
+ def test_normal_virtual_path(self, tmp_workspace):
+ backend = CustomSandboxBackend(root_dir=tmp_workspace, virtual_mode=True)
+ resolved = backend._resolve_path("/src/main.py")
+ assert str(resolved).endswith("src/main.py")
diff --git a/tests/test_imports.py b/tests/test_imports.py
new file mode 100644
index 0000000..107741b
--- /dev/null
+++ b/tests/test_imports.py
@@ -0,0 +1,59 @@
+"""Smoke tests verifying package structure is intact."""
+
+import os
+import pytest
+
+needs_api_key = pytest.mark.skipif(
+ not os.getenv("ANTHROPIC_API_KEY"),
+ reason="ANTHROPIC_API_KEY not set",
+)
+
+
+def test_import_stream_utils():
+ from EvoScientist.stream.utils import (
+ is_success,
+ format_tool_compact,
+ truncate,
+ has_args,
+ count_lines,
+ truncate_with_line_hint,
+ )
+ assert callable(is_success)
+
+
+def test_import_stream_emitter():
+ from EvoScientist.stream.emitter import StreamEventEmitter, StreamEvent
+ assert callable(StreamEventEmitter.thinking)
+
+
+def test_import_stream_tracker():
+ from EvoScientist.stream.tracker import ToolCallTracker, ToolCallInfo
+ assert ToolCallTracker is not None
+
+
+def test_import_backends():
+ from EvoScientist.backends import (
+ validate_command,
+ convert_virtual_paths_in_command,
+ CustomSandboxBackend,
+ ReadOnlyFilesystemBackend,
+ )
+ assert callable(validate_command)
+
+
+def test_import_prompts():
+ from EvoScientist.prompts import get_system_prompt, RESEARCHER_INSTRUCTIONS
+ assert callable(get_system_prompt)
+
+
+def test_import_tools():
+ from EvoScientist.tools import think_tool
+ assert think_tool is not None
+ assert hasattr(think_tool, "invoke")
+
+
+@needs_api_key
+def test_import_package_exports():
+ from EvoScientist import EvoScientist_agent, create_cli_agent
+ # Just verify they are importable; don't call them without API key
+ assert EvoScientist_agent is not None
diff --git a/tests/test_prompts.py b/tests/test_prompts.py
new file mode 100644
index 0000000..2614991
--- /dev/null
+++ b/tests/test_prompts.py
@@ -0,0 +1,35 @@
+"""Tests for EvoScientist/prompts.py."""
+
+from EvoScientist.prompts import get_system_prompt, EXPERIMENT_WORKFLOW, DELEGATION_STRATEGY
+
+
+class TestGetSystemPrompt:
+ def test_returns_non_empty(self):
+ result = get_system_prompt()
+ assert isinstance(result, str)
+ assert len(result) > 100
+
+ def test_contains_workflow(self):
+ result = get_system_prompt()
+ assert "Experiment Workflow" in result
+
+ def test_contains_delegation(self):
+ result = get_system_prompt()
+ assert "Sub-Agent Delegation" in result
+
+ def test_default_params_interpolated(self):
+ result = get_system_prompt()
+ assert "3 parallel sub-agents" in result
+ assert "3 delegation rounds" in result
+
+ def test_custom_params(self):
+ result = get_system_prompt(max_concurrent=5, max_iterations=10)
+ assert "5 parallel sub-agents" in result
+ assert "10 delegation rounds" in result
+
+ def test_workflow_constant_not_empty(self):
+ assert len(EXPERIMENT_WORKFLOW) > 0
+
+ def test_delegation_has_placeholders(self):
+ assert "{max_concurrent}" in DELEGATION_STRATEGY
+ assert "{max_iterations}" in DELEGATION_STRATEGY
diff --git a/tests/test_stream_emitter.py b/tests/test_stream_emitter.py
new file mode 100644
index 0000000..da9becd
--- /dev/null
+++ b/tests/test_stream_emitter.py
@@ -0,0 +1,89 @@
+"""Tests for EvoScientist/stream/emitter.py."""
+
+from EvoScientist.stream.emitter import StreamEventEmitter, StreamEvent
+
+
+class TestStreamEventEmitter:
+ def test_thinking(self):
+ ev = StreamEventEmitter.thinking("deep thought", thinking_id=1)
+ assert isinstance(ev, StreamEvent)
+ assert ev.type == "thinking"
+ assert ev.data["content"] == "deep thought"
+ assert ev.data["id"] == 1
+
+ def test_text(self):
+ ev = StreamEventEmitter.text("hello")
+ assert ev.type == "text"
+ assert ev.data["content"] == "hello"
+
+ def test_tool_call(self):
+ ev = StreamEventEmitter.tool_call("execute", {"command": "ls"}, tool_id="tc1")
+ assert ev.type == "tool_call"
+ assert ev.data["name"] == "execute"
+ assert ev.data["args"] == {"command": "ls"}
+ assert ev.data["id"] == "tc1"
+
+ def test_tool_result(self):
+ ev = StreamEventEmitter.tool_result("execute", "[OK] done", success=True)
+ assert ev.type == "tool_result"
+ assert ev.data["name"] == "execute"
+ assert ev.data["content"] == "[OK] done"
+ assert ev.data["success"] is True
+
+ def test_tool_result_failure(self):
+ ev = StreamEventEmitter.tool_result("execute", "Error: fail", success=False)
+ assert ev.data["success"] is False
+
+ def test_subagent_start(self):
+ ev = StreamEventEmitter.subagent_start("research-agent", "Find papers")
+ assert ev.type == "subagent_start"
+ assert ev.data["name"] == "research-agent"
+ assert ev.data["description"] == "Find papers"
+
+ def test_subagent_tool_call(self):
+ ev = StreamEventEmitter.subagent_tool_call("research-agent", "tavily_search", {"query": "q"}, "tc2")
+ assert ev.type == "subagent_tool_call"
+ assert ev.data["subagent"] == "research-agent"
+ assert ev.data["name"] == "tavily_search"
+
+ def test_subagent_tool_result(self):
+ ev = StreamEventEmitter.subagent_tool_result("research-agent", "tavily_search", "results", True)
+ assert ev.type == "subagent_tool_result"
+ assert ev.data["subagent"] == "research-agent"
+
+ def test_subagent_end(self):
+ ev = StreamEventEmitter.subagent_end("research-agent")
+ assert ev.type == "subagent_end"
+ assert ev.data["name"] == "research-agent"
+
+ def test_done(self):
+ ev = StreamEventEmitter.done("final answer")
+ assert ev.type == "done"
+ assert ev.data["response"] == "final answer"
+
+ def test_done_empty(self):
+ ev = StreamEventEmitter.done()
+ assert ev.data["response"] == ""
+
+ def test_error(self):
+ ev = StreamEventEmitter.error("something broke")
+ assert ev.type == "error"
+ assert ev.data["message"] == "something broke"
+
+ def test_all_events_have_type_in_data(self):
+ """Every event's data dict should contain a 'type' key matching the event type."""
+ events = [
+ StreamEventEmitter.thinking("x"),
+ StreamEventEmitter.text("x"),
+ StreamEventEmitter.tool_call("t", {}),
+ StreamEventEmitter.tool_result("t", "x"),
+ StreamEventEmitter.subagent_start("s", "d"),
+ StreamEventEmitter.subagent_tool_call("s", "t", {}),
+ StreamEventEmitter.subagent_tool_result("s", "t", "x"),
+ StreamEventEmitter.subagent_end("s"),
+ StreamEventEmitter.done(),
+ StreamEventEmitter.error("e"),
+ ]
+ for ev in events:
+ assert "type" in ev.data
+ assert ev.data["type"] == ev.type
diff --git a/tests/test_stream_state.py b/tests/test_stream_state.py
new file mode 100644
index 0000000..6bea09b
--- /dev/null
+++ b/tests/test_stream_state.py
@@ -0,0 +1,285 @@
+"""Tests for StreamState, SubAgentState, and display helpers from cli.py."""
+
+from EvoScientist.cli import (
+ SubAgentState,
+ StreamState,
+ _parse_todo_items,
+ _build_todo_stats,
+)
+
+
+# === SubAgentState ===
+
+class TestSubAgentState:
+ def test_add_tool_call(self):
+ sa = SubAgentState("research-agent")
+ sa.add_tool_call("tavily_search", {"query": "test"}, "tc1")
+ assert len(sa.tool_calls) == 1
+ assert sa.tool_calls[0]["name"] == "tavily_search"
+
+ def test_add_tool_call_dedup_by_id(self):
+ sa = SubAgentState("research-agent")
+ sa.add_tool_call("tavily_search", {"query": "test"}, "tc1")
+ sa.add_tool_call("tavily_search", {"query": "updated"}, "tc1")
+ assert len(sa.tool_calls) == 1
+ assert sa.tool_calls[0]["args"]["query"] == "updated"
+
+ def test_add_tool_call_merge_name(self):
+ """When first call has empty name, second should fill it in."""
+ sa = SubAgentState("agent")
+ sa.add_tool_call("", {}, "tc1")
+ # Empty name + empty id → skipped entirely
+ # But with an id, it can be tracked:
+ # Actually, empty name with id is also skipped per the code (not name check)
+ # Let's use a named call first, then merge args
+ sa2 = SubAgentState("agent")
+ sa2.add_tool_call("search", {}, "tc1")
+ sa2.add_tool_call("search", {"query": "test"}, "tc1")
+ assert sa2.tool_calls[0]["args"] == {"query": "test"}
+
+ def test_skip_empty_name_no_id(self):
+ sa = SubAgentState("agent")
+ sa.add_tool_call("", {}, "")
+ assert len(sa.tool_calls) == 0
+
+ def test_add_tool_result_matched(self):
+ sa = SubAgentState("agent")
+ sa.add_tool_call("execute", {}, "tc1")
+ sa.add_tool_result("execute", "output", True)
+ result = sa.get_result_for(sa.tool_calls[0])
+ assert result is not None
+ assert result["content"] == "output"
+
+ def test_add_tool_result_fallback(self):
+ """When name doesn't match, falls back to first unmatched."""
+ sa = SubAgentState("agent")
+ sa.add_tool_call("execute", {}, "tc1")
+ sa.add_tool_result("different_name", "output", True)
+ result = sa.get_result_for(sa.tool_calls[0])
+ assert result is not None
+ assert result["content"] == "output"
+
+ def test_get_result_for_no_match(self):
+ sa = SubAgentState("agent")
+ tc = {"id": "tc_missing", "name": "x", "args": {}}
+ assert sa.get_result_for(tc) is None
+
+ def test_get_result_for_index_fallback(self):
+ """When no id, falls back to index-based matching."""
+ sa = SubAgentState("agent")
+ tc = {"id": "", "name": "execute", "args": {}}
+ sa.tool_calls.append(tc)
+ sa.tool_results.append({"name": "execute", "content": "ok", "success": True})
+ result = sa.get_result_for(tc)
+ assert result is not None
+
+
+# === StreamState ===
+
+class TestStreamState:
+ def test_handle_thinking(self):
+ state = StreamState()
+ result = state.handle_event({"type": "thinking", "content": "hmm"})
+ assert result == "thinking"
+ assert state.is_thinking is True
+ assert state.thinking_text == "hmm"
+
+ def test_handle_text(self):
+ state = StreamState()
+ result = state.handle_event({"type": "text", "content": "hello"})
+ assert result == "text"
+ assert state.is_responding is True
+ assert state.response_text == "hello"
+
+ def test_handle_text_accumulates(self):
+ state = StreamState()
+ state.handle_event({"type": "text", "content": "a"})
+ state.handle_event({"type": "text", "content": "b"})
+ assert state.response_text == "ab"
+
+ def test_handle_tool_call(self):
+ state = StreamState()
+ state.handle_event({
+ "type": "tool_call",
+ "id": "tc1",
+ "name": "execute",
+ "args": {"command": "ls"},
+ })
+ assert len(state.tool_calls) == 1
+ assert state.tool_calls[0]["name"] == "execute"
+
+ def test_handle_tool_call_update_existing(self):
+ state = StreamState()
+ state.handle_event({"type": "tool_call", "id": "tc1", "name": "execute", "args": {}})
+ state.handle_event({"type": "tool_call", "id": "tc1", "name": "execute", "args": {"command": "ls"}})
+ assert len(state.tool_calls) == 1
+ assert state.tool_calls[0]["args"] == {"command": "ls"}
+
+ def test_handle_tool_result(self):
+ state = StreamState()
+ state.handle_event({
+ "type": "tool_result",
+ "name": "execute",
+ "content": "[OK] done",
+ })
+ assert len(state.tool_results) == 1
+ assert state.is_processing is True
+
+ def test_handle_subagent_start(self):
+ state = StreamState()
+ state.handle_event({"type": "subagent_start", "name": "research-agent", "description": "Search"})
+ assert len(state.subagents) == 1
+ assert state.subagents[0].name == "research-agent"
+ assert state.subagents[0].is_active is True
+
+ def test_handle_subagent_tool_call(self):
+ state = StreamState()
+ state.handle_event({
+ "type": "subagent_tool_call",
+ "subagent": "research-agent",
+ "name": "tavily_search",
+ "args": {"query": "test"},
+ "id": "tc_sa1",
+ })
+ assert len(state.subagents) == 1
+ assert len(state.subagents[0].tool_calls) == 1
+
+ def test_handle_subagent_tool_result(self):
+ state = StreamState()
+ state.handle_event({
+ "type": "subagent_tool_call",
+ "subagent": "code-agent",
+ "name": "execute",
+ "args": {},
+ "id": "tc1",
+ })
+ state.handle_event({
+ "type": "subagent_tool_result",
+ "subagent": "code-agent",
+ "name": "execute",
+ "content": "output",
+ "success": True,
+ })
+ sa = state.subagents[0]
+ assert len(sa.tool_results) == 1
+
+ def test_handle_subagent_end(self):
+ state = StreamState()
+ state.handle_event({"type": "subagent_start", "name": "agent-x", "description": ""})
+ state.handle_event({"type": "subagent_end", "name": "agent-x"})
+ assert state.subagents[0].is_active is False
+
+ def test_handle_done(self):
+ state = StreamState()
+ state.handle_event({"type": "done", "response": "Final answer"})
+ assert state.is_processing is False
+ assert state.response_text == "Final answer"
+
+ def test_handle_done_does_not_overwrite_existing_response(self):
+ state = StreamState()
+ state.handle_event({"type": "text", "content": "Already here"})
+ state.handle_event({"type": "done", "response": "Final"})
+ assert state.response_text == "Already here"
+
+ def test_handle_error(self):
+ state = StreamState()
+ state.handle_event({"type": "error", "message": "boom"})
+ assert "[Error] boom" in state.response_text
+ assert state.is_processing is False
+
+ def test_full_event_sequence(self, sample_events):
+ state = StreamState()
+ for event in sample_events:
+ state.handle_event(event)
+ assert state.thinking_text == "Let me think..."
+ assert "Here is the answer." in state.response_text
+ assert len(state.tool_calls) == 1
+ assert len(state.tool_results) == 1
+ assert len(state.subagents) == 1
+ assert state.subagents[0].is_active is False
+
+
+# === Name merging ===
+
+class TestNameMerging:
+ def test_generic_subagent_merged(self):
+ state = StreamState()
+ # First event creates generic "sub-agent"
+ state.handle_event({
+ "type": "subagent_tool_call",
+ "subagent": "sub-agent",
+ "name": "execute",
+ "args": {},
+ "id": "tc1",
+ })
+ assert len(state.subagents) == 1
+ assert state.subagents[0].name == "sub-agent"
+
+ # Proper name arrives, should merge
+ sa = state._get_or_create_subagent("code-agent", "write code")
+ assert len(state.subagents) == 1
+ assert sa.name == "code-agent"
+ assert len(sa.tool_calls) == 1 # preserved from generic entry
+
+ def test_no_merge_when_not_generic(self):
+ state = StreamState()
+ state.handle_event({"type": "subagent_start", "name": "research-agent", "description": ""})
+ state._get_or_create_subagent("code-agent")
+ assert len(state.subagents) == 2
+
+
+# === _parse_todo_items ===
+
+class TestParseTodoItems:
+ def test_json_input(self):
+ import json
+ items = [{"status": "todo", "content": "Do X"}]
+ result = _parse_todo_items(json.dumps(items))
+ assert result is not None
+ assert len(result) == 1
+
+ def test_python_literal(self):
+ result = _parse_todo_items('[{"status": "done", "content": "Y"}]')
+ assert result is not None
+
+ def test_embedded_list(self):
+ text = "Updated todos:\n" + '[{"status": "todo", "content": "A"}]'
+ result = _parse_todo_items(text)
+ assert result is not None
+
+ def test_invalid_input(self):
+ assert _parse_todo_items("not a list at all") is None
+
+ def test_empty_string(self):
+ assert _parse_todo_items("") is None
+
+
+# === _build_todo_stats ===
+
+class TestBuildTodoStats:
+ def test_mixed_statuses(self):
+ items = [
+ {"status": "done"},
+ {"status": "active"},
+ {"status": "todo"},
+ {"status": "completed"},
+ ]
+ result = _build_todo_stats(items)
+ assert "1 active" in result
+ assert "2 done" in result
+ assert "1 pending" in result
+
+ def test_all_done(self):
+ items = [{"status": "done"}, {"status": "complete"}]
+ result = _build_todo_stats(items)
+ assert "2 done" in result
+ assert "active" not in result
+
+ def test_unknown_status_becomes_pending(self):
+ items = [{"status": "unknown_status"}]
+ result = _build_todo_stats(items)
+ assert "1 pending" in result
+
+ def test_empty_items(self):
+ result = _build_todo_stats([])
+ assert "0 items" in result
diff --git a/tests/test_stream_tracker.py b/tests/test_stream_tracker.py
new file mode 100644
index 0000000..2f4304e
--- /dev/null
+++ b/tests/test_stream_tracker.py
@@ -0,0 +1,99 @@
+"""Tests for EvoScientist/stream/tracker.py."""
+
+from EvoScientist.stream.tracker import ToolCallTracker, ToolCallInfo
+
+
+class TestToolCallTracker:
+ def test_update_and_get(self):
+ tracker = ToolCallTracker()
+ tracker.update("tc1", name="execute", args={"command": "ls"})
+ info = tracker.get("tc1")
+ assert info is not None
+ assert info.name == "execute"
+ assert info.args == {"command": "ls"}
+
+ def test_update_merges(self):
+ tracker = ToolCallTracker()
+ tracker.update("tc1", name="execute")
+ tracker.update("tc1", args={"command": "ls"})
+ info = tracker.get("tc1")
+ assert info.name == "execute"
+ assert info.args == {"command": "ls"}
+
+ def test_get_missing(self):
+ tracker = ToolCallTracker()
+ assert tracker.get("nonexistent") is None
+
+ def test_is_ready_true(self):
+ tracker = ToolCallTracker()
+ tracker.update("tc1", name="execute")
+ assert tracker.is_ready("tc1") is True
+
+ def test_is_ready_no_name(self):
+ tracker = ToolCallTracker()
+ tracker.update("tc1")
+ assert tracker.is_ready("tc1") is False
+
+ def test_is_ready_already_emitted(self):
+ tracker = ToolCallTracker()
+ tracker.update("tc1", name="execute")
+ tracker.mark_emitted("tc1")
+ assert tracker.is_ready("tc1") is False
+
+ def test_is_ready_missing_id(self):
+ tracker = ToolCallTracker()
+ assert tracker.is_ready("tc_missing") is False
+
+ def test_append_json_delta_and_finalize(self):
+ tracker = ToolCallTracker()
+ tracker.update("tc1", name="execute")
+ tracker.append_json_delta('{"comma')
+ tracker.append_json_delta('nd": "ls"}')
+ tracker.finalize_all()
+ info = tracker.get("tc1")
+ assert info.args == {"command": "ls"}
+ assert info.args_complete is True
+
+ def test_finalize_invalid_json(self):
+ tracker = ToolCallTracker()
+ tracker.update("tc1", name="execute")
+ tracker.append_json_delta("{invalid json")
+ tracker.finalize_all()
+ info = tracker.get("tc1")
+ # Args should remain empty since JSON is invalid
+ assert info.args == {}
+ assert info.args_complete is True
+
+ def test_emit_all_pending(self):
+ tracker = ToolCallTracker()
+ tracker.update("tc1", name="execute")
+ tracker.update("tc2", name="read_file")
+ pending = tracker.emit_all_pending()
+ assert len(pending) == 2
+ # All should now be marked emitted
+ assert tracker.get("tc1").emitted is True
+ assert tracker.get("tc2").emitted is True
+ # Second call should return empty
+ assert tracker.emit_all_pending() == []
+
+ def test_get_pending(self):
+ tracker = ToolCallTracker()
+ tracker.update("tc1", name="execute")
+ tracker.update("tc2", name="read_file")
+ tracker.mark_emitted("tc1")
+ pending = tracker.get_pending()
+ assert len(pending) == 1
+ assert pending[0].id == "tc2"
+
+ def test_get_all(self):
+ tracker = ToolCallTracker()
+ tracker.update("tc1", name="execute")
+ tracker.update("tc2", name="read_file")
+ assert len(tracker.get_all()) == 2
+
+ def test_clear(self):
+ tracker = ToolCallTracker()
+ tracker.update("tc1", name="execute")
+ tracker.clear()
+ assert tracker.get("tc1") is None
+ assert tracker.get_all() == []
diff --git a/tests/test_stream_utils.py b/tests/test_stream_utils.py
new file mode 100644
index 0000000..829ddd1
--- /dev/null
+++ b/tests/test_stream_utils.py
@@ -0,0 +1,199 @@
+"""Tests for EvoScientist/stream/utils.py pure functions."""
+
+from EvoScientist.stream.utils import (
+ is_success,
+ format_tool_compact,
+ truncate,
+ has_args,
+ count_lines,
+ truncate_with_line_hint,
+ _shorten_path,
+)
+
+
+# === is_success ===
+
+class TestIsSuccess:
+ def test_ok_prefix(self):
+ assert is_success("[OK] all good") is True
+
+ def test_failed_prefix(self):
+ assert is_success("[FAILED] bad") is False
+
+ def test_traceback(self):
+ assert is_success("Traceback (most recent call last)\n File ...") is False
+
+ def test_exception(self):
+ assert is_success("Exception: something went wrong") is False
+
+ def test_error(self):
+ assert is_success("Error: file not found") is False
+
+ def test_clean_output(self):
+ assert is_success("file1.py\nfile2.py") is True
+
+ def test_whitespace_stripped(self):
+ assert is_success(" [OK] with spaces ") is True
+
+
+# === format_tool_compact ===
+
+class TestFormatToolCompact:
+ def test_no_args(self):
+ assert format_tool_compact("execute", None) == "execute()"
+ assert format_tool_compact("execute", {}) == "execute()"
+
+ def test_execute(self):
+ result = format_tool_compact("execute", {"command": "ls -la"})
+ assert result == "execute(ls -la)"
+
+ def test_execute_long_command(self):
+ long_cmd = "x" * 60
+ result = format_tool_compact("execute", {"command": long_cmd})
+ assert len(result) < 70
+ assert result.endswith("...)")
+
+ def test_read_file(self):
+ result = format_tool_compact("read_file", {"path": "src/main.py"})
+ assert result == "read_file(src/main.py)"
+
+ def test_write_file(self):
+ result = format_tool_compact("write_file", {"path": "out.txt"})
+ assert result == "write_file(out.txt)"
+
+ def test_edit_file(self):
+ result = format_tool_compact("edit_file", {"path": "f.py"})
+ assert result == "edit_file(f.py)"
+
+ def test_glob(self):
+ result = format_tool_compact("glob", {"pattern": "*.py"})
+ assert result == "glob(*.py)"
+
+ def test_grep(self):
+ result = format_tool_compact("grep", {"pattern": "TODO", "path": "src/"})
+ assert result == "grep(TODO, src/)"
+
+ def test_ls(self):
+ assert format_tool_compact("ls", {"path": "/src"}) == "ls(/src)"
+
+ def test_write_todos_list(self):
+ todos = [{"status": "todo", "content": "a"}, {"status": "todo", "content": "b"}]
+ result = format_tool_compact("write_todos", {"todos": todos})
+ assert result == "write_todos(2 items)"
+
+ def test_write_todos_non_list(self):
+ result = format_tool_compact("write_todos", {"todos": "something"})
+ assert result == "write_todos(...)"
+
+ def test_read_todos(self):
+ assert format_tool_compact("read_todos", {}) == "read_todos()"
+
+ def test_task_with_type_and_desc(self):
+ result = format_tool_compact("task", {"subagent_type": "research-agent", "description": "Find papers"})
+ assert "Cooking with research-agent" in result
+ assert "Find papers" in result
+
+ def test_task_with_type_only(self):
+ result = format_tool_compact("task", {"subagent_type": "code-agent"})
+ assert result == "Cooking with code-agent"
+
+ def test_task_with_desc_only(self):
+ result = format_tool_compact("task", {"description": "do stuff"})
+ assert "Cooking with sub-agent" in result
+
+ def test_task_no_info(self):
+ result = format_tool_compact("task", {"other": "value"})
+ assert result == "Cooking with sub-agent"
+
+ def test_load_skill(self):
+ result = format_tool_compact("load_skill", {"skill_name": "vllm"})
+ assert result == "load_skill(vllm)"
+
+ def test_load_skill_name_key(self):
+ result = format_tool_compact("load_skill", {"name": "peft"})
+ assert result == "load_skill(peft)"
+
+ def test_tavily_search(self):
+ result = format_tool_compact("tavily_search", {"query": "python testing"})
+ assert result == "tavily_search(python testing)"
+
+ def test_think_tool(self):
+ result = format_tool_compact("think_tool", {"reflection": "need more data"})
+ assert result == "think_tool(need more data)"
+
+ def test_unknown_tool(self):
+ result = format_tool_compact("custom_tool", {"key": "value"})
+ assert "custom_tool(" in result
+ assert "key=value" in result
+
+ def test_unknown_tool_long_value(self):
+ result = format_tool_compact("custom_tool", {"key": "a" * 30})
+ assert "..." in result
+
+
+# === truncate ===
+
+class TestTruncate:
+ def test_within_limit(self):
+ assert truncate("hello", 10) == "hello"
+
+ def test_at_limit(self):
+ assert truncate("hello", 5) == "hello"
+
+ def test_over_limit(self):
+ result = truncate("hello world", 5)
+ assert result.startswith("hello")
+ assert "truncated" in result
+
+
+# === _shorten_path ===
+
+class TestShortenPath:
+ def test_short_path(self):
+ assert _shorten_path("src/main.py") == "src/main.py"
+
+ def test_long_path(self):
+ long_path = "a/b/c/d/e/f/g/h/i/j/k/l/m/n/o/p/q/r.py"
+ result = _shorten_path(long_path, max_len=20)
+ assert result.startswith(".../")
+ assert result.endswith("r.py")
+
+
+# === has_args ===
+
+class TestHasArgs:
+ def test_none(self):
+ assert has_args(None) is False
+
+ def test_empty_dict(self):
+ assert has_args({}) is False
+
+ def test_non_empty(self):
+ assert has_args({"key": "val"}) is True
+
+
+# === count_lines ===
+
+class TestCountLines:
+ def test_empty(self):
+ assert count_lines("") == 0
+
+ def test_single_line(self):
+ assert count_lines("hello") == 1
+
+ def test_multi_line(self):
+ assert count_lines("a\nb\nc") == 3
+
+
+# === truncate_with_line_hint ===
+
+class TestTruncateWithLineHint:
+ def test_within_limit(self):
+ text, remaining = truncate_with_line_hint("a\nb\nc", max_lines=5)
+ assert remaining == 0
+ assert "a" in text
+
+ def test_over_limit(self):
+ text, remaining = truncate_with_line_hint("a\nb\nc\nd\ne\nf", max_lines=3)
+ assert remaining == 3
+ assert "d" not in text
diff --git a/tests/test_tools.py b/tests/test_tools.py
new file mode 100644
index 0000000..a85a70f
--- /dev/null
+++ b/tests/test_tools.py
@@ -0,0 +1,18 @@
+"""Tests for EvoScientist/tools.py — only non-API tools."""
+
+from EvoScientist.tools import think_tool
+
+
+class TestThinkTool:
+ def test_returns_confirmation(self):
+ result = think_tool.invoke({"reflection": "I need more data on topic X"})
+ assert isinstance(result, str)
+ assert "I need more data on topic X" in result
+
+ def test_reflection_recorded(self):
+ result = think_tool.invoke({"reflection": "gap analysis"})
+ assert "Reflection recorded" in result
+
+ def test_empty_reflection(self):
+ result = think_tool.invoke({"reflection": ""})
+ assert "Reflection recorded" in result