diff --git a/EvoScientist/EvoScientist.py b/EvoScientist/EvoScientist.py index 50e0d09..defd61f 100644 --- a/EvoScientist/EvoScientist.py +++ b/EvoScientist/EvoScientist.py @@ -21,7 +21,7 @@ from deepagents import create_deep_agent from deepagents.backends import FilesystemBackend, CompositeBackend from langchain.chat_models import init_chat_model -from .backends import CustomSandboxBackend, ReadOnlyFilesystemBackend +from .backends import CustomSandboxBackend, MergedReadOnlyBackend from .middleware import create_skills_middleware from .prompts import RESEARCHER_INSTRUCTIONS, get_system_prompt from .utils import load_subagents @@ -40,7 +40,7 @@ MAX_ITERATIONS = 3 # Max delegation rounds # Workspace settings WORKSPACE_DIR = "./workspace/" -SKILLS_DIR = "./skills/" +SKILLS_DIR = str(Path(__file__).parent / "skills") SUBAGENTS_CONFIG = Path(__file__).parent / "subagent.yaml" # ============================================================================= @@ -76,10 +76,10 @@ else: virtual_mode=True, ) -# Skills backend: read-only access to ./skills/ -_skills_backend = ReadOnlyFilesystemBackend( - root_dir=SKILLS_DIR, - virtual_mode=True, +# Skills backend: merge user-installed (workspace) and system (package) skills +_skills_backend = MergedReadOnlyBackend( + primary_dir=str(Path(WORKSPACE_DIR) / "skills"), # user-installed, takes priority + secondary_dir=SKILLS_DIR, # package built-in, fallback ) # Composite backend: workspace as default, skills mounted at /skills/ @@ -134,9 +134,9 @@ def create_cli_agent(workspace_dir: str | None = None): virtual_mode=True, timeout=300, ) - sk_backend = ReadOnlyFilesystemBackend( - root_dir=SKILLS_DIR, - virtual_mode=True, + sk_backend = MergedReadOnlyBackend( + primary_dir=str(Path(workspace_dir) / "skills"), + secondary_dir=SKILLS_DIR, ) be = CompositeBackend( default=ws_backend, diff --git a/EvoScientist/backends.py b/EvoScientist/backends.py index dbf97bb..f6ed5d1 100644 --- a/EvoScientist/backends.py +++ b/EvoScientist/backends.py @@ -9,6 +9,8 @@ from deepagents.backends import FilesystemBackend from deepagents.backends.filesystem import WriteResult, EditResult from deepagents.backends.protocol import ( ExecuteResponse, + FileDownloadResponse, + FileUploadResponse, SandboxBackendProtocol, ) @@ -223,6 +225,30 @@ class MergedReadOnlyBackend: async def aedit(self, file_path: str, old_string: str, new_string: str, replace_all: bool = False) -> EditResult: return self.edit(file_path, old_string, new_string, replace_all) + # -- download / upload (required by BackendProtocol) -- + + def download_files(self, paths: list[str]) -> list[FileDownloadResponse]: + """Download files, trying primary then secondary.""" + responses: list[FileDownloadResponse] = [] + for path in paths: + resp = self._primary.download_files([path])[0] + if resp.error is not None: + resp = self._secondary.download_files([path])[0] + responses.append(resp) + return responses + + async def adownload_files(self, paths: list[str]) -> list[FileDownloadResponse]: + return self.download_files(paths) + + def upload_files(self, files: list[tuple[str, bytes]]) -> list[FileUploadResponse]: + return [ + FileUploadResponse(path=path, error="permission_denied") + for path, _ in files + ] + + async def aupload_files(self, files: list[tuple[str, bytes]]) -> list[FileUploadResponse]: + return self.upload_files(files) + class CustomSandboxBackend(FilesystemBackend, SandboxBackendProtocol): """ diff --git a/EvoScientist/middleware.py b/EvoScientist/middleware.py index 44734b1..74b9fe0 100644 --- a/EvoScientist/middleware.py +++ b/EvoScientist/middleware.py @@ -1,27 +1,35 @@ """Middleware configuration for the EvoScientist agent.""" -from deepagents.backends import FilesystemBackend +from pathlib import Path + from deepagents.middleware.skills import SkillsMiddleware +from .backends import MergedReadOnlyBackend + +_DEFAULT_SKILLS_DIR = str(Path(__file__).parent / "skills") + def create_skills_middleware( - skills_dir: str = "./skills/", + skills_dir: str = _DEFAULT_SKILLS_DIR, workspace_dir: str = "./workspace/", ) -> SkillsMiddleware: """Create a SkillsMiddleware that loads skills. - All skills (system and user-installed) live in ./skills/. - The --user flag in install_skill.py also installs to ./skills/. + Merges user-installed skills (workspace/skills/) with system skills + (package built-in). User skills take priority on name conflicts. Args: - skills_dir: Path to the skills directory - workspace_dir: Unused, kept for API compatibility + skills_dir: Path to the system skills directory (package built-in) + workspace_dir: Path to the workspace root (user skills live under workspace/skills/) Returns: Configured SkillsMiddleware instance """ - skills_backend = FilesystemBackend(root_dir=skills_dir, virtual_mode=True) + merged = MergedReadOnlyBackend( + primary_dir=str(Path(workspace_dir) / "skills"), + secondary_dir=skills_dir, + ) return SkillsMiddleware( - backend=skills_backend, + backend=merged, sources=["/"], ) diff --git a/skills/accelerate/SKILL.md b/EvoScientist/skills/accelerate/SKILL.md similarity index 100% rename from skills/accelerate/SKILL.md rename to EvoScientist/skills/accelerate/SKILL.md diff --git a/skills/accelerate/references/custom-plugins.md b/EvoScientist/skills/accelerate/references/custom-plugins.md similarity index 100% rename from skills/accelerate/references/custom-plugins.md rename to EvoScientist/skills/accelerate/references/custom-plugins.md diff --git a/skills/accelerate/references/megatron-integration.md b/EvoScientist/skills/accelerate/references/megatron-integration.md similarity index 100% rename from skills/accelerate/references/megatron-integration.md rename to EvoScientist/skills/accelerate/references/megatron-integration.md diff --git a/skills/accelerate/references/performance.md b/EvoScientist/skills/accelerate/references/performance.md similarity index 100% rename from skills/accelerate/references/performance.md rename to EvoScientist/skills/accelerate/references/performance.md diff --git a/skills/bitsandbytes/SKILL.md b/EvoScientist/skills/bitsandbytes/SKILL.md similarity index 100% rename from skills/bitsandbytes/SKILL.md rename to EvoScientist/skills/bitsandbytes/SKILL.md diff --git a/skills/bitsandbytes/references/memory-optimization.md b/EvoScientist/skills/bitsandbytes/references/memory-optimization.md similarity index 100% rename from skills/bitsandbytes/references/memory-optimization.md rename to EvoScientist/skills/bitsandbytes/references/memory-optimization.md diff --git a/skills/bitsandbytes/references/qlora-training.md b/EvoScientist/skills/bitsandbytes/references/qlora-training.md similarity index 100% rename from skills/bitsandbytes/references/qlora-training.md rename to EvoScientist/skills/bitsandbytes/references/qlora-training.md diff --git a/skills/bitsandbytes/references/quantization-formats.md b/EvoScientist/skills/bitsandbytes/references/quantization-formats.md similarity index 100% rename from skills/bitsandbytes/references/quantization-formats.md rename to EvoScientist/skills/bitsandbytes/references/quantization-formats.md diff --git a/skills/find-skills/SKILL.md b/EvoScientist/skills/find-skills/SKILL.md similarity index 100% rename from skills/find-skills/SKILL.md rename to EvoScientist/skills/find-skills/SKILL.md diff --git a/skills/find-skills/scripts/install_skill.py b/EvoScientist/skills/find-skills/scripts/install_skill.py similarity index 100% rename from skills/find-skills/scripts/install_skill.py rename to EvoScientist/skills/find-skills/scripts/install_skill.py diff --git a/skills/flash-attention/SKILL.md b/EvoScientist/skills/flash-attention/SKILL.md similarity index 100% rename from skills/flash-attention/SKILL.md rename to EvoScientist/skills/flash-attention/SKILL.md diff --git a/skills/flash-attention/references/benchmarks.md b/EvoScientist/skills/flash-attention/references/benchmarks.md similarity index 100% rename from skills/flash-attention/references/benchmarks.md rename to EvoScientist/skills/flash-attention/references/benchmarks.md diff --git a/skills/flash-attention/references/transformers-integration.md b/EvoScientist/skills/flash-attention/references/transformers-integration.md similarity index 100% rename from skills/flash-attention/references/transformers-integration.md rename to EvoScientist/skills/flash-attention/references/transformers-integration.md diff --git a/skills/llama-cpp/SKILL.md b/EvoScientist/skills/llama-cpp/SKILL.md similarity index 100% rename from skills/llama-cpp/SKILL.md rename to EvoScientist/skills/llama-cpp/SKILL.md diff --git a/skills/llama-cpp/references/optimization.md b/EvoScientist/skills/llama-cpp/references/optimization.md similarity index 100% rename from skills/llama-cpp/references/optimization.md rename to EvoScientist/skills/llama-cpp/references/optimization.md diff --git a/skills/llama-cpp/references/quantization.md b/EvoScientist/skills/llama-cpp/references/quantization.md similarity index 100% rename from skills/llama-cpp/references/quantization.md rename to EvoScientist/skills/llama-cpp/references/quantization.md diff --git a/skills/llama-cpp/references/server.md b/EvoScientist/skills/llama-cpp/references/server.md similarity index 100% rename from skills/llama-cpp/references/server.md rename to EvoScientist/skills/llama-cpp/references/server.md diff --git a/skills/lm-evaluation-harness/SKILL.md b/EvoScientist/skills/lm-evaluation-harness/SKILL.md similarity index 100% rename from skills/lm-evaluation-harness/SKILL.md rename to EvoScientist/skills/lm-evaluation-harness/SKILL.md diff --git a/skills/lm-evaluation-harness/references/api-evaluation.md b/EvoScientist/skills/lm-evaluation-harness/references/api-evaluation.md similarity index 100% rename from skills/lm-evaluation-harness/references/api-evaluation.md rename to EvoScientist/skills/lm-evaluation-harness/references/api-evaluation.md diff --git a/skills/lm-evaluation-harness/references/benchmark-guide.md b/EvoScientist/skills/lm-evaluation-harness/references/benchmark-guide.md similarity index 100% rename from skills/lm-evaluation-harness/references/benchmark-guide.md rename to EvoScientist/skills/lm-evaluation-harness/references/benchmark-guide.md diff --git a/skills/lm-evaluation-harness/references/custom-tasks.md b/EvoScientist/skills/lm-evaluation-harness/references/custom-tasks.md similarity index 100% rename from skills/lm-evaluation-harness/references/custom-tasks.md rename to EvoScientist/skills/lm-evaluation-harness/references/custom-tasks.md diff --git a/skills/lm-evaluation-harness/references/distributed-eval.md b/EvoScientist/skills/lm-evaluation-harness/references/distributed-eval.md similarity index 100% rename from skills/lm-evaluation-harness/references/distributed-eval.md rename to EvoScientist/skills/lm-evaluation-harness/references/distributed-eval.md diff --git a/skills/ml-paper-writing/.openskills.json b/EvoScientist/skills/ml-paper-writing/.openskills.json similarity index 100% rename from skills/ml-paper-writing/.openskills.json rename to EvoScientist/skills/ml-paper-writing/.openskills.json diff --git a/skills/ml-paper-writing/SKILL.md b/EvoScientist/skills/ml-paper-writing/SKILL.md similarity index 100% rename from skills/ml-paper-writing/SKILL.md rename to EvoScientist/skills/ml-paper-writing/SKILL.md diff --git a/skills/ml-paper-writing/references/checklists.md b/EvoScientist/skills/ml-paper-writing/references/checklists.md similarity index 100% rename from skills/ml-paper-writing/references/checklists.md rename to EvoScientist/skills/ml-paper-writing/references/checklists.md diff --git a/skills/ml-paper-writing/references/citation-workflow.md b/EvoScientist/skills/ml-paper-writing/references/citation-workflow.md similarity index 100% rename from skills/ml-paper-writing/references/citation-workflow.md rename to EvoScientist/skills/ml-paper-writing/references/citation-workflow.md diff --git a/skills/ml-paper-writing/references/reviewer-guidelines.md b/EvoScientist/skills/ml-paper-writing/references/reviewer-guidelines.md similarity index 100% rename from skills/ml-paper-writing/references/reviewer-guidelines.md rename to EvoScientist/skills/ml-paper-writing/references/reviewer-guidelines.md diff --git a/skills/ml-paper-writing/references/sources.md b/EvoScientist/skills/ml-paper-writing/references/sources.md similarity index 100% rename from skills/ml-paper-writing/references/sources.md rename to EvoScientist/skills/ml-paper-writing/references/sources.md diff --git a/skills/ml-paper-writing/references/writing-guide.md b/EvoScientist/skills/ml-paper-writing/references/writing-guide.md similarity index 100% rename from skills/ml-paper-writing/references/writing-guide.md rename to EvoScientist/skills/ml-paper-writing/references/writing-guide.md diff --git a/skills/ml-paper-writing/templates/README.md b/EvoScientist/skills/ml-paper-writing/templates/README.md similarity index 100% rename from skills/ml-paper-writing/templates/README.md rename to EvoScientist/skills/ml-paper-writing/templates/README.md diff --git a/skills/ml-paper-writing/templates/aaai2026/README.md b/EvoScientist/skills/ml-paper-writing/templates/aaai2026/README.md similarity index 100% rename from skills/ml-paper-writing/templates/aaai2026/README.md rename to EvoScientist/skills/ml-paper-writing/templates/aaai2026/README.md diff --git a/skills/ml-paper-writing/templates/aaai2026/aaai2026-unified-supp.tex b/EvoScientist/skills/ml-paper-writing/templates/aaai2026/aaai2026-unified-supp.tex similarity index 100% rename from skills/ml-paper-writing/templates/aaai2026/aaai2026-unified-supp.tex rename to EvoScientist/skills/ml-paper-writing/templates/aaai2026/aaai2026-unified-supp.tex diff --git a/skills/ml-paper-writing/templates/aaai2026/aaai2026-unified-template.tex b/EvoScientist/skills/ml-paper-writing/templates/aaai2026/aaai2026-unified-template.tex similarity index 100% rename from skills/ml-paper-writing/templates/aaai2026/aaai2026-unified-template.tex rename to EvoScientist/skills/ml-paper-writing/templates/aaai2026/aaai2026-unified-template.tex diff --git a/skills/ml-paper-writing/templates/aaai2026/aaai2026.bib b/EvoScientist/skills/ml-paper-writing/templates/aaai2026/aaai2026.bib similarity index 100% rename from skills/ml-paper-writing/templates/aaai2026/aaai2026.bib rename to EvoScientist/skills/ml-paper-writing/templates/aaai2026/aaai2026.bib diff --git a/skills/ml-paper-writing/templates/aaai2026/aaai2026.bst b/EvoScientist/skills/ml-paper-writing/templates/aaai2026/aaai2026.bst similarity index 100% rename from skills/ml-paper-writing/templates/aaai2026/aaai2026.bst rename to EvoScientist/skills/ml-paper-writing/templates/aaai2026/aaai2026.bst diff --git a/skills/ml-paper-writing/templates/aaai2026/aaai2026.sty b/EvoScientist/skills/ml-paper-writing/templates/aaai2026/aaai2026.sty similarity index 100% rename from skills/ml-paper-writing/templates/aaai2026/aaai2026.sty rename to EvoScientist/skills/ml-paper-writing/templates/aaai2026/aaai2026.sty diff --git a/skills/ml-paper-writing/templates/acl/README.md b/EvoScientist/skills/ml-paper-writing/templates/acl/README.md similarity index 100% rename from skills/ml-paper-writing/templates/acl/README.md rename to EvoScientist/skills/ml-paper-writing/templates/acl/README.md diff --git a/skills/ml-paper-writing/templates/acl/acl.sty b/EvoScientist/skills/ml-paper-writing/templates/acl/acl.sty similarity index 100% rename from skills/ml-paper-writing/templates/acl/acl.sty rename to EvoScientist/skills/ml-paper-writing/templates/acl/acl.sty diff --git a/skills/ml-paper-writing/templates/acl/acl_latex.tex b/EvoScientist/skills/ml-paper-writing/templates/acl/acl_latex.tex similarity index 100% rename from skills/ml-paper-writing/templates/acl/acl_latex.tex rename to EvoScientist/skills/ml-paper-writing/templates/acl/acl_latex.tex diff --git a/skills/ml-paper-writing/templates/acl/acl_lualatex.tex b/EvoScientist/skills/ml-paper-writing/templates/acl/acl_lualatex.tex similarity index 100% rename from skills/ml-paper-writing/templates/acl/acl_lualatex.tex rename to EvoScientist/skills/ml-paper-writing/templates/acl/acl_lualatex.tex diff --git a/skills/ml-paper-writing/templates/acl/acl_natbib.bst b/EvoScientist/skills/ml-paper-writing/templates/acl/acl_natbib.bst similarity index 100% rename from skills/ml-paper-writing/templates/acl/acl_natbib.bst rename to EvoScientist/skills/ml-paper-writing/templates/acl/acl_natbib.bst diff --git a/skills/ml-paper-writing/templates/acl/anthology.bib.txt b/EvoScientist/skills/ml-paper-writing/templates/acl/anthology.bib.txt similarity index 100% rename from skills/ml-paper-writing/templates/acl/anthology.bib.txt rename to EvoScientist/skills/ml-paper-writing/templates/acl/anthology.bib.txt diff --git a/skills/ml-paper-writing/templates/acl/custom.bib b/EvoScientist/skills/ml-paper-writing/templates/acl/custom.bib similarity index 100% rename from skills/ml-paper-writing/templates/acl/custom.bib rename to EvoScientist/skills/ml-paper-writing/templates/acl/custom.bib diff --git a/skills/ml-paper-writing/templates/acl/formatting.md b/EvoScientist/skills/ml-paper-writing/templates/acl/formatting.md similarity index 100% rename from skills/ml-paper-writing/templates/acl/formatting.md rename to EvoScientist/skills/ml-paper-writing/templates/acl/formatting.md diff --git a/skills/ml-paper-writing/templates/colm2025/README.md b/EvoScientist/skills/ml-paper-writing/templates/colm2025/README.md similarity index 100% rename from skills/ml-paper-writing/templates/colm2025/README.md rename to EvoScientist/skills/ml-paper-writing/templates/colm2025/README.md diff --git a/skills/ml-paper-writing/templates/colm2025/colm2025_conference.bib b/EvoScientist/skills/ml-paper-writing/templates/colm2025/colm2025_conference.bib similarity index 100% rename from skills/ml-paper-writing/templates/colm2025/colm2025_conference.bib rename to EvoScientist/skills/ml-paper-writing/templates/colm2025/colm2025_conference.bib diff --git a/skills/ml-paper-writing/templates/colm2025/colm2025_conference.bst b/EvoScientist/skills/ml-paper-writing/templates/colm2025/colm2025_conference.bst similarity index 100% rename from skills/ml-paper-writing/templates/colm2025/colm2025_conference.bst rename to EvoScientist/skills/ml-paper-writing/templates/colm2025/colm2025_conference.bst diff --git a/skills/ml-paper-writing/templates/colm2025/colm2025_conference.pdf b/EvoScientist/skills/ml-paper-writing/templates/colm2025/colm2025_conference.pdf similarity index 100% rename from skills/ml-paper-writing/templates/colm2025/colm2025_conference.pdf rename to EvoScientist/skills/ml-paper-writing/templates/colm2025/colm2025_conference.pdf diff --git a/skills/ml-paper-writing/templates/colm2025/colm2025_conference.sty b/EvoScientist/skills/ml-paper-writing/templates/colm2025/colm2025_conference.sty similarity index 100% rename from skills/ml-paper-writing/templates/colm2025/colm2025_conference.sty rename to EvoScientist/skills/ml-paper-writing/templates/colm2025/colm2025_conference.sty diff --git a/skills/ml-paper-writing/templates/colm2025/colm2025_conference.tex b/EvoScientist/skills/ml-paper-writing/templates/colm2025/colm2025_conference.tex similarity index 100% rename from skills/ml-paper-writing/templates/colm2025/colm2025_conference.tex rename to EvoScientist/skills/ml-paper-writing/templates/colm2025/colm2025_conference.tex diff --git a/skills/ml-paper-writing/templates/colm2025/fancyhdr.sty b/EvoScientist/skills/ml-paper-writing/templates/colm2025/fancyhdr.sty similarity index 100% rename from skills/ml-paper-writing/templates/colm2025/fancyhdr.sty rename to EvoScientist/skills/ml-paper-writing/templates/colm2025/fancyhdr.sty diff --git a/skills/ml-paper-writing/templates/colm2025/math_commands.tex b/EvoScientist/skills/ml-paper-writing/templates/colm2025/math_commands.tex similarity index 100% rename from skills/ml-paper-writing/templates/colm2025/math_commands.tex rename to EvoScientist/skills/ml-paper-writing/templates/colm2025/math_commands.tex diff --git a/skills/ml-paper-writing/templates/colm2025/natbib.sty b/EvoScientist/skills/ml-paper-writing/templates/colm2025/natbib.sty similarity index 100% rename from skills/ml-paper-writing/templates/colm2025/natbib.sty rename to EvoScientist/skills/ml-paper-writing/templates/colm2025/natbib.sty diff --git a/skills/ml-paper-writing/templates/iclr2026/fancyhdr.sty b/EvoScientist/skills/ml-paper-writing/templates/iclr2026/fancyhdr.sty similarity index 100% rename from skills/ml-paper-writing/templates/iclr2026/fancyhdr.sty rename to EvoScientist/skills/ml-paper-writing/templates/iclr2026/fancyhdr.sty diff --git a/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.bib b/EvoScientist/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.bib similarity index 100% rename from skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.bib rename to EvoScientist/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.bib diff --git a/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.bst b/EvoScientist/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.bst similarity index 100% rename from skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.bst rename to EvoScientist/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.bst diff --git a/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.pdf b/EvoScientist/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.pdf similarity index 100% rename from skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.pdf rename to EvoScientist/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.pdf diff --git a/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.sty b/EvoScientist/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.sty similarity index 100% rename from skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.sty rename to EvoScientist/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.sty diff --git a/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.tex b/EvoScientist/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.tex similarity index 100% rename from skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.tex rename to EvoScientist/skills/ml-paper-writing/templates/iclr2026/iclr2026_conference.tex diff --git a/skills/ml-paper-writing/templates/iclr2026/math_commands.tex b/EvoScientist/skills/ml-paper-writing/templates/iclr2026/math_commands.tex similarity index 100% rename from skills/ml-paper-writing/templates/iclr2026/math_commands.tex rename to EvoScientist/skills/ml-paper-writing/templates/iclr2026/math_commands.tex diff --git a/skills/ml-paper-writing/templates/iclr2026/natbib.sty b/EvoScientist/skills/ml-paper-writing/templates/iclr2026/natbib.sty similarity index 100% rename from skills/ml-paper-writing/templates/iclr2026/natbib.sty rename to EvoScientist/skills/ml-paper-writing/templates/iclr2026/natbib.sty diff --git a/skills/ml-paper-writing/templates/icml2026/algorithm.sty b/EvoScientist/skills/ml-paper-writing/templates/icml2026/algorithm.sty similarity index 100% rename from skills/ml-paper-writing/templates/icml2026/algorithm.sty rename to EvoScientist/skills/ml-paper-writing/templates/icml2026/algorithm.sty diff --git a/skills/ml-paper-writing/templates/icml2026/algorithmic.sty b/EvoScientist/skills/ml-paper-writing/templates/icml2026/algorithmic.sty similarity index 100% rename from skills/ml-paper-writing/templates/icml2026/algorithmic.sty rename to EvoScientist/skills/ml-paper-writing/templates/icml2026/algorithmic.sty diff --git a/skills/ml-paper-writing/templates/icml2026/example_paper.bib b/EvoScientist/skills/ml-paper-writing/templates/icml2026/example_paper.bib similarity index 100% rename from skills/ml-paper-writing/templates/icml2026/example_paper.bib rename to EvoScientist/skills/ml-paper-writing/templates/icml2026/example_paper.bib diff --git a/skills/ml-paper-writing/templates/icml2026/example_paper.pdf b/EvoScientist/skills/ml-paper-writing/templates/icml2026/example_paper.pdf similarity index 100% rename from skills/ml-paper-writing/templates/icml2026/example_paper.pdf rename to EvoScientist/skills/ml-paper-writing/templates/icml2026/example_paper.pdf diff --git a/skills/ml-paper-writing/templates/icml2026/example_paper.tex b/EvoScientist/skills/ml-paper-writing/templates/icml2026/example_paper.tex similarity index 100% rename from skills/ml-paper-writing/templates/icml2026/example_paper.tex rename to EvoScientist/skills/ml-paper-writing/templates/icml2026/example_paper.tex diff --git a/skills/ml-paper-writing/templates/icml2026/fancyhdr.sty b/EvoScientist/skills/ml-paper-writing/templates/icml2026/fancyhdr.sty similarity index 100% rename from skills/ml-paper-writing/templates/icml2026/fancyhdr.sty rename to EvoScientist/skills/ml-paper-writing/templates/icml2026/fancyhdr.sty diff --git a/skills/ml-paper-writing/templates/icml2026/icml2026.bst b/EvoScientist/skills/ml-paper-writing/templates/icml2026/icml2026.bst similarity index 100% rename from skills/ml-paper-writing/templates/icml2026/icml2026.bst rename to EvoScientist/skills/ml-paper-writing/templates/icml2026/icml2026.bst diff --git a/skills/ml-paper-writing/templates/icml2026/icml2026.sty b/EvoScientist/skills/ml-paper-writing/templates/icml2026/icml2026.sty similarity index 100% rename from skills/ml-paper-writing/templates/icml2026/icml2026.sty rename to EvoScientist/skills/ml-paper-writing/templates/icml2026/icml2026.sty diff --git a/skills/ml-paper-writing/templates/icml2026/icml_numpapers.pdf b/EvoScientist/skills/ml-paper-writing/templates/icml2026/icml_numpapers.pdf similarity index 100% rename from skills/ml-paper-writing/templates/icml2026/icml_numpapers.pdf rename to EvoScientist/skills/ml-paper-writing/templates/icml2026/icml_numpapers.pdf diff --git a/skills/ml-paper-writing/templates/neurips2025/Makefile b/EvoScientist/skills/ml-paper-writing/templates/neurips2025/Makefile similarity index 100% rename from skills/ml-paper-writing/templates/neurips2025/Makefile rename to EvoScientist/skills/ml-paper-writing/templates/neurips2025/Makefile diff --git a/skills/ml-paper-writing/templates/neurips2025/extra_pkgs.tex b/EvoScientist/skills/ml-paper-writing/templates/neurips2025/extra_pkgs.tex similarity index 100% rename from skills/ml-paper-writing/templates/neurips2025/extra_pkgs.tex rename to EvoScientist/skills/ml-paper-writing/templates/neurips2025/extra_pkgs.tex diff --git a/skills/ml-paper-writing/templates/neurips2025/main.tex b/EvoScientist/skills/ml-paper-writing/templates/neurips2025/main.tex similarity index 100% rename from skills/ml-paper-writing/templates/neurips2025/main.tex rename to EvoScientist/skills/ml-paper-writing/templates/neurips2025/main.tex diff --git a/skills/ml-paper-writing/templates/neurips2025/neurips.sty b/EvoScientist/skills/ml-paper-writing/templates/neurips2025/neurips.sty similarity index 100% rename from skills/ml-paper-writing/templates/neurips2025/neurips.sty rename to EvoScientist/skills/ml-paper-writing/templates/neurips2025/neurips.sty diff --git a/skills/peft/SKILL.md b/EvoScientist/skills/peft/SKILL.md similarity index 100% rename from skills/peft/SKILL.md rename to EvoScientist/skills/peft/SKILL.md diff --git a/skills/peft/references/advanced-usage.md b/EvoScientist/skills/peft/references/advanced-usage.md similarity index 100% rename from skills/peft/references/advanced-usage.md rename to EvoScientist/skills/peft/references/advanced-usage.md diff --git a/skills/peft/references/troubleshooting.md b/EvoScientist/skills/peft/references/troubleshooting.md similarity index 100% rename from skills/peft/references/troubleshooting.md rename to EvoScientist/skills/peft/references/troubleshooting.md diff --git a/skills/ray-data/SKILL.md b/EvoScientist/skills/ray-data/SKILL.md similarity index 100% rename from skills/ray-data/SKILL.md rename to EvoScientist/skills/ray-data/SKILL.md diff --git a/skills/ray-data/references/integration.md b/EvoScientist/skills/ray-data/references/integration.md similarity index 100% rename from skills/ray-data/references/integration.md rename to EvoScientist/skills/ray-data/references/integration.md diff --git a/skills/ray-data/references/transformations.md b/EvoScientist/skills/ray-data/references/transformations.md similarity index 100% rename from skills/ray-data/references/transformations.md rename to EvoScientist/skills/ray-data/references/transformations.md diff --git a/skills/skill-creator/LICENSE.txt b/EvoScientist/skills/skill-creator/LICENSE.txt similarity index 100% rename from skills/skill-creator/LICENSE.txt rename to EvoScientist/skills/skill-creator/LICENSE.txt diff --git a/skills/skill-creator/SKILL.md b/EvoScientist/skills/skill-creator/SKILL.md similarity index 100% rename from skills/skill-creator/SKILL.md rename to EvoScientist/skills/skill-creator/SKILL.md diff --git a/skills/skill-creator/references/output-patterns.md b/EvoScientist/skills/skill-creator/references/output-patterns.md similarity index 100% rename from skills/skill-creator/references/output-patterns.md rename to EvoScientist/skills/skill-creator/references/output-patterns.md diff --git a/skills/skill-creator/references/workflows.md b/EvoScientist/skills/skill-creator/references/workflows.md similarity index 100% rename from skills/skill-creator/references/workflows.md rename to EvoScientist/skills/skill-creator/references/workflows.md diff --git a/skills/skill-creator/scripts/init_skill.py b/EvoScientist/skills/skill-creator/scripts/init_skill.py similarity index 100% rename from skills/skill-creator/scripts/init_skill.py rename to EvoScientist/skills/skill-creator/scripts/init_skill.py diff --git a/skills/skill-creator/scripts/package_skill.py b/EvoScientist/skills/skill-creator/scripts/package_skill.py similarity index 100% rename from skills/skill-creator/scripts/package_skill.py rename to EvoScientist/skills/skill-creator/scripts/package_skill.py diff --git a/skills/skill-creator/scripts/quick_validate.py b/EvoScientist/skills/skill-creator/scripts/quick_validate.py similarity index 100% rename from skills/skill-creator/scripts/quick_validate.py rename to EvoScientist/skills/skill-creator/scripts/quick_validate.py diff --git a/README.md b/README.md index 341df92..879285c 100644 --- a/README.md +++ b/README.md @@ -9,11 +9,12 @@ Typing SVG -[![Project Page](https://img.shields.io/badge/Project-Page-4285F4?style=for-the-badge&logo=googlelens&logoColor=4285F4)]() +[![PyPI](https://img.shields.io/badge/PyPI-EvoScientist%20v0.0.1-3da9fc?style=for-the-badge&logo=python&logoColor=3da9fc)](https://pypi.org/project/EvoScientist/) +[![Project Page](https://img.shields.io/badge/Project-Page-ff8e3c?style=for-the-badge&logo=googlelens&logoColor=ff8e3c)]() [![arXiv](https://img.shields.io/badge/arXiv-xxxx.xxxx-b31b1b?style=for-the-badge&logo=arxiv&logoColor=b31b1b)]() -[![Gradio Demo](https://img.shields.io/badge/Gradio-Online_Demo-FFCC00?style=for-the-badge&logo=gradio&logoColor=yellow&labelColor=grey)]() -[![Evaluation Split](https://img.shields.io/badge/HF-Test_Dataset-AECBFA?style=for-the-badge&logo=huggingface&logoColor=FFCC00&labelColor=grey)]() [![License](https://img.shields.io/badge/License-MIT-green?style=for-the-badge)]() + diff --git a/pyproject.toml b/pyproject.toml index a1857f1..bda752c 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "EvoScientist" -version = "0.1.0" +version = "0.0.1" description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery" readme = "README.md" requires-python = ">=3.11" @@ -49,4 +49,7 @@ build-backend = "setuptools.build_meta" include = ["EvoScientist*"] [tool.setuptools.package-data] -EvoScientist = ["subagent.yaml"] +EvoScientist = ["subagent.yaml", "skills/**/*"] + +[tool.pytest.ini_options] +testpaths = ["tests"] diff --git a/skills/clip/SKILL.md b/skills/clip/SKILL.md deleted file mode 100644 index e5282ae..0000000 --- a/skills/clip/SKILL.md +++ /dev/null @@ -1,253 +0,0 @@ ---- -name: clip -description: OpenAI's model connecting vision and language. Enables zero-shot image classification, image-text matching, and cross-modal retrieval. Trained on 400M image-text pairs. Use for image search, content moderation, or vision-language tasks without fine-tuning. Best for general-purpose image understanding. -version: 1.0.0 -author: Orchestra Research -license: MIT -tags: [Multimodal, CLIP, Vision-Language, Zero-Shot, Image Classification, OpenAI, Image Search, Cross-Modal Retrieval, Content Moderation] -dependencies: [transformers, torch, pillow] ---- - -# CLIP - Contrastive Language-Image Pre-Training - -OpenAI's model that understands images from natural language. - -## When to use CLIP - -**Use when:** -- Zero-shot image classification (no training data needed) -- Image-text similarity/matching -- Semantic image search -- Content moderation (detect NSFW, violence) -- Visual question answering -- Cross-modal retrieval (image→text, text→image) - -**Metrics**: -- **25,300+ GitHub stars** -- Trained on 400M image-text pairs -- Matches ResNet-50 on ImageNet (zero-shot) -- MIT License - -**Use alternatives instead**: -- **BLIP-2**: Better captioning -- **LLaVA**: Vision-language chat -- **Segment Anything**: Image segmentation - -## Quick start - -### Installation - -```bash -pip install git+https://github.com/openai/CLIP.git -pip install torch torchvision ftfy regex tqdm -``` - -### Zero-shot classification - -```python -import torch -import clip -from PIL import Image - -# Load model -device = "cuda" if torch.cuda.is_available() else "cpu" -model, preprocess = clip.load("ViT-B/32", device=device) - -# Load image -image = preprocess(Image.open("photo.jpg")).unsqueeze(0).to(device) - -# Define possible labels -text = clip.tokenize(["a dog", "a cat", "a bird", "a car"]).to(device) - -# Compute similarity -with torch.no_grad(): - image_features = model.encode_image(image) - text_features = model.encode_text(text) - - # Cosine similarity - logits_per_image, logits_per_text = model(image, text) - probs = logits_per_image.softmax(dim=-1).cpu().numpy() - -# Print results -labels = ["a dog", "a cat", "a bird", "a car"] -for label, prob in zip(labels, probs[0]): - print(f"{label}: {prob:.2%}") -``` - -## Available models - -```python -# Models (sorted by size) -models = [ - "RN50", # ResNet-50 - "RN101", # ResNet-101 - "ViT-B/32", # Vision Transformer (recommended) - "ViT-B/16", # Better quality, slower - "ViT-L/14", # Best quality, slowest -] - -model, preprocess = clip.load("ViT-B/32") -``` - -| Model | Parameters | Speed | Quality | -|-------|------------|-------|---------| -| RN50 | 102M | Fast | Good | -| ViT-B/32 | 151M | Medium | Better | -| ViT-L/14 | 428M | Slow | Best | - -## Image-text similarity - -```python -# Compute embeddings -image_features = model.encode_image(image) -text_features = model.encode_text(text) - -# Normalize -image_features /= image_features.norm(dim=-1, keepdim=True) -text_features /= text_features.norm(dim=-1, keepdim=True) - -# Cosine similarity -similarity = (image_features @ text_features.T).item() -print(f"Similarity: {similarity:.4f}") -``` - -## Semantic image search - -```python -# Index images -image_paths = ["img1.jpg", "img2.jpg", "img3.jpg"] -image_embeddings = [] - -for img_path in image_paths: - image = preprocess(Image.open(img_path)).unsqueeze(0).to(device) - with torch.no_grad(): - embedding = model.encode_image(image) - embedding /= embedding.norm(dim=-1, keepdim=True) - image_embeddings.append(embedding) - -image_embeddings = torch.cat(image_embeddings) - -# Search with text query -query = "a sunset over the ocean" -text_input = clip.tokenize([query]).to(device) -with torch.no_grad(): - text_embedding = model.encode_text(text_input) - text_embedding /= text_embedding.norm(dim=-1, keepdim=True) - -# Find most similar images -similarities = (text_embedding @ image_embeddings.T).squeeze(0) -top_k = similarities.topk(3) - -for idx, score in zip(top_k.indices, top_k.values): - print(f"{image_paths[idx]}: {score:.3f}") -``` - -## Content moderation - -```python -# Define categories -categories = [ - "safe for work", - "not safe for work", - "violent content", - "graphic content" -] - -text = clip.tokenize(categories).to(device) - -# Check image -with torch.no_grad(): - logits_per_image, _ = model(image, text) - probs = logits_per_image.softmax(dim=-1) - -# Get classification -max_idx = probs.argmax().item() -max_prob = probs[0, max_idx].item() - -print(f"Category: {categories[max_idx]} ({max_prob:.2%})") -``` - -## Batch processing - -```python -# Process multiple images -images = [preprocess(Image.open(f"img{i}.jpg")) for i in range(10)] -images = torch.stack(images).to(device) - -with torch.no_grad(): - image_features = model.encode_image(images) - image_features /= image_features.norm(dim=-1, keepdim=True) - -# Batch text -texts = ["a dog", "a cat", "a bird"] -text_tokens = clip.tokenize(texts).to(device) - -with torch.no_grad(): - text_features = model.encode_text(text_tokens) - text_features /= text_features.norm(dim=-1, keepdim=True) - -# Similarity matrix (10 images × 3 texts) -similarities = image_features @ text_features.T -print(similarities.shape) # (10, 3) -``` - -## Integration with vector databases - -```python -# Store CLIP embeddings in Chroma/FAISS -import chromadb - -client = chromadb.Client() -collection = client.create_collection("image_embeddings") - -# Add image embeddings -for img_path, embedding in zip(image_paths, image_embeddings): - collection.add( - embeddings=[embedding.cpu().numpy().tolist()], - metadatas=[{"path": img_path}], - ids=[img_path] - ) - -# Query with text -query = "a sunset" -text_embedding = model.encode_text(clip.tokenize([query])) -results = collection.query( - query_embeddings=[text_embedding.cpu().numpy().tolist()], - n_results=5 -) -``` - -## Best practices - -1. **Use ViT-B/32 for most cases** - Good balance -2. **Normalize embeddings** - Required for cosine similarity -3. **Batch processing** - More efficient -4. **Cache embeddings** - Expensive to recompute -5. **Use descriptive labels** - Better zero-shot performance -6. **GPU recommended** - 10-50× faster -7. **Preprocess images** - Use provided preprocess function - -## Performance - -| Operation | CPU | GPU (V100) | -|-----------|-----|------------| -| Image encoding | ~200ms | ~20ms | -| Text encoding | ~50ms | ~5ms | -| Similarity compute | <1ms | <1ms | - -## Limitations - -1. **Not for fine-grained tasks** - Best for broad categories -2. **Requires descriptive text** - Vague labels perform poorly -3. **Biased on web data** - May have dataset biases -4. **No bounding boxes** - Whole image only -5. **Limited spatial understanding** - Position/counting weak - -## Resources - -- **GitHub**: https://github.com/openai/CLIP ⭐ 25,300+ -- **Paper**: https://arxiv.org/abs/2103.00020 -- **Colab**: https://colab.research.google.com/github/openai/clip/ -- **License**: MIT - - diff --git a/skills/clip/references/applications.md b/skills/clip/references/applications.md deleted file mode 100644 index 38e9a05..0000000 --- a/skills/clip/references/applications.md +++ /dev/null @@ -1,207 +0,0 @@ -# CLIP Applications Guide - -Practical applications and use cases for CLIP. - -## Zero-shot image classification - -```python -import torch -import clip -from PIL import Image - -model, preprocess = clip.load("ViT-B/32") - -# Define categories -categories = [ - "a photo of a dog", - "a photo of a cat", - "a photo of a bird", - "a photo of a car", - "a photo of a person" -] - -# Prepare image -image = preprocess(Image.open("photo.jpg")).unsqueeze(0) -text = clip.tokenize(categories) - -# Classify -with torch.no_grad(): - image_features = model.encode_image(image) - text_features = model.encode_text(text) - - logits_per_image, _ = model(image, text) - probs = logits_per_image.softmax(dim=-1).cpu().numpy() - -# Print results -for category, prob in zip(categories, probs[0]): - print(f"{category}: {prob:.2%}") -``` - -## Semantic image search - -```python -# Index images -image_database = [] -image_paths = ["img1.jpg", "img2.jpg", "img3.jpg"] - -for img_path in image_paths: - image = preprocess(Image.open(img_path)).unsqueeze(0) - with torch.no_grad(): - features = model.encode_image(image) - features /= features.norm(dim=-1, keepdim=True) - image_database.append((img_path, features)) - -# Search with text -query = "a sunset over mountains" -text_input = clip.tokenize([query]) - -with torch.no_grad(): - text_features = model.encode_text(text_input) - text_features /= text_features.norm(dim=-1, keepdim=True) - -# Find matches -similarities = [] -for img_path, img_features in image_database: - similarity = (text_features @ img_features.T).item() - similarities.append((img_path, similarity)) - -# Sort by similarity -similarities.sort(key=lambda x: x[1], reverse=True) -for img_path, score in similarities[:3]: - print(f"{img_path}: {score:.3f}") -``` - -## Content moderation - -```python -# Define safety categories -categories = [ - "safe for work content", - "not safe for work content", - "violent or graphic content", - "hate speech or offensive content", - "spam or misleading content" -] - -text = clip.tokenize(categories) - -# Check image -with torch.no_grad(): - logits, _ = model(image, text) - probs = logits.softmax(dim=-1) - -# Get classification -max_idx = probs.argmax().item() -confidence = probs[0, max_idx].item() - -if confidence > 0.7: - print(f"Classified as: {categories[max_idx]} ({confidence:.2%})") -else: - print(f"Uncertain classification (confidence: {confidence:.2%})") -``` - -## Image-to-text retrieval - -```python -# Text database -captions = [ - "A beautiful sunset over the ocean", - "A cute dog playing in the park", - "A modern city skyline at night", - "A delicious pizza with toppings" -] - -# Encode captions -caption_features = [] -for caption in captions: - text = clip.tokenize([caption]) - with torch.no_grad(): - features = model.encode_text(text) - features /= features.norm(dim=-1, keepdim=True) - caption_features.append(features) - -caption_features = torch.cat(caption_features) - -# Find matching captions for image -with torch.no_grad(): - image_features = model.encode_image(image) - image_features /= image_features.norm(dim=-1, keepdim=True) - -similarities = (image_features @ caption_features.T).squeeze(0) -top_k = similarities.topk(3) - -for idx, score in zip(top_k.indices, top_k.values): - print(f"{captions[idx]}: {score:.3f}") -``` - -## Visual question answering - -```python -# Create yes/no questions -image = preprocess(Image.open("photo.jpg")).unsqueeze(0) - -questions = [ - "a photo showing people", - "a photo showing animals", - "a photo taken indoors", - "a photo taken outdoors", - "a photo taken during daytime", - "a photo taken at night" -] - -text = clip.tokenize(questions) - -with torch.no_grad(): - logits, _ = model(image, text) - probs = logits.softmax(dim=-1) - -# Answer questions -for question, prob in zip(questions, probs[0]): - answer = "Yes" if prob > 0.5 else "No" - print(f"{question}: {answer} ({prob:.2%})") -``` - -## Image deduplication - -```python -# Detect duplicate/similar images -def compute_similarity(img1_path, img2_path): - img1 = preprocess(Image.open(img1_path)).unsqueeze(0) - img2 = preprocess(Image.open(img2_path)).unsqueeze(0) - - with torch.no_grad(): - feat1 = model.encode_image(img1) - feat2 = model.encode_image(img2) - - feat1 /= feat1.norm(dim=-1, keepdim=True) - feat2 /= feat2.norm(dim=-1, keepdim=True) - - similarity = (feat1 @ feat2.T).item() - - return similarity - -# Check for duplicates -threshold = 0.95 -image_pairs = [("img1.jpg", "img2.jpg"), ("img1.jpg", "img3.jpg")] - -for img1, img2 in image_pairs: - sim = compute_similarity(img1, img2) - if sim > threshold: - print(f"{img1} and {img2} are duplicates (similarity: {sim:.3f})") -``` - -## Best practices - -1. **Use descriptive labels** - "a photo of X" works better than just "X" -2. **Normalize embeddings** - Always normalize for cosine similarity -3. **Batch processing** - Process multiple images/texts together -4. **Cache embeddings** - Expensive to recompute -5. **Set appropriate thresholds** - Test on validation data -6. **Use GPU** - 10-50× faster than CPU -7. **Consider model size** - ViT-B/32 good default, ViT-L/14 for best quality - -## Resources - -- **Paper**: https://arxiv.org/abs/2103.00020 -- **GitHub**: https://github.com/openai/CLIP -- **Colab**: https://colab.research.google.com/github/openai/clip/ diff --git a/skills/langgraph-docs/SKILL.md b/skills/langgraph-docs/SKILL.md deleted file mode 100644 index 360a17f..0000000 --- a/skills/langgraph-docs/SKILL.md +++ /dev/null @@ -1,36 +0,0 @@ ---- -name: langgraph-docs -description: Use this skill for requests related to LangGraph in order to fetch relevant documentation to provide accurate, up-to-date guidance. ---- - -# langgraph-docs - -## Overview - -This skill explains how to access LangGraph Python documentation to help answer questions and guide implementation. - -## Instructions - -### 1. Fetch the Documentation Index - -Use the fetch_url tool to read the following URL: -https://docs.langchain.com/llms.txt - -This provides a structured list of all available documentation with descriptions. - -### 2. Select Relevant Documentation - -Based on the question, identify 2-4 most relevant documentation URLs from the index. Prioritize: - -- Specific how-to guides for implementation questions -- Core concept pages for understanding questions -- Tutorials for end-to-end examples -- Reference docs for API details - -### 3. Fetch Selected Documentation - -Use the fetch_url tool to read the selected documentation URLs. - -### 4. Provide Accurate Guidance - -After reading the documentation, complete the user's request. \ No newline at end of file diff --git a/skills/tensorboard/SKILL.md b/skills/tensorboard/SKILL.md deleted file mode 100644 index 2764d52..0000000 --- a/skills/tensorboard/SKILL.md +++ /dev/null @@ -1,629 +0,0 @@ ---- -name: tensorboard -description: Visualize training metrics, debug models with histograms, compare experiments, visualize model graphs, and profile performance with TensorBoard - Google's ML visualization toolkit -version: 1.0.0 -author: Orchestra Research -license: MIT -tags: [MLOps, TensorBoard, Visualization, Training Metrics, Model Debugging, PyTorch, TensorFlow, Experiment Tracking, Performance Profiling] -dependencies: [tensorboard, torch, tensorflow] ---- - -# TensorBoard: Visualization Toolkit for ML - -## When to Use This Skill - -Use TensorBoard when you need to: -- **Visualize training metrics** like loss and accuracy over time -- **Debug models** with histograms and distributions -- **Compare experiments** across multiple runs -- **Visualize model graphs** and architecture -- **Project embeddings** to lower dimensions (t-SNE, PCA) -- **Track hyperparameter** experiments -- **Profile performance** and identify bottlenecks -- **Visualize images and text** during training - -**Users**: 20M+ downloads/year | **GitHub Stars**: 27k+ | **License**: Apache 2.0 - -## Installation - -```bash -# Install TensorBoard -pip install tensorboard - -# PyTorch integration -pip install torch torchvision tensorboard - -# TensorFlow integration (TensorBoard included) -pip install tensorflow - -# Launch TensorBoard -tensorboard --logdir=runs -# Access at http://localhost:6006 -``` - -## Quick Start - -### PyTorch - -```python -from torch.utils.tensorboard import SummaryWriter - -# Create writer -writer = SummaryWriter('runs/experiment_1') - -# Training loop -for epoch in range(10): - train_loss = train_epoch() - val_acc = validate() - - # Log metrics - writer.add_scalar('Loss/train', train_loss, epoch) - writer.add_scalar('Accuracy/val', val_acc, epoch) - -# Close writer -writer.close() - -# Launch: tensorboard --logdir=runs -``` - -### TensorFlow/Keras - -```python -import tensorflow as tf - -# Create callback -tensorboard_callback = tf.keras.callbacks.TensorBoard( - log_dir='logs/fit', - histogram_freq=1 -) - -# Train model -model.fit( - x_train, y_train, - epochs=10, - validation_data=(x_val, y_val), - callbacks=[tensorboard_callback] -) - -# Launch: tensorboard --logdir=logs -``` - -## Core Concepts - -### 1. SummaryWriter (PyTorch) - -```python -from torch.utils.tensorboard import SummaryWriter - -# Default directory: runs/CURRENT_DATETIME -writer = SummaryWriter() - -# Custom directory -writer = SummaryWriter('runs/experiment_1') - -# Custom comment (appended to default directory) -writer = SummaryWriter(comment='baseline') - -# Log data -writer.add_scalar('Loss/train', 0.5, step=0) -writer.add_scalar('Loss/train', 0.3, step=1) - -# Flush and close -writer.flush() -writer.close() -``` - -### 2. Logging Scalars - -```python -# PyTorch -from torch.utils.tensorboard import SummaryWriter -writer = SummaryWriter() - -for epoch in range(100): - train_loss = train() - val_loss = validate() - - # Log individual metrics - writer.add_scalar('Loss/train', train_loss, epoch) - writer.add_scalar('Loss/val', val_loss, epoch) - writer.add_scalar('Accuracy/train', train_acc, epoch) - writer.add_scalar('Accuracy/val', val_acc, epoch) - - # Learning rate - lr = optimizer.param_groups[0]['lr'] - writer.add_scalar('Learning_rate', lr, epoch) - -writer.close() -``` - -```python -# TensorFlow -import tensorflow as tf - -train_summary_writer = tf.summary.create_file_writer('logs/train') -val_summary_writer = tf.summary.create_file_writer('logs/val') - -for epoch in range(100): - with train_summary_writer.as_default(): - tf.summary.scalar('loss', train_loss, step=epoch) - tf.summary.scalar('accuracy', train_acc, step=epoch) - - with val_summary_writer.as_default(): - tf.summary.scalar('loss', val_loss, step=epoch) - tf.summary.scalar('accuracy', val_acc, step=epoch) -``` - -### 3. Logging Multiple Scalars - -```python -# PyTorch: Group related metrics -writer.add_scalars('Loss', { - 'train': train_loss, - 'validation': val_loss, - 'test': test_loss -}, epoch) - -writer.add_scalars('Metrics', { - 'accuracy': accuracy, - 'precision': precision, - 'recall': recall, - 'f1': f1_score -}, epoch) -``` - -### 4. Logging Images - -```python -# PyTorch -import torch -from torchvision.utils import make_grid - -# Single image -writer.add_image('Input/sample', img_tensor, epoch) - -# Multiple images as grid -img_grid = make_grid(images[:64], nrow=8) -writer.add_image('Batch/inputs', img_grid, epoch) - -# Predictions visualization -pred_grid = make_grid(predictions[:16], nrow=4) -writer.add_image('Predictions', pred_grid, epoch) -``` - -```python -# TensorFlow -import tensorflow as tf - -with file_writer.as_default(): - # Encode images as PNG - tf.summary.image('Training samples', images, step=epoch, max_outputs=25) -``` - -### 5. Logging Histograms - -```python -# PyTorch: Track weight distributions -for name, param in model.named_parameters(): - writer.add_histogram(name, param, epoch) - - # Track gradients - if param.grad is not None: - writer.add_histogram(f'{name}.grad', param.grad, epoch) - -# Track activations -writer.add_histogram('Activations/relu1', activations, epoch) -``` - -```python -# TensorFlow -with file_writer.as_default(): - tf.summary.histogram('weights/layer1', layer1.kernel, step=epoch) - tf.summary.histogram('activations/relu1', activations, step=epoch) -``` - -### 6. Logging Model Graph - -```python -# PyTorch -import torch - -model = MyModel() -dummy_input = torch.randn(1, 3, 224, 224) - -writer.add_graph(model, dummy_input) -writer.close() -``` - -```python -# TensorFlow (automatic with Keras) -tensorboard_callback = tf.keras.callbacks.TensorBoard( - log_dir='logs', - write_graph=True -) - -model.fit(x, y, callbacks=[tensorboard_callback]) -``` - -## Advanced Features - -### Embedding Projector - -Visualize high-dimensional data (embeddings, features) in 2D/3D. - -```python -import torch -from torch.utils.tensorboard import SummaryWriter - -# Get embeddings (e.g., word embeddings, image features) -embeddings = model.get_embeddings(data) # Shape: (N, embedding_dim) - -# Metadata (labels for each point) -metadata = ['class_1', 'class_2', 'class_1', ...] - -# Images (optional, for image embeddings) -label_images = torch.stack([img1, img2, img3, ...]) - -# Log to TensorBoard -writer.add_embedding( - embeddings, - metadata=metadata, - label_img=label_images, - global_step=epoch -) -``` - -**In TensorBoard:** -- Navigate to "Projector" tab -- Choose PCA, t-SNE, or UMAP visualization -- Search, filter, and explore clusters - -### Hyperparameter Tuning - -```python -from torch.utils.tensorboard import SummaryWriter - -# Try different hyperparameters -for lr in [0.001, 0.01, 0.1]: - for batch_size in [16, 32, 64]: - # Create unique run directory - writer = SummaryWriter(f'runs/lr{lr}_bs{batch_size}') - - # Log hyperparameters - writer.add_hparams( - {'lr': lr, 'batch_size': batch_size}, - {'hparam/accuracy': final_acc, 'hparam/loss': final_loss} - ) - - # Train and log - for epoch in range(10): - loss = train(lr, batch_size) - writer.add_scalar('Loss/train', loss, epoch) - - writer.close() - -# Compare in TensorBoard's "HParams" tab -``` - -### Text Logging - -```python -# PyTorch: Log text (e.g., model predictions, summaries) -writer.add_text('Predictions', f'Epoch {epoch}: {predictions}', epoch) -writer.add_text('Config', str(config), 0) - -# Log markdown tables -markdown_table = """ -| Metric | Value | -|--------|-------| -| Accuracy | 0.95 | -| F1 Score | 0.93 | -""" -writer.add_text('Results', markdown_table, epoch) -``` - -### PR Curves - -Precision-Recall curves for classification. - -```python -from torch.utils.tensorboard import SummaryWriter - -# Get predictions and labels -predictions = model(test_data) # Shape: (N, num_classes) -labels = test_labels # Shape: (N,) - -# Log PR curve for each class -for i in range(num_classes): - writer.add_pr_curve( - f'PR_curve/class_{i}', - labels == i, - predictions[:, i], - global_step=epoch - ) -``` - -## Integration Examples - -### PyTorch Training Loop - -```python -import torch -import torch.nn as nn -from torch.utils.tensorboard import SummaryWriter - -# Setup -writer = SummaryWriter('runs/resnet_experiment') -model = ResNet50() -optimizer = torch.optim.Adam(model.parameters(), lr=0.001) -criterion = nn.CrossEntropyLoss() - -# Log model graph -dummy_input = torch.randn(1, 3, 224, 224) -writer.add_graph(model, dummy_input) - -# Training loop -for epoch in range(50): - model.train() - train_loss = 0.0 - train_correct = 0 - - for batch_idx, (data, target) in enumerate(train_loader): - optimizer.zero_grad() - output = model(data) - loss = criterion(output, target) - loss.backward() - optimizer.step() - - train_loss += loss.item() - pred = output.argmax(dim=1) - train_correct += pred.eq(target).sum().item() - - # Log batch metrics (every 100 batches) - if batch_idx % 100 == 0: - global_step = epoch * len(train_loader) + batch_idx - writer.add_scalar('Loss/train_batch', loss.item(), global_step) - - # Epoch metrics - train_loss /= len(train_loader) - train_acc = train_correct / len(train_loader.dataset) - - # Validation - model.eval() - val_loss = 0.0 - val_correct = 0 - - with torch.no_grad(): - for data, target in val_loader: - output = model(data) - val_loss += criterion(output, target).item() - pred = output.argmax(dim=1) - val_correct += pred.eq(target).sum().item() - - val_loss /= len(val_loader) - val_acc = val_correct / len(val_loader.dataset) - - # Log epoch metrics - writer.add_scalars('Loss', {'train': train_loss, 'val': val_loss}, epoch) - writer.add_scalars('Accuracy', {'train': train_acc, 'val': val_acc}, epoch) - - # Log learning rate - writer.add_scalar('Learning_rate', optimizer.param_groups[0]['lr'], epoch) - - # Log histograms (every 5 epochs) - if epoch % 5 == 0: - for name, param in model.named_parameters(): - writer.add_histogram(name, param, epoch) - - # Log sample predictions - if epoch % 10 == 0: - sample_images = data[:8] - writer.add_image('Sample_inputs', make_grid(sample_images), epoch) - -writer.close() -``` - -### TensorFlow/Keras Training - -```python -import tensorflow as tf - -# Define model -model = tf.keras.models.Sequential([ - tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(28, 28, 1)), - tf.keras.layers.MaxPooling2D(), - tf.keras.layers.Flatten(), - tf.keras.layers.Dense(128, activation='relu'), - tf.keras.layers.Dense(10, activation='softmax') -]) - -model.compile( - optimizer='adam', - loss='sparse_categorical_crossentropy', - metrics=['accuracy'] -) - -# TensorBoard callback -tensorboard_callback = tf.keras.callbacks.TensorBoard( - log_dir='logs/fit', - histogram_freq=1, # Log histograms every epoch - write_graph=True, # Visualize model graph - write_images=True, # Visualize weights as images - update_freq='epoch', # Log metrics every epoch - profile_batch='500,520', # Profile batches 500-520 - embeddings_freq=1 # Log embeddings every epoch -) - -# Train -model.fit( - x_train, y_train, - epochs=10, - validation_data=(x_val, y_val), - callbacks=[tensorboard_callback] -) -``` - -## Comparing Experiments - -### Multiple Runs - -```bash -# Run experiments with different configs -python train.py --lr 0.001 --logdir runs/exp1 -python train.py --lr 0.01 --logdir runs/exp2 -python train.py --lr 0.1 --logdir runs/exp3 - -# View all runs together -tensorboard --logdir=runs -``` - -**In TensorBoard:** -- All runs appear in the same dashboard -- Toggle runs on/off for comparison -- Use regex to filter run names -- Overlay charts to compare metrics - -### Organizing Experiments - -```python -# Hierarchical organization -runs/ -├── baseline/ -│ ├── run_1/ -│ └── run_2/ -├── improved/ -│ ├── run_1/ -│ └── run_2/ -└── final/ - └── run_1/ - -# Log with hierarchy -writer = SummaryWriter('runs/baseline/run_1') -``` - -## Best Practices - -### 1. Use Descriptive Run Names - -```python -# ✅ Good: Descriptive names -from datetime import datetime -timestamp = datetime.now().strftime('%Y%m%d_%H%M%S') -writer = SummaryWriter(f'runs/resnet50_lr0.001_bs32_{timestamp}') - -# ❌ Bad: Auto-generated names -writer = SummaryWriter() # Creates runs/Jan01_12-34-56_hostname -``` - -### 2. Group Related Metrics - -```python -# ✅ Good: Grouped metrics -writer.add_scalar('Loss/train', train_loss, step) -writer.add_scalar('Loss/val', val_loss, step) -writer.add_scalar('Accuracy/train', train_acc, step) -writer.add_scalar('Accuracy/val', val_acc, step) - -# ❌ Bad: Flat namespace -writer.add_scalar('train_loss', train_loss, step) -writer.add_scalar('val_loss', val_loss, step) -``` - -### 3. Log Regularly but Not Too Often - -```python -# ✅ Good: Log epoch metrics always, batch metrics occasionally -for epoch in range(100): - for batch_idx, (data, target) in enumerate(train_loader): - loss = train_step(data, target) - - # Log every 100 batches - if batch_idx % 100 == 0: - writer.add_scalar('Loss/batch', loss, global_step) - - # Always log epoch metrics - writer.add_scalar('Loss/epoch', epoch_loss, epoch) - -# ❌ Bad: Log every batch (creates huge log files) -for batch in train_loader: - writer.add_scalar('Loss', loss, step) # Too frequent -``` - -### 4. Close Writer When Done - -```python -# ✅ Good: Use context manager -with SummaryWriter('runs/exp1') as writer: - for epoch in range(10): - writer.add_scalar('Loss', loss, epoch) -# Automatically closes - -# Or manually -writer = SummaryWriter('runs/exp1') -# ... logging ... -writer.close() -``` - -### 5. Use Separate Writers for Train/Val - -```python -# ✅ Good: Separate log directories -train_writer = SummaryWriter('runs/exp1/train') -val_writer = SummaryWriter('runs/exp1/val') - -train_writer.add_scalar('loss', train_loss, epoch) -val_writer.add_scalar('loss', val_loss, epoch) -``` - -## Performance Profiling - -### TensorFlow Profiler - -```python -# Enable profiling -tensorboard_callback = tf.keras.callbacks.TensorBoard( - log_dir='logs', - profile_batch='10,20' # Profile batches 10-20 -) - -model.fit(x, y, callbacks=[tensorboard_callback]) - -# View in TensorBoard Profile tab -# Shows: GPU utilization, kernel stats, memory usage, bottlenecks -``` - -### PyTorch Profiler - -```python -import torch.profiler as profiler - -with profiler.profile( - activities=[ - profiler.ProfilerActivity.CPU, - profiler.ProfilerActivity.CUDA - ], - on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/profiler'), - record_shapes=True, - with_stack=True -) as prof: - for batch in train_loader: - loss = train_step(batch) - prof.step() - -# View in TensorBoard Profile tab -``` - -## Resources - -- **Documentation**: https://www.tensorflow.org/tensorboard -- **PyTorch Integration**: https://pytorch.org/docs/stable/tensorboard.html -- **GitHub**: https://github.com/tensorflow/tensorboard (27k+ stars) -- **TensorBoard.dev**: https://tensorboard.dev (share experiments publicly) - -## See Also - -- `references/visualization.md` - Comprehensive visualization guide -- `references/profiling.md` - Performance profiling patterns -- `references/integrations.md` - Framework-specific integration examples - - diff --git a/skills/tensorboard/references/integrations.md b/skills/tensorboard/references/integrations.md deleted file mode 100644 index 5865350..0000000 --- a/skills/tensorboard/references/integrations.md +++ /dev/null @@ -1,638 +0,0 @@ -# Framework Integration Guide - -Complete guide to integrating TensorBoard with popular ML frameworks. - -## Table of Contents -- PyTorch -- TensorFlow/Keras -- PyTorch Lightning -- HuggingFace Transformers -- Fast.ai -- JAX -- scikit-learn - -## PyTorch - -### Basic Integration - -```python -import torch -import torch.nn as nn -from torch.utils.tensorboard import SummaryWriter - -# Create writer -writer = SummaryWriter('runs/pytorch_experiment') - -# Model and optimizer -model = ResNet50() -optimizer = torch.optim.Adam(model.parameters(), lr=0.001) -criterion = nn.CrossEntropyLoss() - -# Log model graph -dummy_input = torch.randn(1, 3, 224, 224) -writer.add_graph(model, dummy_input) - -# Training loop -for epoch in range(100): - model.train() - train_loss = 0.0 - - for batch_idx, (data, target) in enumerate(train_loader): - optimizer.zero_grad() - output = model(data) - loss = criterion(output, target) - loss.backward() - optimizer.step() - - train_loss += loss.item() - - # Log batch metrics - if batch_idx % 100 == 0: - global_step = epoch * len(train_loader) + batch_idx - writer.add_scalar('Loss/train_batch', loss.item(), global_step) - - # Epoch metrics - train_loss /= len(train_loader) - writer.add_scalar('Loss/train_epoch', train_loss, epoch) - - # Log histograms - for name, param in model.named_parameters(): - writer.add_histogram(name, param, epoch) - -writer.close() -``` - -### torchvision Integration - -```python -from torchvision.utils import make_grid - -# Log image batch -for batch_idx, (images, labels) in enumerate(train_loader): - if batch_idx == 0: # First batch - img_grid = make_grid(images[:64], nrow=8) - writer.add_image('Training_batch', img_grid, epoch) - break -``` - -### Distributed Training - -```python -import torch.distributed as dist -from torch.nn.parallel import DistributedDataParallel as DDP - -# Setup -dist.init_process_group(backend='nccl') -rank = dist.get_rank() - -# Only log from rank 0 -if rank == 0: - writer = SummaryWriter('runs/distributed_experiment') - -model = DDP(model, device_ids=[rank]) - -for epoch in range(100): - train_loss = train_epoch() - - # Log only from rank 0 - if rank == 0: - writer.add_scalar('Loss/train', train_loss, epoch) -``` - -## TensorFlow/Keras - -### Keras Callback - -```python -import tensorflow as tf - -# TensorBoard callback -tensorboard_callback = tf.keras.callbacks.TensorBoard( - log_dir='logs/keras_experiment', - histogram_freq=1, # Log histograms every epoch - write_graph=True, # Visualize model graph - write_images=True, # Visualize layer weights as images - update_freq='epoch', # Log metrics per epoch (or 'batch', or integer) - profile_batch='10,20', # Profile batches 10-20 - embeddings_freq=1 # Log embeddings every epoch -) - -# Compile model -model.compile( - optimizer='adam', - loss='sparse_categorical_crossentropy', - metrics=['accuracy'] -) - -# Train with callback -history = model.fit( - x_train, y_train, - epochs=10, - validation_data=(x_val, y_val), - callbacks=[tensorboard_callback] -) -``` - -### Custom Training Loop - -```python -import tensorflow as tf - -# Create summary writers -train_summary_writer = tf.summary.create_file_writer('logs/train') -val_summary_writer = tf.summary.create_file_writer('logs/val') - -# Training loop -for epoch in range(100): - # Training - for step, (x_batch, y_batch) in enumerate(train_dataset): - with tf.GradientTape() as tape: - predictions = model(x_batch, training=True) - loss = loss_fn(y_batch, predictions) - - gradients = tape.gradient(loss, model.trainable_variables) - optimizer.apply_gradients(zip(gradients, model.trainable_variables)) - - # Log training metrics - with train_summary_writer.as_default(): - tf.summary.scalar('loss', loss, step=epoch * len(train_dataset) + step) - - # Validation - for x_batch, y_batch in val_dataset: - predictions = model(x_batch, training=False) - val_loss = loss_fn(y_batch, predictions) - val_acc = accuracy_fn(y_batch, predictions) - - # Log validation metrics - with val_summary_writer.as_default(): - tf.summary.scalar('loss', val_loss, step=epoch) - tf.summary.scalar('accuracy', val_acc, step=epoch) - - # Log histograms - with train_summary_writer.as_default(): - for layer in model.layers: - for weight in layer.weights: - tf.summary.histogram(weight.name, weight, step=epoch) -``` - -### tf.data Integration - -```python -# Log dataset samples -for images, labels in train_dataset.take(1): - with file_writer.as_default(): - tf.summary.image('Training samples', images, step=0, max_outputs=25) -``` - -## PyTorch Lightning - -### Built-in Logger - -```python -import pytorch_lightning as pl -from pytorch_lightning.loggers import TensorBoardLogger - -# Create logger -logger = TensorBoardLogger('logs', name='lightning_experiment') - -# Lightning module -class LitModel(pl.LightningModule): - def __init__(self): - super().__init__() - self.model = ResNet50() - - def training_step(self, batch, batch_idx): - x, y = batch - y_hat = self.model(x) - loss = F.cross_entropy(y_hat, y) - - # Log metrics - self.log('train_loss', loss, on_step=True, on_epoch=True) - - return loss - - def validation_step(self, batch, batch_idx): - x, y = batch - y_hat = self.model(x) - loss = F.cross_entropy(y_hat, y) - acc = (y_hat.argmax(dim=1) == y).float().mean() - - # Log metrics - self.log('val_loss', loss, on_epoch=True) - self.log('val_acc', acc, on_epoch=True) - - return loss - - def configure_optimizers(self): - return torch.optim.Adam(self.parameters(), lr=0.001) - -# Trainer -trainer = pl.Trainer( - max_epochs=100, - logger=logger, - log_every_n_steps=50 -) - -# Train -model = LitModel() -trainer.fit(model, train_loader, val_loader) -``` - -### Custom Logging - -```python -class LitModel(pl.LightningModule): - def training_step(self, batch, batch_idx): - x, y = batch - y_hat = self.model(x) - loss = F.cross_entropy(y_hat, y) - - # Log scalar - self.log('train_loss', loss) - - # Log images (every 100 batches) - if batch_idx % 100 == 0: - from torchvision.utils import make_grid - img_grid = make_grid(x[:8]) - self.logger.experiment.add_image('train_images', img_grid, self.global_step) - - # Log histogram - self.logger.experiment.add_histogram('predictions', y_hat, self.global_step) - - return loss -``` - -## HuggingFace Transformers - -### TrainingArguments Integration - -```python -from transformers import Trainer, TrainingArguments - -training_args = TrainingArguments( - output_dir='./results', - num_train_epochs=3, - per_device_train_batch_size=16, - per_device_eval_batch_size=64, - logging_dir='./logs', # TensorBoard log directory - logging_steps=100, # Log every 100 steps - evaluation_strategy='epoch', - save_strategy='epoch', - load_best_model_at_end=True, - report_to='tensorboard' # Enable TensorBoard -) - -trainer = Trainer( - model=model, - args=training_args, - train_dataset=train_dataset, - eval_dataset=eval_dataset, - tokenizer=tokenizer -) - -# Train (automatically logs to TensorBoard) -trainer.train() -``` - -### Custom Metrics - -```python -from transformers import Trainer, TrainingArguments -import numpy as np - -def compute_metrics(eval_pred): - """Custom metrics for evaluation.""" - predictions, labels = eval_pred - predictions = np.argmax(predictions, axis=1) - - accuracy = (predictions == labels).mean() - f1 = f1_score(labels, predictions, average='weighted') - - return { - 'accuracy': accuracy, - 'f1': f1 - } - -trainer = Trainer( - model=model, - args=training_args, - train_dataset=train_dataset, - eval_dataset=eval_dataset, - compute_metrics=compute_metrics # Custom metrics logged to TensorBoard -) -``` - -### Manual Logging - -```python -from transformers import TrainerCallback -from torch.utils.tensorboard import SummaryWriter - -class TensorBoardCallback(TrainerCallback): - """Custom TensorBoard logging.""" - - def __init__(self, log_dir='logs'): - self.writer = SummaryWriter(log_dir) - - def on_log(self, args, state, control, logs=None, **kwargs): - """Called when logging.""" - if logs: - for key, value in logs.items(): - self.writer.add_scalar(key, value, state.global_step) - - def on_train_end(self, args, state, control, **kwargs): - """Close writer.""" - self.writer.close() - -# Use callback -trainer = Trainer( - model=model, - args=training_args, - train_dataset=train_dataset, - callbacks=[TensorBoardCallback()] -) -``` - -## Fast.ai - -### Learner Integration - -```python -from fastai.vision.all import * -from fastai.callback.tensorboard import TensorBoardCallback - -# Create data loaders -dls = ImageDataLoaders.from_folder(path, train='train', valid='valid') - -# Create learner -learn = cnn_learner(dls, resnet50, metrics=accuracy) - -# Train with TensorBoard logging -learn.fit_one_cycle( - 10, - cbs=TensorBoardCallback('logs/fastai', trace_model=True) -) - -# View logs -# tensorboard --logdir=logs/fastai -``` - -### Custom Callbacks - -```python -from fastai.callback.core import Callback -from torch.utils.tensorboard import SummaryWriter - -class CustomTensorBoardCallback(Callback): - """Custom TensorBoard callback.""" - - def __init__(self, log_dir='logs'): - self.writer = SummaryWriter(log_dir) - - def after_batch(self): - """Log after each batch.""" - if self.train_iter % 100 == 0: - self.writer.add_scalar('Loss/train', self.loss, self.train_iter) - - def after_epoch(self): - """Log after each epoch.""" - self.writer.add_scalar('Loss/train_epoch', self.recorder.train_loss, self.epoch) - self.writer.add_scalar('Loss/val_epoch', self.recorder.valid_loss, self.epoch) - - # Log metrics - for i, metric in enumerate(self.recorder.metrics): - metric_name = self.recorder.metric_names[i+1] - self.writer.add_scalar(f'Metrics/{metric_name}', metric, self.epoch) - -# Use callback -learn.fit_one_cycle(10, cbs=[CustomTensorBoardCallback()]) -``` - -## JAX - -### Basic Integration - -```python -import jax -import jax.numpy as jnp -from torch.utils.tensorboard import SummaryWriter - -writer = SummaryWriter('logs/jax_experiment') - -# Training loop -for epoch in range(100): - for batch in train_batches: - # JAX training step - state, loss = train_step(state, batch) - - # Log to TensorBoard (convert JAX array to numpy) - writer.add_scalar('Loss/train', float(loss), epoch) - - # Validation - val_loss = evaluate(state, val_batches) - writer.add_scalar('Loss/val', float(val_loss), epoch) - -writer.close() -``` - -### Flax Integration - -```python -from flax.training import train_state -import optax -from torch.utils.tensorboard import SummaryWriter - -writer = SummaryWriter('logs/flax_experiment') - -# Create train state -state = train_state.TrainState.create( - apply_fn=model.apply, - params=params, - tx=optax.adam(0.001) -) - -# Training loop -for epoch in range(100): - for batch in train_loader: - state, loss = train_step(state, batch) - - # Log metrics - writer.add_scalar('Loss/train', loss.item(), epoch) - - # Log parameters - for name, param in state.params.items(): - writer.add_histogram(f'Params/{name}', jnp.array(param), epoch) - -writer.close() -``` - -## scikit-learn - -### Manual Logging - -```python -from sklearn.ensemble import RandomForestClassifier -from sklearn.model_selection import cross_val_score -from torch.utils.tensorboard import SummaryWriter - -writer = SummaryWriter('logs/sklearn_experiment') - -# Hyperparameter search -for n_estimators in [10, 50, 100, 200]: - for max_depth in [3, 5, 10, None]: - # Train model - model = RandomForestClassifier( - n_estimators=n_estimators, - max_depth=max_depth, - random_state=42 - ) - - # Cross-validation - scores = cross_val_score(model, X_train, y_train, cv=5) - - # Log results - run_name = f'n{n_estimators}_d{max_depth}' - writer.add_scalar(f'{run_name}/cv_mean', scores.mean(), 0) - writer.add_scalar(f'{run_name}/cv_std', scores.std(), 0) - - # Log hyperparameters - writer.add_hparams( - {'n_estimators': n_estimators, 'max_depth': max_depth or -1}, - {'cv_accuracy': scores.mean()} - ) - -writer.close() -``` - -### GridSearchCV Logging - -```python -from sklearn.model_selection import GridSearchCV -from torch.utils.tensorboard import SummaryWriter - -writer = SummaryWriter('logs/gridsearch') - -# Grid search -param_grid = { - 'n_estimators': [10, 50, 100], - 'max_depth': [3, 5, 10] -} - -grid_search = GridSearchCV( - RandomForestClassifier(), - param_grid, - cv=5, - return_train_score=True -) - -grid_search.fit(X_train, y_train) - -# Log all results -for i, params in enumerate(grid_search.cv_results_['params']): - mean_train_score = grid_search.cv_results_['mean_train_score'][i] - mean_test_score = grid_search.cv_results_['mean_test_score'][i] - - param_str = '_'.join([f'{k}{v}' for k, v in params.items()]) - - writer.add_scalar(f'{param_str}/train', mean_train_score, 0) - writer.add_scalar(f'{param_str}/test', mean_test_score, 0) - -# Log best params -writer.add_text('Best_params', str(grid_search.best_params_), 0) -writer.add_scalar('Best_score', grid_search.best_score_, 0) - -writer.close() -``` - -## Best Practices - -### 1. Consistent Naming Conventions - -```python -# ✅ Good: Hierarchical names across frameworks -writer.add_scalar('Loss/train', train_loss, step) -writer.add_scalar('Loss/val', val_loss, step) -writer.add_scalar('Metrics/accuracy', accuracy, step) - -# Works the same in PyTorch, TensorFlow, Lightning -``` - -### 2. Use Framework-Specific Features - -```python -# PyTorch: Use SummaryWriter -from torch.utils.tensorboard import SummaryWriter - -# TensorFlow: Use tf.summary -import tensorflow as tf -tf.summary.scalar('loss', loss, step=step) - -# Lightning: Use self.log() -self.log('train_loss', loss) - -# Transformers: Use report_to='tensorboard' -training_args = TrainingArguments(report_to='tensorboard') -``` - -### 3. Centralize Logging Logic - -```python -class MetricLogger: - """Universal metric logger.""" - - def __init__(self, log_dir='logs'): - self.writer = SummaryWriter(log_dir) - - def log_scalar(self, name, value, step): - self.writer.add_scalar(name, value, step) - - def log_image(self, name, image, step): - self.writer.add_image(name, image, step) - - def log_histogram(self, name, values, step): - self.writer.add_histogram(name, values, step) - - def close(self): - self.writer.close() - -# Use across frameworks -logger = MetricLogger('logs/universal') -logger.log_scalar('Loss/train', train_loss, epoch) -``` - -### 4. Framework Detection - -```python -def get_tensorboard_writer(framework='auto', log_dir='logs'): - """Get TensorBoard writer for any framework.""" - if framework == 'auto': - # Auto-detect framework - try: - import torch - framework = 'pytorch' - except ImportError: - try: - import tensorflow as tf - framework = 'tensorflow' - except ImportError: - raise ValueError("No supported framework found") - - if framework == 'pytorch': - from torch.utils.tensorboard import SummaryWriter - return SummaryWriter(log_dir) - - elif framework == 'tensorflow': - import tensorflow as tf - return tf.summary.create_file_writer(log_dir) - -# Use it -writer = get_tensorboard_writer(log_dir='logs/auto') -``` - -## Resources - -- **PyTorch**: https://pytorch.org/docs/stable/tensorboard.html -- **TensorFlow**: https://www.tensorflow.org/tensorboard -- **Lightning**: https://pytorch-lightning.readthedocs.io/en/stable/extensions/logging.html -- **Transformers**: https://huggingface.co/docs/transformers/main_classes/trainer -- **Fast.ai**: https://docs.fast.ai/callback.tensorboard.html diff --git a/skills/tensorboard/references/profiling.md b/skills/tensorboard/references/profiling.md deleted file mode 100644 index 5b1da6b..0000000 --- a/skills/tensorboard/references/profiling.md +++ /dev/null @@ -1,545 +0,0 @@ -# Performance Profiling Guide - -Complete guide to profiling and optimizing ML models with TensorBoard. - -## Table of Contents -- PyTorch Profiler -- TensorFlow Profiler -- GPU Utilization -- Memory Profiling -- Bottleneck Detection -- Optimization Strategies - -## PyTorch Profiler - -### Basic Profiling - -```python -import torch -import torch.profiler as profiler - -model = MyModel().cuda() -optimizer = torch.optim.Adam(model.parameters()) - -# Profile training loop -with profiler.profile( - activities=[ - profiler.ProfilerActivity.CPU, - profiler.ProfilerActivity.CUDA, - ], - on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/profiler'), - record_shapes=True, - with_stack=True -) as prof: - for step, (data, target) in enumerate(train_loader): - optimizer.zero_grad() - output = model(data.cuda()) - loss = F.cross_entropy(output, target.cuda()) - loss.backward() - optimizer.step() - - # Mark step for profiler - prof.step() - - if step >= 10: # Profile first 10 steps - break -``` - -### Profiler Configuration - -```python -with profiler.profile( - activities=[ - profiler.ProfilerActivity.CPU, # Profile CPU ops - profiler.ProfilerActivity.CUDA, # Profile GPU ops - ], - schedule=profiler.schedule( - wait=1, # Warmup steps (skip profiling) - warmup=1, # Steps to warmup profiler - active=3, # Steps to actively profile - repeat=2 # Repeat cycle 2 times - ), - on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/profiler'), - record_shapes=True, # Record tensor shapes - profile_memory=True, # Track memory allocation - with_stack=True, # Record source code stack traces - with_flops=True # Estimate FLOPS -) as prof: - for step, batch in enumerate(train_loader): - train_step(batch) - prof.step() -``` - -### Profile Inference - -```python -model.eval() - -with profiler.profile( - activities=[profiler.ProfilerActivity.CPU, profiler.ProfilerActivity.CUDA], - on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/inference_profiler') -) as prof: - with torch.no_grad(): - for i in range(100): - data = torch.randn(1, 3, 224, 224).cuda() - output = model(data) - prof.step() -``` - -### Analyze Profile Data - -```python -# Print profiler summary -print(prof.key_averages().table(sort_by="cuda_time_total", row_limit=10)) - -# Export Chrome trace (for chrome://tracing) -prof.export_chrome_trace("trace.json") - -# View in TensorBoard -# tensorboard --logdir=runs/profiler -``` - -**TensorBoard Profile Tab shows:** -- Overview: GPU utilization, step time breakdown -- Operator view: Time spent in each operation -- Kernel view: GPU kernel execution -- Trace view: Timeline of operations -- Memory view: Memory allocation over time - -## TensorFlow Profiler - -### Profile with Callback - -```python -import tensorflow as tf - -# Create profiler callback -tensorboard_callback = tf.keras.callbacks.TensorBoard( - log_dir='logs/profiler', - profile_batch='10,20' # Profile batches 10-20 -) - -# Train with profiling -model.fit( - x_train, y_train, - epochs=5, - callbacks=[tensorboard_callback] -) - -# Launch TensorBoard -# tensorboard --logdir=logs/profiler -``` - -### Programmatic Profiling - -```python -import tensorflow as tf - -# Start profiler -tf.profiler.experimental.start('logs/profiler') - -# Training code -for epoch in range(5): - for step, (x, y) in enumerate(train_dataset): - with tf.GradientTape() as tape: - predictions = model(x, training=True) - loss = loss_fn(y, predictions) - - gradients = tape.gradient(loss, model.trainable_variables) - optimizer.apply_gradients(zip(gradients, model.trainable_variables)) - - # Profile specific steps - if epoch == 2 and step == 10: - tf.profiler.experimental.start('logs/profiler_step10') - - if epoch == 2 and step == 20: - tf.profiler.experimental.stop() - -# Stop profiler -tf.profiler.experimental.stop() -``` - -### Profile Custom Training Loop - -```python -# Profile with context manager -with tf.profiler.experimental.Profile('logs/profiler'): - for epoch in range(3): - for step, (x, y) in enumerate(train_dataset): - train_step(x, y) -``` - -## GPU Utilization - -### Monitor GPU Usage - -```python -import torch -import torch.profiler as profiler - -with profiler.profile( - activities=[profiler.ProfilerActivity.CUDA], - on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/gpu_profile'), - with_stack=True -) as prof: - for step, batch in enumerate(train_loader): - # Your training step - output = model(batch.cuda()) - loss = criterion(output, target.cuda()) - loss.backward() - optimizer.step() - - prof.step() - -# View in TensorBoard > Profile > Overview -# Shows: GPU utilization %, kernel efficiency, memory bandwidth -``` - -### Optimize GPU Utilization - -```python -# ✅ Good: Keep GPU busy -def train_step(batch): - # Overlap data transfer with computation - data = batch.cuda(non_blocking=True) # Async transfer - - # Mixed precision for faster computation - with torch.cuda.amp.autocast(): - output = model(data) - loss = criterion(output, target) - - return loss - -# ❌ Bad: GPU idle during data transfer -def train_step_slow(batch): - data = batch.cuda() # Blocking transfer - output = model(data) - return loss -``` - -### Reduce CPU-GPU Synchronization - -```python -# ✅ Good: Minimize synchronization -for epoch in range(100): - for batch in train_loader: - loss = train_step(batch) - - # Accumulate losses (no sync) - total_loss += loss.item() - - # Synchronize once per epoch - avg_loss = total_loss / len(train_loader) - -# ❌ Bad: Frequent synchronization -for batch in train_loader: - loss = train_step(batch) - print(f"Loss: {loss.item()}") # Syncs every batch! -``` - -## Memory Profiling - -### Track Memory Allocation - -```python -import torch -import torch.profiler as profiler - -with profiler.profile( - activities=[profiler.ProfilerActivity.CUDA], - profile_memory=True, - record_shapes=True, - on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/memory_profile') -) as prof: - for step, batch in enumerate(train_loader): - train_step(batch) - prof.step() - -# View in TensorBoard > Profile > Memory View -# Shows: Memory allocation over time, peak memory, allocation stack traces -``` - -### Find Memory Leaks - -```python -import torch - -# Record memory snapshots -torch.cuda.memory._record_memory_history( - enabled=True, - max_entries=100000 -) - -# Training -for batch in train_loader: - train_step(batch) - -# Save memory snapshot -snapshot = torch.cuda.memory._snapshot() -torch.cuda.memory._dump_snapshot("memory_snapshot.pickle") - -# Analyze with: -# python -m torch.cuda.memory_viz trace_plot memory_snapshot.pickle -o memory_trace.html -``` - -### Optimize Memory Usage - -```python -# ✅ Good: Gradient accumulation for large batches -accumulation_steps = 4 - -for i, batch in enumerate(train_loader): - # Forward - output = model(batch) - loss = criterion(output, target) / accumulation_steps - - # Backward - loss.backward() - - # Step optimizer every accumulation_steps - if (i + 1) % accumulation_steps == 0: - optimizer.step() - optimizer.zero_grad() - -# ✅ Good: Release memory explicitly -del intermediate_tensor -torch.cuda.empty_cache() - -# ✅ Good: Use gradient checkpointing -from torch.utils.checkpoint import checkpoint - -def custom_forward(module, input): - return checkpoint(module, input) -``` - -## Bottleneck Detection - -### Identify Slow Operations - -```python -with profiler.profile( - activities=[profiler.ProfilerActivity.CPU, profiler.ProfilerActivity.CUDA], - on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/bottleneck_profile'), - with_stack=True -) as prof: - for step, batch in enumerate(train_loader): - train_step(batch) - prof.step() - -# Print slowest operations -print(prof.key_averages().table( - sort_by="cuda_time_total", - row_limit=20 -)) - -# Expected output: -# Name | CPU time | CUDA time | Calls -# aten::conv2d | 5.2 ms | 45.3 ms | 32 -# aten::batch_norm | 1.1 ms | 8.7 ms | 32 -# aten::relu | 0.3 ms | 2.1 ms | 32 -``` - -### Optimize Data Loading - -```python -# ✅ Good: Efficient data loading -train_loader = torch.utils.data.DataLoader( - dataset, - batch_size=32, - num_workers=4, # Parallel data loading - pin_memory=True, # Faster GPU transfer - prefetch_factor=2, # Prefetch batches - persistent_workers=True # Reuse workers -) - -# Profile data loading -import time - -start = time.time() -for batch in train_loader: - pass -print(f"Data loading time: {time.time() - start:.2f}s") - -# ❌ Bad: Single worker, no pinning -train_loader = torch.utils.data.DataLoader( - dataset, - batch_size=32, - num_workers=0 # Slow! -) -``` - -### Profile Specific Operations - -```python -# Context manager for specific code blocks -with profiler.record_function("data_preprocessing"): - data = preprocess(batch) - -with profiler.record_function("forward_pass"): - output = model(data) - -with profiler.record_function("loss_computation"): - loss = criterion(output, target) - -# View in TensorBoard > Profile > Trace View -``` - -## Optimization Strategies - -### Mixed Precision Training - -```python -import torch -from torch.cuda.amp import autocast, GradScaler - -scaler = GradScaler() - -for batch in train_loader: - optimizer.zero_grad() - - # Mixed precision forward pass - with autocast(): - output = model(batch.cuda()) - loss = criterion(output, target.cuda()) - - # Scaled backward pass - scaler.scale(loss).backward() - scaler.step(optimizer) - scaler.update() - -# Profile to verify speedup -with profiler.profile( - activities=[profiler.ProfilerActivity.CUDA], - on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/mixed_precision') -) as prof: - train_with_mixed_precision() - prof.step() -``` - -### Kernel Fusion - -```python -# ✅ Good: Fused operations -# torch.nn.functional.gelu() is fused -output = F.gelu(x) - -# ❌ Bad: Separate operations -# Manual GELU (slower due to multiple kernels) -output = 0.5 * x * (1 + torch.tanh(math.sqrt(2 / math.pi) * (x + 0.044715 * x**3))) - -# Use torch.jit to fuse custom operations -@torch.jit.script -def fused_gelu(x): - return 0.5 * x * (1 + torch.tanh(math.sqrt(2 / math.pi) * (x + 0.044715 * x**3))) -``` - -### Reduce Host-Device Transfers - -```python -# ✅ Good: Keep data on GPU -data = data.cuda() # Transfer once -for epoch in range(100): - output = model(data) # No transfer - loss = criterion(output, target) - -# ❌ Bad: Frequent transfers -for epoch in range(100): - output = model(data.cuda()) # Transfer every epoch! - loss = criterion(output.cpu(), target.cpu()) # Transfer back! -``` - -### Batch Size Optimization - -```python -# Find optimal batch size with profiling -for batch_size in [16, 32, 64, 128, 256]: - train_loader = DataLoader(dataset, batch_size=batch_size) - - with profiler.profile( - activities=[profiler.ProfilerActivity.CUDA], - profile_memory=True, - on_trace_ready=torch.profiler.tensorboard_trace_handler(f'./runs/bs{batch_size}') - ) as prof: - for step, batch in enumerate(train_loader): - train_step(batch) - prof.step() - - if step >= 10: - break - -# Compare in TensorBoard: -# - GPU utilization -# - Memory usage -# - Throughput (samples/sec) -``` - -## Best Practices - -### 1. Profile Representative Workloads - -```python -# ✅ Good: Profile realistic training scenario -with profiler.profile(...) as prof: - for epoch in range(3): # Profile multiple epochs - for step, batch in enumerate(train_loader): - train_step(batch) - prof.step() - -# ❌ Bad: Profile single step -with profiler.profile(...) as prof: - train_step(single_batch) -``` - -### 2. Profile Periodically - -```python -# Profile every N epochs -if epoch % 10 == 0: - with profiler.profile( - activities=[profiler.ProfilerActivity.CUDA], - on_trace_ready=torch.profiler.tensorboard_trace_handler(f'./runs/epoch{epoch}') - ) as prof: - train_epoch() -``` - -### 3. Compare Before/After Optimizations - -```python -# Baseline -with profiler.profile(...) as prof: - baseline_train() - prof.step() - -# After optimization -with profiler.profile(...) as prof: - optimized_train() - prof.step() - -# Compare in TensorBoard -``` - -### 4. Profile Inference - -```python -# Production inference profiling -model.eval() - -with profiler.profile( - activities=[profiler.ProfilerActivity.CUDA], - on_trace_ready=torch.profiler.tensorboard_trace_handler('./runs/inference') -) as prof: - with torch.no_grad(): - for i in range(1000): # Realistic load - data = get_production_request() - output = model(data) - prof.step() - -# Analyze latency percentiles in TensorBoard -``` - -## Resources - -- **PyTorch Profiler**: https://pytorch.org/tutorials/recipes/recipes/profiler_recipe.html -- **TensorFlow Profiler**: https://www.tensorflow.org/guide/profiler -- **NVIDIA Nsight**: https://developer.nvidia.com/nsight-systems -- **PyTorch Bottleneck**: https://pytorch.org/docs/stable/bottleneck.html diff --git a/skills/tensorboard/references/visualization.md b/skills/tensorboard/references/visualization.md deleted file mode 100644 index be9e7d3..0000000 --- a/skills/tensorboard/references/visualization.md +++ /dev/null @@ -1,620 +0,0 @@ -# Comprehensive Visualization Guide - -Complete guide to visualizing ML experiments with TensorBoard. - -## Table of Contents -- Scalars -- Images -- Histograms & Distributions -- Graphs -- Embeddings -- Text -- PR Curves -- Custom Visualizations - -## Scalars - -### Basic Scalar Logging - -```python -from torch.utils.tensorboard import SummaryWriter - -writer = SummaryWriter('runs/scalars_demo') - -# Log single metric -for step in range(100): - loss = compute_loss() - writer.add_scalar('Loss', loss, step) - -writer.close() -``` - -### Multiple Scalars - -```python -# Group related metrics -writer.add_scalars('Loss', { - 'train': train_loss, - 'validation': val_loss, - 'test': test_loss -}, epoch) - -writer.add_scalars('Metrics/Classification', { - 'accuracy': accuracy, - 'precision': precision, - 'recall': recall, - 'f1_score': f1 -}, epoch) -``` - -### Time-Series Metrics - -```python -# Track metrics over training -for epoch in range(100): - # Training metrics - train_loss = 0.0 - for batch in train_loader: - loss = train_batch(batch) - train_loss += loss - - train_loss /= len(train_loader) - - # Validation metrics - val_loss, val_acc = validate() - - # Log - writer.add_scalar('Loss/train', train_loss, epoch) - writer.add_scalar('Loss/val', val_loss, epoch) - writer.add_scalar('Accuracy/val', val_acc, epoch) - - # Log learning rate - current_lr = optimizer.param_groups[0]['lr'] - writer.add_scalar('Learning_rate', current_lr, epoch) -``` - -### Custom Smoothing - -TensorBoard UI allows smoothing scalars: -- Slider from 0 (no smoothing) to 1 (maximum smoothing) -- Exponential moving average -- Useful for noisy metrics - -## Images - -### Single Image - -```python -import torch -from torch.utils.tensorboard import SummaryWriter - -writer = SummaryWriter('runs/images_demo') - -# Log single image (C, H, W) -img = torch.rand(3, 224, 224) -writer.add_image('Sample_image', img, 0) -``` - -### Image Grid - -```python -from torchvision.utils import make_grid - -# Create grid from batch -images = torch.rand(64, 3, 224, 224) # Batch of 64 images -img_grid = make_grid(images, nrow=8) # 8 images per row - -writer.add_image('Image_grid', img_grid, epoch) -``` - -### Training Visualizations - -```python -# Visualize inputs, predictions, and ground truth -for epoch in range(10): - # Get batch - images, labels = next(iter(val_loader)) - - # Predict - with torch.no_grad(): - predictions = model(images) - - # Visualize inputs - input_grid = make_grid(images[:16], nrow=4) - writer.add_image('Inputs', input_grid, epoch) - - # Visualize predictions (if images) - if isinstance(predictions, torch.Tensor) and predictions.dim() == 4: - pred_grid = make_grid(predictions[:16], nrow=4) - writer.add_image('Predictions', pred_grid, epoch) -``` - -### Attention Maps - -```python -# Visualize attention weights -attention_maps = model.get_attention(images) # (B, H, W) - -# Normalize to [0, 1] -attention_maps = (attention_maps - attention_maps.min()) / (attention_maps.max() - attention_maps.min()) - -# Add channel dimension -attention_maps = attention_maps.unsqueeze(1) # (B, 1, H, W) - -# Create grid -attention_grid = make_grid(attention_maps[:16], nrow=4) -writer.add_image('Attention_maps', attention_grid, epoch) -``` - -### TensorFlow Images - -```python -import tensorflow as tf - -file_writer = tf.summary.create_file_writer('logs/images') - -with file_writer.as_default(): - # Log image batch - tf.summary.image('Training_samples', images, step=epoch, max_outputs=25) - - # Log single image - tf.summary.image('Sample', img[tf.newaxis, ...], step=epoch) -``` - -## Histograms & Distributions - -### Weight Histograms - -```python -# PyTorch: Track weight distributions over time -for epoch in range(100): - train_epoch() - - # Log all model parameters - for name, param in model.named_parameters(): - writer.add_histogram(f'Weights/{name}', param, epoch) - - # Log gradients - for name, param in model.named_parameters(): - if param.grad is not None: - writer.add_histogram(f'Gradients/{name}', param.grad, epoch) -``` - -### Activation Histograms - -```python -# Hook to capture activations -activations = {} - -def get_activation(name): - def hook(model, input, output): - activations[name] = output.detach() - return hook - -# Register hooks -model.conv1.register_forward_hook(get_activation('conv1')) -model.conv2.register_forward_hook(get_activation('conv2')) -model.fc.register_forward_hook(get_activation('fc')) - -# Forward pass -output = model(input) - -# Log activations -for name, activation in activations.items(): - writer.add_histogram(f'Activations/{name}', activation, epoch) -``` - -### Custom Distributions - -```python -# Log prediction distributions -predictions = model(test_data) -writer.add_histogram('Predictions', predictions, epoch) - -# Log loss distributions across batches -losses = [] -for batch in val_loader: - loss = compute_loss(batch) - losses.append(loss) - -losses = torch.tensor(losses) -writer.add_histogram('Loss_distribution', losses, epoch) -``` - -### TensorFlow Histograms - -```python -import tensorflow as tf - -file_writer = tf.summary.create_file_writer('logs/histograms') - -with file_writer.as_default(): - # Log weight distributions - for layer in model.layers: - for weight in layer.weights: - tf.summary.histogram(weight.name, weight, step=epoch) -``` - -## Graphs - -### Model Architecture - -```python -import torch -from torch.utils.tensorboard import SummaryWriter - -# PyTorch model -model = ResNet50(num_classes=1000) - -# Create dummy input (same shape as real input) -dummy_input = torch.randn(1, 3, 224, 224) - -# Log graph -writer = SummaryWriter('runs/graph_demo') -writer.add_graph(model, dummy_input) -writer.close() - -# View in TensorBoard "Graphs" tab -``` - -### TensorFlow Graph - -```python -# TensorFlow automatically logs graph with Keras -tensorboard_callback = tf.keras.callbacks.TensorBoard( - log_dir='logs', - write_graph=True # Enable graph logging -) - -model.fit(x, y, callbacks=[tensorboard_callback]) -``` - -## Embeddings - -### Projecting Embeddings - -```python -import torch -from torch.utils.tensorboard import SummaryWriter - -writer = SummaryWriter('runs/embeddings_demo') - -# Get embeddings (e.g., word embeddings, image features) -# Shape: (num_samples, embedding_dim) -embeddings = model.get_embeddings(data) - -# Metadata (labels for each embedding) -metadata = ['cat', 'dog', 'bird', 'cat', 'dog', ...] - -# Optional: Images for each embedding -label_img = torch.stack([img1, img2, img3, ...]) # (num_samples, C, H, W) - -# Log embeddings -writer.add_embedding( - embeddings, - metadata=metadata, - label_img=label_img, - global_step=epoch, - tag='Word_embeddings' -) - -writer.close() -``` - -**In TensorBoard Projector:** -- Choose PCA, t-SNE, or UMAP -- Color by metadata labels -- Search and filter points -- Explore nearest neighbors - -### Image Embeddings - -```python -# Extract features from CNN -features = [] -labels = [] -images = [] - -model.eval() -with torch.no_grad(): - for data, target in test_loader: - # Get features from penultimate layer - feature = model.get_features(data) # (B, feature_dim) - features.append(feature) - labels.extend(target.cpu().numpy()) - images.append(data) - -# Concatenate -features = torch.cat(features) -images = torch.cat(images) - -# Metadata (class names) -class_names = ['airplane', 'car', 'bird', 'cat', 'deer', 'dog', 'frog', 'horse', 'ship', 'truck'] -metadata = [class_names[label] for label in labels] - -# Log to TensorBoard -writer.add_embedding( - features, - metadata=metadata, - label_img=images, - tag='CIFAR10_features' -) -``` - -### Text Embeddings - -```python -# Word2Vec or BERT embeddings -word_embeddings = model.word_embeddings.weight.data # (vocab_size, embedding_dim) -vocabulary = ['the', 'cat', 'dog', 'run', 'jump', ...] - -writer.add_embedding( - word_embeddings, - metadata=vocabulary, - tag='Word2Vec_embeddings' -) -``` - -## Text - -### Basic Text Logging - -```python -from torch.utils.tensorboard import SummaryWriter - -writer = SummaryWriter('runs/text_demo') - -# Log plain text -writer.add_text('Config', str(config), 0) -writer.add_text('Hyperparameters', f'lr={lr}, batch_size={batch_size}', 0) - -# Log predictions -predictions_text = f"Epoch {epoch}:\n" -for i, pred in enumerate(predictions[:5]): - predictions_text += f"Sample {i}: {pred}\n" - -writer.add_text('Predictions', predictions_text, epoch) -``` - -### Markdown Tables - -```python -# Log results as markdown table -results = f""" -| Metric | Train | Validation | Test | -|--------|-------|------------|------| -| Accuracy | {train_acc:.4f} | {val_acc:.4f} | {test_acc:.4f} | -| Loss | {train_loss:.4f} | {val_loss:.4f} | {test_loss:.4f} | -| F1 Score | {train_f1:.4f} | {val_f1:.4f} | {test_f1:.4f} | -""" - -writer.add_text('Results/Summary', results, epoch) -``` - -### Model Summaries - -```python -# Log model architecture as text -from torchinfo import summary - -model_summary = str(summary(model, input_size=(1, 3, 224, 224), verbose=0)) -writer.add_text('Model/Architecture', f'```\n{model_summary}\n```', 0) -``` - -## PR Curves - -### Precision-Recall Curves - -```python -from torch.utils.tensorboard import SummaryWriter -from sklearn.metrics import precision_recall_curve - -writer = SummaryWriter('runs/pr_curves') - -# Get predictions and ground truth -y_true = [] -y_scores = [] - -model.eval() -with torch.no_grad(): - for data, target in test_loader: - output = model(data) - probs = torch.softmax(output, dim=1) - - y_true.extend(target.cpu().numpy()) - y_scores.extend(probs.cpu().numpy()) - -y_true = np.array(y_true) -y_scores = np.array(y_scores) - -# Log PR curve for each class -num_classes = y_scores.shape[1] -for class_idx in range(num_classes): - # Binary classification: class vs rest - labels = (y_true == class_idx).astype(int) - scores = y_scores[:, class_idx] - - # Add PR curve - writer.add_pr_curve( - f'PR_curve/class_{class_idx}', - labels, - scores, - global_step=epoch - ) - -writer.close() -``` - -### ROC Curves - -```python -# TensorBoard doesn't have built-in ROC, but we can log as image -from sklearn.metrics import roc_curve, auc -import matplotlib.pyplot as plt - -fig, ax = plt.subplots() - -for class_idx in range(num_classes): - labels = (y_true == class_idx).astype(int) - scores = y_scores[:, class_idx] - - fpr, tpr, _ = roc_curve(labels, scores) - roc_auc = auc(fpr, tpr) - - ax.plot(fpr, tpr, label=f'Class {class_idx} (AUC = {roc_auc:.2f})') - -ax.plot([0, 1], [0, 1], 'k--') -ax.set_xlabel('False Positive Rate') -ax.set_ylabel('True Positive Rate') -ax.set_title('ROC Curves') -ax.legend() - -# Convert to tensor and log -fig.canvas.draw() -img = np.frombuffer(fig.canvas.tostring_rgb(), dtype=np.uint8) -img = img.reshape(fig.canvas.get_width_height()[::-1] + (3,)) -img = torch.from_numpy(img).permute(2, 0, 1) - -writer.add_image('ROC_curves', img, epoch) -plt.close(fig) -``` - -## Custom Visualizations - -### Confusion Matrix - -```python -import matplotlib.pyplot as plt -import seaborn as sns -from sklearn.metrics import confusion_matrix - -# Compute confusion matrix -cm = confusion_matrix(y_true, y_pred) - -# Plot -fig, ax = plt.subplots(figsize=(10, 10)) -sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', ax=ax) -ax.set_xlabel('Predicted') -ax.set_ylabel('True') -ax.set_title('Confusion Matrix') - -# Convert to tensor and log -fig.canvas.draw() -img = np.frombuffer(fig.canvas.tostring_rgb(), dtype=np.uint8) -img = img.reshape(fig.canvas.get_width_height()[::-1] + (3,)) -img = torch.from_numpy(img).permute(2, 0, 1) - -writer.add_image('Confusion_matrix', img, epoch) -plt.close(fig) -``` - -### Loss Landscape - -```python -# Visualize loss surface around current parameters -import numpy as np - -def compute_loss_landscape(model, data, target, param1, param2): - """Compute loss for a grid of parameter values.""" - # Save original params - original_params = {name: param.clone() for name, param in model.named_parameters()} - - # Grid - param1_range = np.linspace(-1, 1, 50) - param2_range = np.linspace(-1, 1, 50) - losses = np.zeros((50, 50)) - - for i, p1 in enumerate(param1_range): - for j, p2 in enumerate(param2_range): - # Perturb parameters - model.state_dict()[param1].add_(p1) - model.state_dict()[param2].add_(p2) - - # Compute loss - with torch.no_grad(): - output = model(data) - loss = F.cross_entropy(output, target) - losses[i, j] = loss.item() - - # Restore parameters - model.load_state_dict(original_params) - - return losses - -# Plot -fig = plt.figure() -ax = fig.add_subplot(111, projection='3d') -X, Y = np.meshgrid(np.linspace(-1, 1, 50), np.linspace(-1, 1, 50)) -ax.plot_surface(X, Y, losses, cmap='viridis') -ax.set_title('Loss Landscape') - -# Log -fig.canvas.draw() -img = np.frombuffer(fig.canvas.tostring_rgb(), dtype=np.uint8) -img = img.reshape(fig.canvas.get_width_height()[::-1] + (3,)) -img = torch.from_numpy(img).permute(2, 0, 1) -writer.add_image('Loss_landscape', img, epoch) -plt.close(fig) -``` - -## Best Practices - -### 1. Use Hierarchical Tags - -```python -# ✅ Good: Organized with hierarchy -writer.add_scalar('Loss/train', train_loss, step) -writer.add_scalar('Loss/val', val_loss, step) -writer.add_scalar('Metrics/accuracy', accuracy, step) -writer.add_scalar('Metrics/f1_score', f1, step) - -# ❌ Bad: Flat namespace -writer.add_scalar('train_loss', train_loss, step) -writer.add_scalar('val_loss', val_loss, step) -``` - -### 2. Log Regularly but Not Excessively - -```python -# ✅ Good: Epoch-level + periodic batch-level -for epoch in range(100): - for batch_idx, batch in enumerate(train_loader): - loss = train_step(batch) - - # Log every 100 batches - if batch_idx % 100 == 0: - global_step = epoch * len(train_loader) + batch_idx - writer.add_scalar('Loss/train_batch', loss, global_step) - - # Always log epoch metrics - writer.add_scalar('Loss/train_epoch', epoch_loss, epoch) - -# ❌ Bad: Every batch (creates huge logs) -for batch in train_loader: - writer.add_scalar('Loss', loss, step) -``` - -### 3. Visualize Sample Predictions - -```python -# Log predictions periodically -if epoch % 5 == 0: - model.eval() - with torch.no_grad(): - sample_images, sample_labels = next(iter(val_loader)) - predictions = model(sample_images) - - # Visualize - img_grid = make_grid(sample_images[:16], nrow=4) - writer.add_image('Samples/inputs', img_grid, epoch) - - # Add predictions as text - pred_text = '\n'.join([f'{i}: {pred.argmax()}' for i, pred in enumerate(predictions[:16])]) - writer.add_text('Samples/predictions', pred_text, epoch) -``` - -## Resources - -- **TensorBoard Documentation**: https://www.tensorflow.org/tensorboard -- **PyTorch TensorBoard**: https://pytorch.org/docs/stable/tensorboard.html -- **Projector Guide**: https://www.tensorflow.org/tensorboard/tensorboard_projector_plugin diff --git a/skills/vllm/SKILL.md b/skills/vllm/SKILL.md deleted file mode 100644 index 164a99b..0000000 --- a/skills/vllm/SKILL.md +++ /dev/null @@ -1,364 +0,0 @@ ---- -name: vllm -description: Serves LLMs with high throughput using vLLM's PagedAttention and continuous batching. Use when deploying production LLM APIs, optimizing inference latency/throughput, or serving models with limited GPU memory. Supports OpenAI-compatible endpoints, quantization (GPTQ/AWQ/FP8), and tensor parallelism. -version: 1.0.0 -author: Orchestra Research -license: MIT -tags: [vLLM, Inference Serving, PagedAttention, Continuous Batching, High Throughput, Production, OpenAI API, Quantization, Tensor Parallelism] -dependencies: [vllm, torch, transformers] ---- - -# vLLM - High-Performance LLM Serving - -## Quick start - -vLLM achieves 24x higher throughput than standard transformers through PagedAttention (block-based KV cache) and continuous batching (mixing prefill/decode requests). - -**Installation**: -```bash -pip install vllm -``` - -**Basic offline inference**: -```python -from vllm import LLM, SamplingParams - -llm = LLM(model="meta-llama/Llama-3-8B-Instruct") -sampling = SamplingParams(temperature=0.7, max_tokens=256) - -outputs = llm.generate(["Explain quantum computing"], sampling) -print(outputs[0].outputs[0].text) -``` - -**OpenAI-compatible server**: -```bash -vllm serve meta-llama/Llama-3-8B-Instruct - -# Query with OpenAI SDK -python -c " -from openai import OpenAI -client = OpenAI(base_url='http://localhost:8000/v1', api_key='EMPTY') -print(client.chat.completions.create( - model='meta-llama/Llama-3-8B-Instruct', - messages=[{'role': 'user', 'content': 'Hello!'}] -).choices[0].message.content) -" -``` - -## Common workflows - -### Workflow 1: Production API deployment - -Copy this checklist and track progress: - -``` -Deployment Progress: -- [ ] Step 1: Configure server settings -- [ ] Step 2: Test with limited traffic -- [ ] Step 3: Enable monitoring -- [ ] Step 4: Deploy to production -- [ ] Step 5: Verify performance metrics -``` - -**Step 1: Configure server settings** - -Choose configuration based on your model size: - -```bash -# For 7B-13B models on single GPU -vllm serve meta-llama/Llama-3-8B-Instruct \ - --gpu-memory-utilization 0.9 \ - --max-model-len 8192 \ - --port 8000 - -# For 30B-70B models with tensor parallelism -vllm serve meta-llama/Llama-2-70b-hf \ - --tensor-parallel-size 4 \ - --gpu-memory-utilization 0.9 \ - --quantization awq \ - --port 8000 - -# For production with caching and metrics -vllm serve meta-llama/Llama-3-8B-Instruct \ - --gpu-memory-utilization 0.9 \ - --enable-prefix-caching \ - --enable-metrics \ - --metrics-port 9090 \ - --port 8000 \ - --host 0.0.0.0 -``` - -**Step 2: Test with limited traffic** - -Run load test before production: - -```bash -# Install load testing tool -pip install locust - -# Create test_load.py with sample requests -# Run: locust -f test_load.py --host http://localhost:8000 -``` - -Verify TTFT (time to first token) < 500ms and throughput > 100 req/sec. - -**Step 3: Enable monitoring** - -vLLM exposes Prometheus metrics on port 9090: - -```bash -curl http://localhost:9090/metrics | grep vllm -``` - -Key metrics to monitor: -- `vllm:time_to_first_token_seconds` - Latency -- `vllm:num_requests_running` - Active requests -- `vllm:gpu_cache_usage_perc` - KV cache utilization - -**Step 4: Deploy to production** - -Use Docker for consistent deployment: - -```bash -# Run vLLM in Docker -docker run --gpus all -p 8000:8000 \ - vllm/vllm-openai:latest \ - --model meta-llama/Llama-3-8B-Instruct \ - --gpu-memory-utilization 0.9 \ - --enable-prefix-caching -``` - -**Step 5: Verify performance metrics** - -Check that deployment meets targets: -- TTFT < 500ms (for short prompts) -- Throughput > target req/sec -- GPU utilization > 80% -- No OOM errors in logs - -### Workflow 2: Offline batch inference - -For processing large datasets without server overhead. - -Copy this checklist: - -``` -Batch Processing: -- [ ] Step 1: Prepare input data -- [ ] Step 2: Configure LLM engine -- [ ] Step 3: Run batch inference -- [ ] Step 4: Process results -``` - -**Step 1: Prepare input data** - -```python -# Load prompts from file -prompts = [] -with open("prompts.txt") as f: - prompts = [line.strip() for line in f] - -print(f"Loaded {len(prompts)} prompts") -``` - -**Step 2: Configure LLM engine** - -```python -from vllm import LLM, SamplingParams - -llm = LLM( - model="meta-llama/Llama-3-8B-Instruct", - tensor_parallel_size=2, # Use 2 GPUs - gpu_memory_utilization=0.9, - max_model_len=4096 -) - -sampling = SamplingParams( - temperature=0.7, - top_p=0.95, - max_tokens=512, - stop=["", "\n\n"] -) -``` - -**Step 3: Run batch inference** - -vLLM automatically batches requests for efficiency: - -```python -# Process all prompts in one call -outputs = llm.generate(prompts, sampling) - -# vLLM handles batching internally -# No need to manually chunk prompts -``` - -**Step 4: Process results** - -```python -# Extract generated text -results = [] -for output in outputs: - prompt = output.prompt - generated = output.outputs[0].text - results.append({ - "prompt": prompt, - "generated": generated, - "tokens": len(output.outputs[0].token_ids) - }) - -# Save to file -import json -with open("results.jsonl", "w") as f: - for result in results: - f.write(json.dumps(result) + "\n") - -print(f"Processed {len(results)} prompts") -``` - -### Workflow 3: Quantized model serving - -Fit large models in limited GPU memory. - -``` -Quantization Setup: -- [ ] Step 1: Choose quantization method -- [ ] Step 2: Find or create quantized model -- [ ] Step 3: Launch with quantization flag -- [ ] Step 4: Verify accuracy -``` - -**Step 1: Choose quantization method** - -- **AWQ**: Best for 70B models, minimal accuracy loss -- **GPTQ**: Wide model support, good compression -- **FP8**: Fastest on H100 GPUs - -**Step 2: Find or create quantized model** - -Use pre-quantized models from HuggingFace: - -```bash -# Search for AWQ models -# Example: TheBloke/Llama-2-70B-AWQ -``` - -**Step 3: Launch with quantization flag** - -```bash -# Using pre-quantized model -vllm serve TheBloke/Llama-2-70B-AWQ \ - --quantization awq \ - --tensor-parallel-size 1 \ - --gpu-memory-utilization 0.95 - -# Results: 70B model in ~40GB VRAM -``` - -**Step 4: Verify accuracy** - -Test outputs match expected quality: - -```python -# Compare quantized vs non-quantized responses -# Verify task-specific performance unchanged -``` - -## When to use vs alternatives - -**Use vLLM when:** -- Deploying production LLM APIs (100+ req/sec) -- Serving OpenAI-compatible endpoints -- Limited GPU memory but need large models -- Multi-user applications (chatbots, assistants) -- Need low latency with high throughput - -**Use alternatives instead:** -- **llama.cpp**: CPU/edge inference, single-user -- **HuggingFace transformers**: Research, prototyping, one-off generation -- **TensorRT-LLM**: NVIDIA-only, need absolute maximum performance -- **Text-Generation-Inference**: Already in HuggingFace ecosystem - -## Common issues - -**Issue: Out of memory during model loading** - -Reduce memory usage: -```bash -vllm serve MODEL \ - --gpu-memory-utilization 0.7 \ - --max-model-len 4096 -``` - -Or use quantization: -```bash -vllm serve MODEL --quantization awq -``` - -**Issue: Slow first token (TTFT > 1 second)** - -Enable prefix caching for repeated prompts: -```bash -vllm serve MODEL --enable-prefix-caching -``` - -For long prompts, enable chunked prefill: -```bash -vllm serve MODEL --enable-chunked-prefill -``` - -**Issue: Model not found error** - -Use `--trust-remote-code` for custom models: -```bash -vllm serve MODEL --trust-remote-code -``` - -**Issue: Low throughput (<50 req/sec)** - -Increase concurrent sequences: -```bash -vllm serve MODEL --max-num-seqs 512 -``` - -Check GPU utilization with `nvidia-smi` - should be >80%. - -**Issue: Inference slower than expected** - -Verify tensor parallelism uses power of 2 GPUs: -```bash -vllm serve MODEL --tensor-parallel-size 4 # Not 3 -``` - -Enable speculative decoding for faster generation: -```bash -vllm serve MODEL --speculative-model DRAFT_MODEL -``` - -## Advanced topics - -**Server deployment patterns**: See [references/server-deployment.md](references/server-deployment.md) for Docker, Kubernetes, and load balancing configurations. - -**Performance optimization**: See [references/optimization.md](references/optimization.md) for PagedAttention tuning, continuous batching details, and benchmark results. - -**Quantization guide**: See [references/quantization.md](references/quantization.md) for AWQ/GPTQ/FP8 setup, model preparation, and accuracy comparisons. - -**Troubleshooting**: See [references/troubleshooting.md](references/troubleshooting.md) for detailed error messages, debugging steps, and performance diagnostics. - -## Hardware requirements - -- **Small models (7B-13B)**: 1x A10 (24GB) or A100 (40GB) -- **Medium models (30B-40B)**: 2x A100 (40GB) with tensor parallelism -- **Large models (70B+)**: 4x A100 (40GB) or 2x A100 (80GB), use AWQ/GPTQ - -Supported platforms: NVIDIA (primary), AMD ROCm, Intel GPUs, TPUs - -## Resources - -- Official docs: https://docs.vllm.ai -- GitHub: https://github.com/vllm-project/vllm -- Paper: "Efficient Memory Management for Large Language Model Serving with PagedAttention" (SOSP 2023) -- Community: https://discuss.vllm.ai - - - diff --git a/skills/vllm/references/optimization.md b/skills/vllm/references/optimization.md deleted file mode 100644 index 3d0cac5..0000000 --- a/skills/vllm/references/optimization.md +++ /dev/null @@ -1,226 +0,0 @@ -# Performance Optimization - -## Contents -- PagedAttention explained -- Continuous batching mechanics -- Prefix caching strategies -- Speculative decoding setup -- Benchmark results and comparisons -- Performance tuning guide - -## PagedAttention explained - -**Traditional attention problem**: -- KV cache stored in contiguous memory -- Wastes ~50% GPU memory due to fragmentation -- Cannot dynamically reallocate for varying sequence lengths - -**PagedAttention solution**: -- Divides KV cache into fixed-size blocks (like OS virtual memory) -- Dynamic allocation from free block queue -- Shares blocks across sequences (for prefix caching) - -**Memory savings example**: -``` -Traditional: 70B model needs 160GB KV cache → OOM on 8x A100 -PagedAttention: 70B model needs 80GB KV cache → Fits on 4x A100 -``` - -**Configuration**: -```bash -# Block size (default: 16 tokens) -vllm serve MODEL --block-size 16 - -# Number of GPU blocks (auto-calculated) -# Controlled by --gpu-memory-utilization -vllm serve MODEL --gpu-memory-utilization 0.9 -``` - -## Continuous batching mechanics - -**Traditional batching**: -- Wait for all sequences in batch to finish -- GPU idle while waiting for longest sequence -- Low GPU utilization (~40-60%) - -**Continuous batching**: -- Add new requests as slots become available -- Mix prefill (new requests) and decode (ongoing) in same batch -- High GPU utilization (>90%) - -**Throughput improvement**: -``` -Traditional batching: 50 req/sec @ 50% GPU util -Continuous batching: 200 req/sec @ 90% GPU util -= 4x throughput improvement -``` - -**Tuning parameters**: -```bash -# Max concurrent sequences (higher = more batching) -vllm serve MODEL --max-num-seqs 256 - -# Prefill/decode schedule (auto-balanced by default) -# No manual tuning needed -``` - -## Prefix caching strategies - -Reuse computed KV cache for common prompt prefixes. - -**Use cases**: -- System prompts repeated across requests -- Few-shot examples in every prompt -- RAG contexts with overlapping chunks - -**Example savings**: -``` -Prompt: [System: 500 tokens] + [User: 100 tokens] - -Without caching: Compute 600 tokens every request -With caching: Compute 500 tokens once, then 100 tokens/request -= 83% faster TTFT -``` - -**Enable prefix caching**: -```bash -vllm serve MODEL --enable-prefix-caching -``` - -**Automatic prefix detection**: -- vLLM detects common prefixes automatically -- No code changes required -- Works with OpenAI-compatible API - -**Cache hit rate monitoring**: -```bash -curl http://localhost:9090/metrics | grep cache_hit -# vllm_cache_hit_rate: 0.75 (75% hit rate) -``` - -## Speculative decoding setup - -Use smaller "draft" model to propose tokens, larger model to verify. - -**Speed improvement**: -``` -Standard: Generate 1 token per forward pass -Speculative: Generate 3-5 tokens per forward pass -= 2-3x faster generation -``` - -**How it works**: -1. Draft model proposes K tokens (fast) -2. Target model verifies all K tokens in parallel (one pass) -3. Accept verified tokens, restart from first rejection - -**Setup with separate draft model**: -```bash -vllm serve meta-llama/Llama-3-70B-Instruct \ - --speculative-model TinyLlama/TinyLlama-1.1B-Chat-v1.0 \ - --num-speculative-tokens 5 -``` - -**Setup with n-gram draft** (no separate model): -```bash -vllm serve MODEL \ - --speculative-method ngram \ - --num-speculative-tokens 3 -``` - -**When to use**: -- Output length > 100 tokens -- Draft model 5-10x smaller than target -- Acceptable 2-3% accuracy trade-off - -## Benchmark results - -**vLLM vs HuggingFace Transformers** (Llama 3 8B, A100): -``` -Metric | HF Transformers | vLLM | Improvement -------------------------|-----------------|--------|------------ -Throughput (req/sec) | 12 | 280 | 23x -TTFT (ms) | 850 | 120 | 7x -Tokens/sec | 45 | 2,100 | 47x -GPU Memory (GB) | 28 | 16 | 1.75x less -``` - -**vLLM vs TensorRT-LLM** (Llama 2 70B, 4x A100): -``` -Metric | TensorRT-LLM | vLLM | Notes -------------------------|--------------|--------|------------------ -Throughput (req/sec) | 320 | 285 | TRT 12% faster -Setup complexity | High | Low | vLLM much easier -NVIDIA-only | Yes | No | vLLM multi-platform -Quantization support | FP8, INT8 | AWQ/GPTQ/FP8 | vLLM more options -``` - -## Performance tuning guide - -**Step 1: Measure baseline** - -```bash -# Install benchmarking tool -pip install locust - -# Run baseline benchmark -vllm bench throughput \ - --model MODEL \ - --input-tokens 128 \ - --output-tokens 256 \ - --num-prompts 1000 - -# Record: throughput, TTFT, tokens/sec -``` - -**Step 2: Tune memory utilization** - -```bash -# Try different values: 0.7, 0.85, 0.9, 0.95 -vllm serve MODEL --gpu-memory-utilization 0.9 -``` - -Higher = more batch capacity = higher throughput, but risk OOM. - -**Step 3: Tune concurrency** - -```bash -# Try values: 128, 256, 512, 1024 -vllm serve MODEL --max-num-seqs 256 -``` - -Higher = more batching opportunity, but may increase latency. - -**Step 4: Enable optimizations** - -```bash -vllm serve MODEL \ - --enable-prefix-caching \ # For repeated prompts - --enable-chunked-prefill \ # For long prompts - --gpu-memory-utilization 0.9 \ - --max-num-seqs 512 -``` - -**Step 5: Re-benchmark and compare** - -Target improvements: -- Throughput: +30-100% -- TTFT: -20-50% -- GPU utilization: >85% - -**Common performance issues**: - -**Low throughput (<50 req/sec)**: -- Increase `--max-num-seqs` -- Enable `--enable-prefix-caching` -- Check GPU utilization (should be >80%) - -**High TTFT (>1 second)**: -- Enable `--enable-chunked-prefill` -- Reduce `--max-model-len` if possible -- Check if model is too large for GPU - -**OOM errors**: -- Reduce `--gpu-memory-utilization` to 0.7 -- Reduce `--max-model-len` -- Use quantization (`--quantization awq`) diff --git a/skills/vllm/references/quantization.md b/skills/vllm/references/quantization.md deleted file mode 100644 index 44901a2..0000000 --- a/skills/vllm/references/quantization.md +++ /dev/null @@ -1,284 +0,0 @@ -# Quantization Guide - -## Contents -- Quantization methods comparison -- AWQ setup and usage -- GPTQ setup and usage -- FP8 quantization (H100) -- Model preparation -- Accuracy vs compression trade-offs - -## Quantization methods comparison - -| Method | Compression | Accuracy Loss | Speed | Best For | -|--------|-------------|---------------|-------|----------| -| **AWQ** | 4-bit (75%) | <1% | Fast | 70B models, production | -| **GPTQ** | 4-bit (75%) | 1-2% | Fast | Wide model support | -| **FP8** | 8-bit (50%) | <0.5% | Fastest | H100 GPUs only | -| **SqueezeLLM** | 3-4 bit (75-80%) | 2-3% | Medium | Extreme compression | - -**Recommendation**: -- **Production**: Use AWQ for 70B models -- **H100 GPUs**: Use FP8 for best speed -- **Maximum compatibility**: Use GPTQ -- **Extreme compression**: Use SqueezeLLM - -## AWQ setup and usage - -**AWQ** (Activation-aware Weight Quantization) achieves best accuracy at 4-bit. - -**Step 1: Find pre-quantized model** - -Search HuggingFace for AWQ models: -```bash -# Example: TheBloke/Llama-2-70B-AWQ -# Example: TheBloke/Mixtral-8x7B-Instruct-v0.1-AWQ -``` - -**Step 2: Launch with AWQ** - -```bash -vllm serve TheBloke/Llama-2-70B-AWQ \ - --quantization awq \ - --tensor-parallel-size 1 \ - --gpu-memory-utilization 0.95 -``` - -**Memory savings**: -``` -Llama 2 70B fp16: 140GB VRAM (4x A100 needed) -Llama 2 70B AWQ: 35GB VRAM (1x A100 40GB) -= 4x memory reduction -``` - -**Step 3: Verify performance** - -Test that outputs are acceptable: -```python -from openai import OpenAI - -client = OpenAI(base_url="http://localhost:8000/v1", api_key="EMPTY") - -# Test complex reasoning -response = client.chat.completions.create( - model="TheBloke/Llama-2-70B-AWQ", - messages=[{"role": "user", "content": "Explain quantum entanglement"}] -) - -print(response.choices[0].message.content) -# Verify quality matches your requirements -``` - -**Quantize your own model** (requires GPU with 80GB+ VRAM): - -```python -from awq import AutoAWQForCausalLM -from transformers import AutoTokenizer - -model_path = "meta-llama/Llama-2-70b-hf" -quant_path = "llama-2-70b-awq" - -# Load model -model = AutoAWQForCausalLM.from_pretrained(model_path) -tokenizer = AutoTokenizer.from_pretrained(model_path) - -# Quantize -quant_config = {"zero_point": True, "q_group_size": 128, "w_bit": 4} -model.quantize(tokenizer, quant_config=quant_config) - -# Save -model.save_quantized(quant_path) -tokenizer.save_pretrained(quant_path) -``` - -## GPTQ setup and usage - -**GPTQ** has widest model support and good compression. - -**Step 1: Find GPTQ model** - -```bash -# Example: TheBloke/Llama-2-13B-GPTQ -# Example: TheBloke/CodeLlama-34B-GPTQ -``` - -**Step 2: Launch with GPTQ** - -```bash -vllm serve TheBloke/Llama-2-13B-GPTQ \ - --quantization gptq \ - --dtype float16 -``` - -**GPTQ configuration options**: -```bash -# Specify GPTQ parameters if needed -vllm serve MODEL \ - --quantization gptq \ - --gptq-act-order \ # Activation ordering - --dtype float16 -``` - -**Quantize your own model**: - -```python -from auto_gptq import AutoGPTQForCausalLM, BaseQuantizeConfig -from transformers import AutoTokenizer - -model_name = "meta-llama/Llama-2-13b-hf" -quantized_name = "llama-2-13b-gptq" - -# Load model -tokenizer = AutoTokenizer.from_pretrained(model_name) -model = AutoGPTQForCausalLM.from_pretrained(model_name, quantize_config) - -# Prepare calibration data -calib_data = [...] # List of sample texts - -# Quantize -quantize_config = BaseQuantizeConfig( - bits=4, - group_size=128, - desc_act=True -) -model.quantize(calib_data) - -# Save -model.save_quantized(quantized_name) -``` - -## FP8 quantization (H100) - -**FP8** (8-bit floating point) offers best speed on H100 GPUs with minimal accuracy loss. - -**Requirements**: -- H100 or H800 GPU -- CUDA 12.3+ (12.8 recommended) -- Hopper architecture support - -**Step 1: Enable FP8** - -```bash -vllm serve meta-llama/Llama-3-70B-Instruct \ - --quantization fp8 \ - --tensor-parallel-size 2 -``` - -**Performance gains on H100**: -``` -fp16: 180 tokens/sec -FP8: 320 tokens/sec -= 1.8x speedup -``` - -**Step 2: Verify accuracy** - -FP8 typically has <0.5% accuracy degradation: -```python -# Run evaluation suite -# Compare FP8 vs FP16 on your tasks -# Verify acceptable accuracy -``` - -**Dynamic FP8 quantization** (no pre-quantized model needed): - -```bash -# vLLM automatically quantizes at runtime -vllm serve MODEL --quantization fp8 -# No model preparation required -``` - -## Model preparation - -**Pre-quantized models (easiest)**: - -1. Search HuggingFace: `[model name] AWQ` or `[model name] GPTQ` -2. Download or use directly: `TheBloke/[Model]-AWQ` -3. Launch with appropriate `--quantization` flag - -**Quantize your own model**: - -**AWQ**: -```bash -# Install AutoAWQ -pip install autoawq - -# Run quantization script -python quantize_awq.py --model MODEL --output OUTPUT -``` - -**GPTQ**: -```bash -# Install AutoGPTQ -pip install auto-gptq - -# Run quantization script -python quantize_gptq.py --model MODEL --output OUTPUT -``` - -**Calibration data**: -- Use 128-512 diverse examples from target domain -- Representative of production inputs -- Higher quality calibration = better accuracy - -## Accuracy vs compression trade-offs - -**Empirical results** (Llama 2 70B on MMLU benchmark): - -| Quantization | Accuracy | Memory | Speed | Production-Ready | -|--------------|----------|--------|-------|------------------| -| FP16 (baseline) | 100% | 140GB | 1.0x | ✅ (if memory available) | -| FP8 | 99.5% | 70GB | 1.8x | ✅ (H100 only) | -| AWQ 4-bit | 99.0% | 35GB | 1.5x | ✅ (best for 70B) | -| GPTQ 4-bit | 98.5% | 35GB | 1.5x | ✅ (good compatibility) | -| SqueezeLLM 3-bit | 96.0% | 26GB | 1.3x | ⚠️ (check accuracy) | - -**When to use each**: - -**No quantization (FP16)**: -- Have sufficient GPU memory -- Need absolute best accuracy -- Model <13B parameters - -**FP8**: -- Using H100/H800 GPUs -- Need best speed with minimal accuracy loss -- Production deployment - -**AWQ 4-bit**: -- Need to fit 70B model in 40GB GPU -- Production deployment -- <1% accuracy loss acceptable - -**GPTQ 4-bit**: -- Wide model support needed -- Not on H100 (use FP8 instead) -- 1-2% accuracy loss acceptable - -**Testing strategy**: - -1. **Baseline**: Measure FP16 accuracy on your evaluation set -2. **Quantize**: Create quantized version -3. **Evaluate**: Compare quantized vs baseline on same tasks -4. **Decide**: Accept if degradation < threshold (typically 1-2%) - -**Example evaluation**: -```python -from evaluate import load_evaluation_suite - -# Run on FP16 baseline -baseline_score = evaluate(model_fp16, eval_suite) - -# Run on quantized -quant_score = evaluate(model_awq, eval_suite) - -# Compare -degradation = (baseline_score - quant_score) / baseline_score * 100 -print(f"Accuracy degradation: {degradation:.2f}%") - -# Decision -if degradation < 1.0: - print("✅ Quantization acceptable for production") -else: - print("⚠️ Review accuracy loss") -``` diff --git a/skills/vllm/references/server-deployment.md b/skills/vllm/references/server-deployment.md deleted file mode 100644 index da5b837..0000000 --- a/skills/vllm/references/server-deployment.md +++ /dev/null @@ -1,255 +0,0 @@ -# Server Deployment Patterns - -## Contents -- Docker deployment -- Kubernetes deployment -- Load balancing with Nginx -- Multi-node distributed serving -- Production configuration examples -- Health checks and monitoring - -## Docker deployment - -**Basic Dockerfile**: -```dockerfile -FROM nvidia/cuda:12.1.0-devel-ubuntu22.04 - -RUN apt-get update && apt-get install -y python3-pip -RUN pip install vllm - -EXPOSE 8000 - -CMD ["vllm", "serve", "meta-llama/Llama-3-8B-Instruct", \ - "--host", "0.0.0.0", "--port", "8000", \ - "--gpu-memory-utilization", "0.9"] -``` - -**Build and run**: -```bash -docker build -t vllm-server . -docker run --gpus all -p 8000:8000 vllm-server -``` - -**Docker Compose** (with metrics): -```yaml -version: '3.8' -services: - vllm: - image: vllm/vllm-openai:latest - command: > - --model meta-llama/Llama-3-8B-Instruct - --gpu-memory-utilization 0.9 - --enable-metrics - --metrics-port 9090 - ports: - - "8000:8000" - - "9090:9090" - deploy: - resources: - reservations: - devices: - - driver: nvidia - count: all - capabilities: [gpu] -``` - -## Kubernetes deployment - -**Deployment manifest**: -```yaml -apiVersion: apps/v1 -kind: Deployment -metadata: - name: vllm-server -spec: - replicas: 2 - selector: - matchLabels: - app: vllm - template: - metadata: - labels: - app: vllm - spec: - containers: - - name: vllm - image: vllm/vllm-openai:latest - args: - - "--model=meta-llama/Llama-3-8B-Instruct" - - "--gpu-memory-utilization=0.9" - - "--enable-prefix-caching" - resources: - limits: - nvidia.com/gpu: 1 - ports: - - containerPort: 8000 - name: http - - containerPort: 9090 - name: metrics - readinessProbe: - httpGet: - path: /health - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 10 - livenessProbe: - httpGet: - path: /health - port: 8000 - initialDelaySeconds: 60 - periodSeconds: 30 ---- -apiVersion: v1 -kind: Service -metadata: - name: vllm-service -spec: - selector: - app: vllm - ports: - - port: 8000 - targetPort: 8000 - name: http - - port: 9090 - targetPort: 9090 - name: metrics - type: LoadBalancer -``` - -## Load balancing with Nginx - -**Nginx configuration**: -```nginx -upstream vllm_backend { - least_conn; # Route to least-loaded server - server localhost:8001; - server localhost:8002; - server localhost:8003; -} - -server { - listen 80; - - location / { - proxy_pass http://vllm_backend; - proxy_set_header Host $host; - proxy_set_header X-Real-IP $remote_addr; - - # Timeouts for long-running inference - proxy_read_timeout 300s; - proxy_connect_timeout 75s; - } - - # Metrics endpoint - location /metrics { - proxy_pass http://localhost:9090/metrics; - } -} -``` - -**Start multiple vLLM instances**: -```bash -# Terminal 1 -vllm serve MODEL --port 8001 --tensor-parallel-size 1 - -# Terminal 2 -vllm serve MODEL --port 8002 --tensor-parallel-size 1 - -# Terminal 3 -vllm serve MODEL --port 8003 --tensor-parallel-size 1 - -# Start Nginx -nginx -c /path/to/nginx.conf -``` - -## Multi-node distributed serving - -For models too large for single node: - -**Node 1** (master): -```bash -export MASTER_ADDR=192.168.1.10 -export MASTER_PORT=29500 -export RANK=0 -export WORLD_SIZE=2 - -vllm serve meta-llama/Llama-2-70b-hf \ - --tensor-parallel-size 8 \ - --pipeline-parallel-size 2 -``` - -**Node 2** (worker): -```bash -export MASTER_ADDR=192.168.1.10 -export MASTER_PORT=29500 -export RANK=1 -export WORLD_SIZE=2 - -vllm serve meta-llama/Llama-2-70b-hf \ - --tensor-parallel-size 8 \ - --pipeline-parallel-size 2 -``` - -## Production configuration examples - -**High throughput** (batch-heavy workload): -```bash -vllm serve MODEL \ - --max-num-seqs 512 \ - --gpu-memory-utilization 0.95 \ - --enable-prefix-caching \ - --trust-remote-code -``` - -**Low latency** (interactive workload): -```bash -vllm serve MODEL \ - --max-num-seqs 64 \ - --gpu-memory-utilization 0.85 \ - --enable-chunked-prefill -``` - -**Memory-constrained** (40GB GPU for 70B model): -```bash -vllm serve TheBloke/Llama-2-70B-AWQ \ - --quantization awq \ - --tensor-parallel-size 1 \ - --gpu-memory-utilization 0.95 \ - --max-model-len 4096 -``` - -## Health checks and monitoring - -**Health check endpoint**: -```bash -curl http://localhost:8000/health -# Returns: {"status": "ok"} -``` - -**Readiness check** (wait for model loaded): -```bash -#!/bin/bash -until curl -f http://localhost:8000/health; do - echo "Waiting for vLLM to be ready..." - sleep 5 -done -echo "vLLM is ready!" -``` - -**Prometheus scraping**: -```yaml -# prometheus.yml -scrape_configs: - - job_name: 'vllm' - static_configs: - - targets: ['localhost:9090'] - metrics_path: '/metrics' - scrape_interval: 15s -``` - -**Grafana dashboard** (key metrics): -- Requests per second: `rate(vllm_request_success_total[5m])` -- TTFT p50: `histogram_quantile(0.5, vllm_time_to_first_token_seconds_bucket)` -- TTFT p99: `histogram_quantile(0.99, vllm_time_to_first_token_seconds_bucket)` -- GPU cache usage: `vllm_gpu_cache_usage_perc` -- Active requests: `vllm_num_requests_running` diff --git a/skills/vllm/references/troubleshooting.md b/skills/vllm/references/troubleshooting.md deleted file mode 100644 index c00cc9a..0000000 --- a/skills/vllm/references/troubleshooting.md +++ /dev/null @@ -1,447 +0,0 @@ -# Troubleshooting Guide - -## Contents -- Out of memory (OOM) errors -- Performance issues -- Model loading errors -- Network and connection issues -- Quantization problems -- Distributed serving issues -- Debugging tools and commands - -## Out of memory (OOM) errors - -### Symptom: `torch.cuda.OutOfMemoryError` during model loading - -**Cause**: Model + KV cache exceeds available VRAM - -**Solutions (try in order)**: - -1. **Reduce GPU memory utilization**: -```bash -vllm serve MODEL --gpu-memory-utilization 0.7 # Try 0.7, 0.75, 0.8 -``` - -2. **Reduce max sequence length**: -```bash -vllm serve MODEL --max-model-len 4096 # Instead of 8192 -``` - -3. **Enable quantization**: -```bash -vllm serve MODEL --quantization awq # 4x memory reduction -``` - -4. **Use tensor parallelism** (multiple GPUs): -```bash -vllm serve MODEL --tensor-parallel-size 2 # Split across 2 GPUs -``` - -5. **Reduce max concurrent sequences**: -```bash -vllm serve MODEL --max-num-seqs 128 # Default is 256 -``` - -### Symptom: OOM during inference (not model loading) - -**Cause**: KV cache fills up during generation - -**Solutions**: - -```bash -# Reduce KV cache allocation -vllm serve MODEL --gpu-memory-utilization 0.85 - -# Reduce batch size -vllm serve MODEL --max-num-seqs 64 - -# Reduce max tokens per request -# Set in client request: max_tokens=512 -``` - -### Symptom: OOM with quantized model - -**Cause**: Quantization overhead or incorrect configuration - -**Solution**: -```bash -# Ensure quantization flag matches model -vllm serve TheBloke/Llama-2-70B-AWQ --quantization awq # Must specify - -# Try different dtype -vllm serve MODEL --quantization awq --dtype float16 -``` - -## Performance issues - -### Symptom: Low throughput (<50 req/sec expected >100) - -**Diagnostic steps**: - -1. **Check GPU utilization**: -```bash -watch -n 1 nvidia-smi -# GPU utilization should be >80% -``` - -If <80%, increase concurrent requests: -```bash -vllm serve MODEL --max-num-seqs 512 # Increase from 256 -``` - -2. **Check if memory-bound**: -```bash -# If memory at 100% but GPU <80%, reduce sequence length -vllm serve MODEL --max-model-len 4096 -``` - -3. **Enable optimizations**: -```bash -vllm serve MODEL \ - --enable-prefix-caching \ - --enable-chunked-prefill \ - --max-num-seqs 512 -``` - -4. **Check tensor parallelism settings**: -```bash -# Must use power-of-2 GPUs -vllm serve MODEL --tensor-parallel-size 4 # Not 3 or 5 -``` - -### Symptom: High TTFT (time to first token >1 second) - -**Causes and solutions**: - -**Long prompts**: -```bash -vllm serve MODEL --enable-chunked-prefill -``` - -**No prefix caching**: -```bash -vllm serve MODEL --enable-prefix-caching # For repeated prompts -``` - -**Too many concurrent requests**: -```bash -vllm serve MODEL --max-num-seqs 64 # Reduce to prioritize latency -``` - -**Model too large for single GPU**: -```bash -vllm serve MODEL --tensor-parallel-size 2 # Parallelize prefill -``` - -### Symptom: Slow token generation (low tokens/sec) - -**Diagnostic**: -```bash -# Check if model is correct size -vllm serve MODEL # Should see model size in logs - -# Check speculative decoding -vllm serve MODEL --speculative-model DRAFT_MODEL -``` - -**For H100 GPUs**, enable FP8: -```bash -vllm serve MODEL --quantization fp8 -``` - -## Model loading errors - -### Symptom: `OSError: MODEL not found` - -**Causes**: - -1. **Model name typo**: -```bash -# Check exact model name on HuggingFace -vllm serve meta-llama/Llama-3-8B-Instruct # Correct capitalization -``` - -2. **Private/gated model**: -```bash -# Login to HuggingFace first -huggingface-cli login -# Then run vLLM -vllm serve meta-llama/Llama-3-70B-Instruct -``` - -3. **Custom model needs trust flag**: -```bash -vllm serve MODEL --trust-remote-code -``` - -### Symptom: `ValueError: Tokenizer not found` - -**Solution**: -```bash -# Download model manually first -python -c "from transformers import AutoTokenizer; AutoTokenizer.from_pretrained('MODEL')" - -# Then launch vLLM -vllm serve MODEL -``` - -### Symptom: `ImportError: No module named 'flash_attn'` - -**Solution**: -```bash -# Install flash attention -pip install flash-attn --no-build-isolation - -# Or disable flash attention -vllm serve MODEL --disable-flash-attn -``` - -## Network and connection issues - -### Symptom: `Connection refused` when querying server - -**Diagnostic**: - -1. **Check server is running**: -```bash -curl http://localhost:8000/health -``` - -2. **Check port binding**: -```bash -# Bind to all interfaces for remote access -vllm serve MODEL --host 0.0.0.0 --port 8000 - -# Check if port is in use -lsof -i :8000 -``` - -3. **Check firewall**: -```bash -# Allow port through firewall -sudo ufw allow 8000 -``` - -### Symptom: Slow response times over network - -**Solutions**: - -1. **Increase timeout**: -```python -from openai import OpenAI - -client = OpenAI( - base_url="http://localhost:8000/v1", - api_key="EMPTY", - timeout=300.0 # 5 minute timeout -) -``` - -2. **Check network latency**: -```bash -ping SERVER_IP # Should be <10ms for local network -``` - -3. **Use connection pooling**: -```python -import requests -from requests.adapters import HTTPAdapter -from urllib3.util.retry import Retry - -session = requests.Session() -retries = Retry(total=3, backoff_factor=1) -session.mount('http://', HTTPAdapter(max_retries=retries)) -``` - -## Quantization problems - -### Symptom: `RuntimeError: Quantization format not supported` - -**Solution**: -```bash -# Ensure correct quantization method -vllm serve MODEL --quantization awq # For AWQ models -vllm serve MODEL --quantization gptq # For GPTQ models - -# Check model card for quantization type -``` - -### Symptom: Poor quality outputs after quantization - -**Diagnostic**: - -1. **Verify model is correctly quantized**: -```bash -# Check model config.json for quantization_config -cat ~/.cache/huggingface/hub/models--MODEL/config.json -``` - -2. **Try different quantization method**: -```bash -# If AWQ quality issues, try FP8 (H100 only) -vllm serve MODEL --quantization fp8 - -# Or use less aggressive quantization -vllm serve MODEL # No quantization -``` - -3. **Increase temperature for better diversity**: -```python -sampling_params = SamplingParams(temperature=0.8, top_p=0.95) -``` - -## Distributed serving issues - -### Symptom: `RuntimeError: Distributed init failed` - -**Diagnostic**: - -1. **Check environment variables**: -```bash -# On all nodes -echo $MASTER_ADDR # Should be same -echo $MASTER_PORT # Should be same -echo $RANK # Should be unique per node (0, 1, 2, ...) -echo $WORLD_SIZE # Should be same (total nodes) -``` - -2. **Check network connectivity**: -```bash -# From node 1 to node 2 -ping NODE2_IP -nc -zv NODE2_IP 29500 # Check port accessibility -``` - -3. **Check NCCL settings**: -```bash -export NCCL_DEBUG=INFO -export NCCL_SOCKET_IFNAME=eth0 # Or your network interface -vllm serve MODEL --tensor-parallel-size 8 -``` - -### Symptom: `NCCL error: unhandled cuda error` - -**Solutions**: - -```bash -# Set NCCL to use correct network interface -export NCCL_SOCKET_IFNAME=eth0 # Replace with your interface - -# Increase timeout -export NCCL_TIMEOUT=1800 # 30 minutes - -# Force P2P for debugging -export NCCL_P2P_DISABLE=1 -``` - -## Debugging tools and commands - -### Enable debug logging - -```bash -export VLLM_LOGGING_LEVEL=DEBUG -vllm serve MODEL -``` - -### Monitor GPU usage - -```bash -# Real-time GPU monitoring -watch -n 1 nvidia-smi - -# Memory breakdown -nvidia-smi --query-gpu=memory.used,memory.free --format=csv -l 1 -``` - -### Profile performance - -```bash -# Built-in benchmarking -vllm bench throughput \ - --model MODEL \ - --input-tokens 128 \ - --output-tokens 256 \ - --num-prompts 100 - -vllm bench latency \ - --model MODEL \ - --input-tokens 128 \ - --output-tokens 256 \ - --batch-size 8 -``` - -### Check metrics - -```bash -# Prometheus metrics -curl http://localhost:9090/metrics - -# Filter for specific metrics -curl http://localhost:9090/metrics | grep vllm_time_to_first_token - -# Key metrics to monitor: -# - vllm_time_to_first_token_seconds -# - vllm_time_per_output_token_seconds -# - vllm_num_requests_running -# - vllm_gpu_cache_usage_perc -# - vllm_request_success_total -``` - -### Test server health - -```bash -# Health check -curl http://localhost:8000/health - -# Model info -curl http://localhost:8000/v1/models - -# Test completion -curl http://localhost:8000/v1/completions \ - -H "Content-Type: application/json" \ - -d '{ - "model": "MODEL", - "prompt": "Hello", - "max_tokens": 10 - }' -``` - -### Common environment variables - -```bash -# CUDA settings -export CUDA_VISIBLE_DEVICES=0,1,2,3 # Limit to specific GPUs - -# vLLM settings -export VLLM_LOGGING_LEVEL=DEBUG -export VLLM_TRACE_FUNCTION=1 # Profile functions -export VLLM_USE_V1=1 # Use v1.0 engine (faster) - -# NCCL settings (distributed) -export NCCL_DEBUG=INFO -export NCCL_SOCKET_IFNAME=eth0 -export NCCL_IB_DISABLE=0 # Enable InfiniBand -``` - -### Collect diagnostic info for bug reports - -```bash -# System info -nvidia-smi -python --version -pip show vllm - -# vLLM version and config -vllm --version -python -c "import vllm; print(vllm.__version__)" - -# Run with debug logging -export VLLM_LOGGING_LEVEL=DEBUG -vllm serve MODEL 2>&1 | tee vllm_debug.log - -# Include in bug report: -# - vllm_debug.log -# - nvidia-smi output -# - Full command used -# - Expected vs actual behavior -```