diff --git a/EvoScientist/skills/skill-creator/assets/eval_review.html b/EvoScientist/skills/skill-creator/assets/eval_review.html index b8372a4..45c1ed7 100644 --- a/EvoScientist/skills/skill-creator/assets/eval_review.html +++ b/EvoScientist/skills/skill-creator/assets/eval_review.html @@ -4,6 +4,7 @@ Eval Set Review - __SKILL_NAME_PLACEHOLDER__ + diff --git a/EvoScientist/skills/skill-creator/eval-viewer/viewer.html b/EvoScientist/skills/skill-creator/eval-viewer/viewer.html index 83b760a..66793ec 100644 --- a/EvoScientist/skills/skill-creator/eval-viewer/viewer.html +++ b/EvoScientist/skills/skill-creator/eval-viewer/viewer.html @@ -4,6 +4,7 @@ Eval Review + @@ -826,8 +827,17 @@ } } - // ---- XLSX rendering via SheetJS ---- + // ---- XLSX rendering via SheetJS (degrades to download link if CDN unavailable) ---- function renderXlsx(container, b64Data) { + if (typeof XLSX === "undefined") { + const a = document.createElement("a"); + a.className = "download-link"; + a.href = "data:application/vnd.openxmlformats-officedocument.spreadsheetml.sheet;base64," + b64Data; + a.download = "spreadsheet.xlsx"; + a.textContent = "Download .xlsx (SheetJS unavailable offline)"; + container.appendChild(a); + return; + } try { const raw = Uint8Array.from(atob(b64Data), c => c.charCodeAt(0)); const wb = XLSX.read(raw, { type: "array" }); diff --git a/EvoScientist/skills/skill-creator/scripts/aggregate_benchmark.py b/EvoScientist/skills/skill-creator/scripts/aggregate_benchmark.py index 3bd9776..6df7bd6 100755 --- a/EvoScientist/skills/skill-creator/scripts/aggregate_benchmark.py +++ b/EvoScientist/skills/skill-creator/scripts/aggregate_benchmark.py @@ -268,10 +268,9 @@ def generate_benchmark(benchmark_dir: Path, skill_name: str = "", skill_path: st "analyzer_model": "", "timestamp": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), "evals_run": eval_ids, - "runs_per_configuration": min( - (len(runs) for runs in results.values() if runs), - default=0, - ) + "runs_per_configuration": { + config: len(runs) for config, runs in results.items() if runs + } }, "runs": runs, "run_summary": run_summary, @@ -281,6 +280,19 @@ def generate_benchmark(benchmark_dir: Path, skill_name: str = "", skill_path: st return benchmark +def _format_runs_per_config(rpc) -> str: + """Format runs_per_configuration for display (handles both int and dict).""" + if isinstance(rpc, int): + return f"{rpc} runs each" + if isinstance(rpc, dict): + counts = sorted(set(rpc.values())) + if len(counts) == 1: + return f"{counts[0]} runs each" + parts = [f"{config}: {n}" for config, n in rpc.items()] + return ", ".join(parts) + return str(rpc) + + def generate_markdown(benchmark: dict) -> str: """Generate human-readable benchmark.md from benchmark data.""" metadata = benchmark["metadata"] @@ -298,7 +310,7 @@ def generate_markdown(benchmark: dict) -> str: "", f"**Model**: {metadata['executor_model']}", f"**Date**: {metadata['timestamp']}", - f"**Evals**: {', '.join(map(str, metadata['evals_run']))} ({metadata['runs_per_configuration']} runs each per configuration)", + f"**Evals**: {', '.join(map(str, metadata['evals_run']))} ({_format_runs_per_config(metadata['runs_per_configuration'])} per configuration)", "", "## Summary", "", diff --git a/EvoScientist/skills/skill-creator/scripts/package_skill.py b/EvoScientist/skills/skill-creator/scripts/package_skill.py index cbe6bea..c1a13cb 100755 --- a/EvoScientist/skills/skill-creator/scripts/package_skill.py +++ b/EvoScientist/skills/skill-creator/scripts/package_skill.py @@ -71,9 +71,9 @@ def package_skill(skill_path, output_dir=None): print(f"❌ Error: SKILL.md not found in {skill_path}") return None - # Run validation before packaging + # Run validation before packaging (strict mode catches TODO placeholders) print("🔍 Validating skill...") - valid, message = validate_skill(skill_path) + valid, message = validate_skill(skill_path, strict=True) if not valid: print(f"❌ Validation failed: {message}") print(" Please fix the validation errors before packaging.") @@ -113,22 +113,21 @@ def package_skill(skill_path, output_dir=None): def main(): - if len(sys.argv) < 2: - print("Usage: python utils/package_skill.py [output-directory]") - print("\nExample:") - print(" python utils/package_skill.py skills/public/my-skill") - print(" python utils/package_skill.py skills/public/my-skill ./dist") - sys.exit(1) + import argparse + parser = argparse.ArgumentParser( + description="Package a skill folder into a distributable .skill file" + ) + parser.add_argument("skill_path", help="Path to the skill folder") + parser.add_argument("output_dir", nargs="?", default=None, + help="Output directory for the .skill file (default: current directory)") + args = parser.parse_args() - skill_path = sys.argv[1] - output_dir = sys.argv[2] if len(sys.argv) > 2 else None - - print(f"📦 Packaging skill: {skill_path}") - if output_dir: - print(f" Output directory: {output_dir}") + print(f"📦 Packaging skill: {args.skill_path}") + if args.output_dir: + print(f" Output directory: {args.output_dir}") print() - result = package_skill(skill_path, output_dir) + result = package_skill(args.skill_path, args.output_dir) if result: sys.exit(0)