{"_id":"@alexanderzzlatkov/skilleval","_rev":"4-644a8713f2a658746261df0bbcc5f0be","name":"@alexanderzzlatkov/skilleval","dist-tags":{"latest":"0.2.1"},"versions":{"0.1.0":{"name":"@alexanderzzlatkov/skilleval","version":"0.1.0","keywords":["ai","agent","skills","eval","llm","claude-code","openrouter","skilleval"],"author":{"url":"https://github.com/zlatkov","name":"Alexander Zlatkov"},"license":"MIT","_id":"@alexanderzzlatkov/skilleval@0.1.0","maintainers":[{"name":"alexanderzzlatkov","email":"alexander.z.zlatkov@gmail.com"}],"homepage":"https://github.com/zlatkov/skilleval#readme","bugs":{"url":"https://github.com/zlatkov/skilleval/issues"},"bin":{"skilleval":"dist/index.js"},"dist":{"shasum":"91b39c03553cf3ce3eccad8e94a6e9592be2c21e","tarball":"https://registry.npmjs.org/@alexanderzzlatkov/skilleval/-/skilleval-0.1.0.tgz","fileCount":21,"integrity":"sha512-IEnE0q3zKEqDogWKkRYLJebHq1QiQkiPg3vEHyfCYmLNI5gRRnKlJ4qR3ovyhR6OXIF6I1+w32gqLncpT1APVA==","signatures":[{"sig":"MEQCIB+6mX8z3gKqjWGoZ2KhFUGRni1N1RHNz6nflLdnZL3yAiAzSBWXRk2XJHwTP4X3+eisFk792AaVw/sGZIhj/MaRmQ==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":56907},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=18"},"gitHead":"c36bbf341b6e6e25d0bacd270c4eb6d89f94b86e","scripts":{"dev":"tsx src/index.ts","build":"tsc","prepublishOnly":"npm run build"},"_npmUser":{"name":"alexanderzzlatkov","email":"alexander.z.zlatkov@gmail.com"},"repository":{"url":"git+https://github.com/zlatkov/skilleval.git","type":"git"},"_npmVersion":"11.6.2","description":"Evaluate how well AI models understand Agent Skills (SKILL.md files)","directories":{},"_nodeVersion":"24.13.0","dependencies":{"ai":"^4.0.0","chalk":"^5.4.0","dotenv":"^17.3.1","commander":"^13.0.0","gray-matter":"^4.0.3","@ai-sdk/google":"^1.0.0","@ai-sdk/openai":"^1.0.0","@ai-sdk/anthropic":"^1.0.0"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.0.0","typescript":"^5.7.0","@types/node":"^22.0.0"},"_npmOperationalInternal":{"tmp":"tmp/skilleval_0.1.0_1774206773334_0.22930897770670455","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@alexanderzzlatkov/skilleval","version":"0.1.1","keywords":["ai","agent","skills","eval","llm","claude-code","openrouter","skilleval"],"author":{"url":"https://github.com/zlatkov","name":"Alexander Zlatkov"},"license":"MIT","_id":"@alexanderzzlatkov/skilleval@0.1.1","maintainers":[{"name":"alexanderzzlatkov","email":"alexander.z.zlatkov@gmail.com"}],"homepage":"https://github.com/zlatkov/skilleval#readme","bugs":{"url":"https://github.com/zlatkov/skilleval/issues"},"bin":{"skilleval":"dist/index.js"},"dist":{"shasum":"2d04b14ca79668a6939e4f5a2eb90b97a38757c5","tarball":"https://registry.npmjs.org/@alexanderzzlatkov/skilleval/-/skilleval-0.1.1.tgz","fileCount":21,"integrity":"sha512-P/g3wb/vuqgQOCYPtrCSYv3enrz70/h2DmVOTc8aR+/UMBT1ecn0D+mIhZAmcFPlzFCNjitS+XNuimvjfG8rIg==","signatures":[{"sig":"MEQCIFBMJpAC4CNvs98xmHfo53Avj3z4osvjoxm116sHu1fBAiAI5Y4g3N9hx7RDwl7+P+2cFmvvtVBX8erSEB5OZm10Ww==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":57107},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=18"},"gitHead":"1653f6646799f60596f5695f236d7556fe36a8e2","scripts":{"dev":"tsx src/index.ts","build":"tsc","prepublishOnly":"npm run build"},"_npmUser":{"name":"alexanderzzlatkov","email":"alexander.z.zlatkov@gmail.com"},"repository":{"url":"git+https://github.com/zlatkov/skilleval.git","type":"git"},"_npmVersion":"11.6.2","description":"Evaluate how well AI models understand Agent Skills (SKILL.md files)","directories":{},"_nodeVersion":"24.13.0","dependencies":{"ai":"^4.0.0","chalk":"^5.4.0","dotenv":"^17.3.1","commander":"^13.0.0","gray-matter":"^4.0.3","@ai-sdk/google":"^1.0.0","@ai-sdk/openai":"^1.0.0","@ai-sdk/anthropic":"^1.0.0"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.0.0","typescript":"^5.7.0","@types/node":"^22.0.0"},"_npmOperationalInternal":{"tmp":"tmp/skilleval_0.1.1_1774208333793_0.890342021766785","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@alexanderzzlatkov/skilleval","version":"0.2.0","keywords":["ai","agent","skills","eval","llm","claude-code","openrouter","skilleval"],"author":{"url":"https://github.com/zlatkov","name":"Alexander Zlatkov"},"license":"MIT","_id":"@alexanderzzlatkov/skilleval@0.2.0","maintainers":[{"name":"alexanderzzlatkov","email":"alexander.z.zlatkov@gmail.com"}],"homepage":"https://github.com/zlatkov/skilleval#readme","bugs":{"url":"https://github.com/zlatkov/skilleval/issues"},"bin":{"skilleval":"dist/index.js"},"dist":{"shasum":"74a4dc25df15406bb5fd51cf28084176cbc997a8","tarball":"https://registry.npmjs.org/@alexanderzzlatkov/skilleval/-/skilleval-0.2.0.tgz","fileCount":25,"integrity":"sha512-XhLJDW/HRW6CN3c8geUKCUtR3ov5CYiKdjnudfh03O1upqiPbqdfc1gpbhriC6aX8p5xuCA840XbvCQBAcjxAw==","signatures":[{"sig":"MEQCIFw5SfiJ4xOGmlGtk2QpeyfF02JJde7NX3V690fCOlOHAiBhs4Pg5yRSbdOtXnXjuc+ecSXIGH4IwYOS/p+ujSas7A==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":89973},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=18"},"gitHead":"3be97b896eb13d8530d4967fc29b1eff12a296c6","scripts":{"dev":"tsx src/index.ts","test":"vitest run","build":"tsc","prepublishOnly":"npm run build"},"_npmUser":{"name":"alexanderzzlatkov","email":"alexander.z.zlatkov@gmail.com"},"repository":{"url":"git+https://github.com/zlatkov/skilleval.git","type":"git"},"_npmVersion":"11.6.2","description":"Evaluate how well AI models understand Agent Skills (SKILL.md files)","directories":{},"_nodeVersion":"24.13.0","dependencies":{"ai":"^4.0.0","chalk":"^5.4.0","dotenv":"^17.3.1","commander":"^13.0.0","gray-matter":"^4.0.3","@ai-sdk/azure":"^1.3.25","@ai-sdk/google":"^1.0.0","@ai-sdk/openai":"^1.0.0","@ai-sdk/anthropic":"^1.0.0"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.0.0","vitest":"^4.1.0","typescript":"^5.7.0","@types/node":"^22.0.0"},"_npmOperationalInternal":{"tmp":"tmp/skilleval_0.2.0_1774389533515_0.5216837702933823","host":"s3://npm-registry-packages-npm-production"}},"0.2.1":{"name":"@alexanderzzlatkov/skilleval","version":"0.2.1","description":"Evaluate how well AI models understand Agent Skills (SKILL.md files)","keywords":["ai","agent","skills","eval","llm","claude-code","openrouter","skilleval"],"homepage":"https://github.com/zlatkov/skilleval#readme","bugs":{"url":"https://github.com/zlatkov/skilleval/issues"},"repository":{"type":"git","url":"git+https://github.com/zlatkov/skilleval.git"},"license":"MIT","author":{"name":"Alexander Zlatkov","url":"https://github.com/zlatkov"},"type":"module","main":"./dist/index.js","types":"./dist/index.d.ts","bin":{"skilleval":"dist/index.js"},"scripts":{"build":"tsc","dev":"tsx src/index.ts","test":"vitest run","prepublishOnly":"npm run build"},"dependencies":{"@ai-sdk/anthropic":"^1.0.0","@ai-sdk/azure":"^1.3.25","@ai-sdk/google":"^1.0.0","@ai-sdk/openai":"^1.0.0","ai":"^4.0.0","chalk":"^5.4.0","commander":"^13.0.0","dotenv":"^17.3.1","gray-matter":"^4.0.3"},"devDependencies":{"@types/node":"^22.0.0","tsx":"^4.0.0","typescript":"^5.7.0","vitest":"^4.1.0"},"engines":{"node":">=18"},"_id":"@alexanderzzlatkov/skilleval@0.2.1","gitHead":"ea4862acc08e1fdf8aaa17d5b25649314725568c","_nodeVersion":"20.20.1","_npmVersion":"10.8.2","dist":{"integrity":"sha512-oRwnlbu41HuWPY8ld1v5P+E5FagyzDzP9Ov3wWYrctE9jWR4eOhj5DsnSHR65vtvySo3qbCaNhxNMjMLbdxsAg==","shasum":"e433e7e11ded2ab329fdfc7dc55d8373dd965a16","tarball":"https://registry.npmjs.org/@alexanderzzlatkov/skilleval/-/skilleval-0.2.1.tgz","fileCount":25,"unpackedSize":89550,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIGcWB/CIL+PVMHJ7kYiGk1VJY0luKKW3IzmHeSW0mlIzAiAJyOHSTCLDjrcUjSwv5gbvxFVjYc5Azr1QI+kNgnJg8A=="}]},"_npmUser":{"name":"alexanderzzlatkov","email":"alexander.z.zlatkov@gmail.com"},"directories":{},"maintainers":[{"name":"alexanderzzlatkov","email":"alexander.z.zlatkov@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/skilleval_0.2.1_1774390117367_0.13783104815027558"},"_hasShrinkwrap":false}},"time":{"created":"2026-03-22T19:12:53.230Z","modified":"2026-03-24T22:08:37.607Z","0.1.0":"2026-03-22T19:12:53.480Z","0.1.1":"2026-03-22T19:38:53.927Z","0.2.0":"2026-03-24T21:58:53.656Z","0.2.1":"2026-03-24T22:08:37.494Z"},"bugs":{"url":"https://github.com/zlatkov/skilleval/issues"},"author":{"name":"Alexander Zlatkov","url":"https://github.com/zlatkov"},"license":"MIT","homepage":"https://github.com/zlatkov/skilleval#readme","keywords":["ai","agent","skills","eval","llm","claude-code","openrouter","skilleval"],"repository":{"type":"git","url":"git+https://github.com/zlatkov/skilleval.git"},"description":"Evaluate how well AI models understand Agent Skills (SKILL.md files)","maintainers":[{"name":"alexanderzzlatkov","email":"alexander.z.zlatkov@gmail.com"}],"readme":"# skilleval\n\nEvaluate how well AI models understand and respond to [Agent Skills](https://agentskills.io/home) (SKILL.md files).\n\nSkill authors write a SKILL.md and have zero idea whether it works on any model besides the one they tested with. `skilleval` fixes that  - it simulates how agents like [OpenClaw](https://openclaw.ai/) and Claude Code inject skills into prompts, following the [OpenSkills](https://github.com/numman-ali/openskills) specification, then tests whether various LLM models correctly trigger and follow the skill's instructions.\n\n### `skilleval ./SKILL.md` - evaluate a skill across multiple models\n\n```\nskilleval v0.1.0\nSkill: pdf-processing\nDescription: Extract text and tables from PDF files\nProvider: openrouter\nModels: 5\n\n┌───────────────────────────────────────────┬──────────────┬────────────────┬─────────┐\n│ Model                                     │ Trigger      │ Compliance     │ Overall │\n├───────────────────────────────────────────┼──────────────┼────────────────┼─────────┤\n│ qwen/qwen3-235b-a22b:free                 │ 10/10        │ 5/5 (92)       │ 98%     │\n│ meta-llama/llama-3.3-70b-instruct:free    │ 9/10         │ 4/5 (85)       │ 82%     │\n│ deepseek/deepseek-r1:free                 │ 9/10         │ 4/5 (80)       │ 81%     │\n│ google/gemma-3-27b-it:free                │ 8/10         │ 3/5 (70)       │ 72%     │\n│ mistralai/mistral-small-3.1-24b:free      │ 7/10         │ 3/5 (65)       │ 66%     │\n└───────────────────────────────────────────┴──────────────┴────────────────┴─────────┘\n\nBest model: qwen/qwen3-235b-a22b:free (98%)\nWorst model: mistralai/mistral-small-3.1-24b:free (66%)\n```\n\n### `skilleval ./skills/ --graph` - visualise dependencies between skills\n\n```\nSkill Dependency Graph\n════════════════════════════════════════\n\n  ◉ orchestrator\n    ├──▶ fetcher (name: \"fetcher\")\n    ├──▶ parser (name: \"parser\")\n    └──▶ formatter (name: \"formatter\")\n  ◉ fetcher\n    ├──▶ parser (name: \"parser\")\n    └──▶ formatter (name: \"formatter\")\n  ◉ parser (depended on by 3)\n  ◉ formatter\n    └──▶ parser (name: \"parser\")\n\n────────────────────────────────────────\n  Nodes: 4 skills\n  Edges: 6 dependencies\n\n  Adjacency Matrix:\n                    1   2   3   4\n  1. orchestrator   ·   ●   ●   ●\n  2. fetcher        ·   ·   ●   ●\n  3. parser         ·   ·   ·   ·\n  4. formatter      ·   ·   ●   ·\n  ● = depends on\n```\n\n## Prerequisites\n\nYou need an API key for at least one supported provider. The default provider is [OpenRouter](https://openrouter.ai)  - create a free API key at [openrouter.ai/keys](https://openrouter.ai/keys).\n\nWhen using OpenRouter as your provider, the same key is used for test models, prompt generation, and evaluation judging (using free models by default). When using a different provider (Anthropic, OpenAI, Google, Azure), you can either:\n- Specify your own `--generator-model` and `--judge-model` to use models from that same provider\n- Or set `OPENROUTER_API_KEY` to use the default free OpenRouter models for generation and judging\n\n**Free model limitations:** OpenRouter's free models (those ending in `:free`) are subject to upstream rate limits and may be temporarily unavailable. If you encounter rate limit errors, you can:\n- Wait and retry  - free model availability fluctuates\n- Use paid models instead (remove the `:free` suffix, e.g. `meta-llama/llama-3.3-70b-instruct`)\n- Provide your own test prompts with `--prompts` to skip the generator model entirely\n\n## Installation\n\n```bash\nnpm install -g @alexanderzzlatkov/skilleval\n```\n\nOr use directly with `npx`:\n\n```bash\nnpx @alexanderzzlatkov/skilleval ./my-skill/SKILL.md\n```\n\n## Quick Start\n\n```bash\n# Set your OpenRouter API key\nexport OPENROUTER_API_KEY=sk-or-...\n\n# Evaluate a local skill\nnpx @alexanderzzlatkov/skilleval ./my-skill/SKILL.md\n\n# Evaluate a skill from a GitHub repo (like skills.sh)\nnpx skilleval owner/repo\n\n# Evaluate a specific skill within a repo\nnpx skilleval owner/repo --skill skill-name\n\n# Evaluate from a GitHub URL\nnpx skilleval https://github.com/user/repo/blob/main/skills/my-skill/SKILL.md\n\n# Evaluate ALL skills in a folder (local or GitHub)\nnpx skilleval ./skills/\nnpx skilleval https://github.com/user/repo/tree/main/skills\n```\n\n## How It Works\n\n`skilleval` follows the [OpenSkills](https://github.com/numman-ali/openskills) specification  - a universal skills format based on Anthropic's SKILL.md system. It's compatible with skills built for agents like Claude Code, [OpenClaw](https://openclaw.ai/), and any agent that uses the SKILL.md format. It simulates how these agents inject skills into system prompts using `<available_skills>` XML blocks  - the same format used in production. This means the evaluation reflects real-world skill behavior, not synthetic benchmarks.\n\n1. **Parse**  - Reads the SKILL.md, extracts name, description, and instructions. If the input is a folder (local or GitHub), it recursively scans for all SKILL.md files.\n2. **Build context**  - The model is presented as a helpful AI agent with access to multiple skills. The system prompt uses `<available_skills>` XML injection where your skill is mixed in with 3 fake distractor skills (e.g. \"git-commit-helper\", \"api-documentation\", \"test-generator\"). When evaluating a folder of skills, the other real skills found in the folder are also included as distractors alongside the dummy ones  - making the trigger test more realistic.\n3. **Generate test prompts**  - A generator model creates 5 positive prompts (should trigger) and 5 negative prompts (should not), per skill.\n4. **Run trigger tests**  - Sends each prompt to each target model with the skill-injected system prompt.\n5. **Evaluate**  - A judge model assesses trigger accuracy and, for correctly triggered prompts, runs a compliance test against the full skill instructions. If the skill references tools (e.g. `WebFetch`, `BraveSearch`, `Read`, `Write`, `Edit`, `Bash`, `Grep`, `Glob`), `skilleval` automatically provides mock tool definitions so the model can make real structured tool calls instead of fabricating results in text. The judge evaluates whether the model called the right tools with the right parameters, not the quality of the mock results.\n6. **Report**  - Prints a compatibility matrix to the terminal. In batch mode, each skill gets its own table plus a combined summary.\n\nSee [AGENTS.md](./AGENTS.md) for detailed pipeline internals.\n\n### Batch Mode\n\nWhen you point `skilleval` at a folder instead of a single file, it automatically discovers all SKILL.md files recursively and evaluates each one. The other skills found in the folder are injected as real distractors alongside the standard dummy skills, making the trigger test harder and more realistic.\n\n```bash\nnpx skilleval ./skills/\n```\n\n```\nskilleval v0.1.0 (batch mode)\nSkills found: 3\n  • ai-news — Fetches the latest AI news from multiple sources\n  • code-review — Reviews code for quality and best practices\n  • pdf-processor — Extract text and tables from PDF files\nProvider: openrouter\nModels: 5\n\n─── Skill: ai-news ───\n┌───────────────────────────────────────────┬──────────────┬────────────────┬─────────┐\n│ Model                                     │ Trigger      │ Compliance     │ Overall │\n├───────────────────────────────────────────┼──────────────┼────────────────┼─────────┤\n│ qwen/qwen3-235b-a22b:free                 │ 10/10        │ 5/5 (90)       │ 97%     │\n│ ...                                       │              │                │         │\n└───────────────────────────────────────────┴──────────────┴────────────────┴─────────┘\n\n─── Skill: code-review ───\n...\n\n=== Batch Summary ===\n┌──────────────────┬───────────────────────────────────┬──────────────┬────────────────┬─────────┐\n│ Skill            │ Model                             │ Trigger      │ Compliance     │ Overall │\n├──────────────────┼───────────────────────────────────┼──────────────┼────────────────┼─────────┤\n│ ai-news          │ qwen/qwen3-235b-a22b:free         │ 10/10        │ 5/5 (90)       │ 97%     │\n│                  │ meta-llama/llama-3.3-70b:free      │ 8/10         │ 4/5 (80)       │ 78%     │\n├──────────────────┼───────────────────────────────────┼──────────────┼────────────────┼─────────┤\n│ code-review      │ qwen/qwen3-235b-a22b:free         │ 9/10         │ 5/5 (85)       │ 92%     │\n│                  │ ...                               │              │                │         │\n└──────────────────┴───────────────────────────────────┴──────────────┴────────────────┴─────────┘\n\nAverage scores per skill:\n  ai-news: 85%\n  code-review: 78%\n  pdf-processor: 91%\n```\n\nSupported folder sources:\n- Local directories: `./skills/`, `/path/to/skills`\n- GitHub tree URLs: `https://github.com/user/repo/tree/main/skills`\n- GitHub repos: `user/repo` or `https://github.com/user/repo` (scans entire repo)\n\nDirectories like `node_modules`, `.git`, and `dist` are automatically skipped during local scans.\n\n### Dependency Graph\n\nUse `--graph` to visualise how skills reference each other. This works with any folder or repository source and requires no API key. References are detected by scanning each skill's content for mentions of other skill names, path references (e.g. `skills/other-skill/`), and frontmatter dependency fields (e.g. `dependencies: [other-skill]`).\n\n```bash\nnpx skilleval ./skills/ --graph\nnpx skilleval https://github.com/user/repo/tree/main/skills --graph\n```\n\nThe graph also detects and warns about circular dependencies. When running in batch evaluation mode (without `--graph`), the dependency graph is automatically shown if any dependencies are found between the scanned skills.\n\nUse `--graph --json` for machine-readable output.\n\n## Usage\n\n```\nskilleval <skill> [options]\n\nArguments:\n  skill                          Path, URL, or GitHub shorthand (owner/repo).\n                                 Can also be a folder path or GitHub tree URL\n                                 to batch-evaluate all SKILL.md files inside.\n\nOptions:\n  -p, --provider <provider>      Provider: openrouter, anthropic, openai, google, azure (default: openrouter)\n  -m, --models <models>          Comma-separated model IDs\n  -s, --skill <name>             Skill name within the repo (looks for skills/<name>/SKILL.md)\n  -k, --key <key>                API key (or use provider-specific env var)\n  --generator-model <model>      Model for test prompt generation (comma-separated for fallbacks)\n  --judge-model <model>          Model for evaluation judging (comma-separated for fallbacks)\n  --graph                        Show dependency graph between skills (folder/repo mode only)\n  --json                         Output results as JSON\n  --verbose                      Show detailed per-prompt results\n  -n, --count <number>           Number of positive+negative test prompts (default: 5, so 5+5=10 total)\n  --prompts <path>               Path to JSON file with custom test prompts\n  -V, --version                  Output the version number\n  -h, --help                     Display help\n```\n\n### Model Roles\n\n`skilleval` uses three types of models, each with a different role in the pipeline:\n\n| Role | Flag | Default | Description |\n|---|---|---|---|\n| **Test models** | `-m, --models` | 5 free OpenRouter models | The models being evaluated. These receive the skill-injected prompt and are scored on how well they trigger and follow the skill. |\n| **Generator models** | `--generator-model` | 3 free OpenRouter models (with fallback) | Generate test prompts (positive + negative) from the skill definition. Count configurable via `-n`. You can provide comma-separated model IDs for fallback. |\n| **Judge models** | `--judge-model` | 3 free OpenRouter models (with fallback) | Evaluate each test model's response  - did it correctly trigger the skill? Did it follow instructions? You can provide comma-separated model IDs for fallback. |\n\nWhen you specify custom `--generator-model` and `--judge-model`, they use the same provider as `--provider`. When omitted, they default to free OpenRouter models (which requires an `OPENROUTER_API_KEY` if your provider isn't OpenRouter).\n\n### Providers\n\nAll providers use the [Vercel AI SDK](https://ai-sdk.dev) under the hood.\n\n| Provider | Flag | Env Var | Notes |\n|---|---|---|---|\n| OpenRouter | `--provider openrouter` | `OPENROUTER_API_KEY` | Default. Access 300+ models including free ones. |\n| Anthropic | `--provider anthropic` | `ANTHROPIC_API_KEY` | Direct API access to Claude models. |\n| OpenAI | `--provider openai` | `OPENAI_API_KEY` | Direct API access to GPT models. |\n| Google | `--provider google` | `GOOGLE_GENERATIVE_AI_API_KEY` | Direct API access to Gemini models. |\n| Azure | `--provider azure` | `AZURE_API_KEY` + `AZURE_RESOURCE_NAME` | Azure AI Foundry. Model IDs are deployment names. |\n\n### Examples\n\n```bash\n# Test against default free OpenRouter models\nnpx skilleval ./SKILL.md\n\n# Test against specific models via OpenRouter\nnpx skilleval ./SKILL.md --models \"anthropic/claude-sonnet-4-20250514,openai/gpt-4o\"\n\n# Test directly against Anthropic\nnpx skilleval ./SKILL.md --provider anthropic --model claude-sonnet-4-20250514\n\n# Test against Azure AI Foundry (model ID = your deployment name)\nnpx skilleval ./SKILL.md --provider azure --models my-gpt4-deployment\n\n# Use a smarter judge model\nnpx skilleval ./SKILL.md --judge-model \"qwen/qwen3-235b-a22b:free\"\n\n# Provide your own test prompts\nnpx skilleval ./SKILL.md --prompts ./my-test-prompts.json\n\n# Machine-readable output\nnpx skilleval ./SKILL.md --json\n\n# Quick test with fewer prompts (1 positive + 1 negative)\nnpx skilleval ./SKILL.md -n 1\n\n# Detailed per-prompt breakdown\nnpx skilleval ./SKILL.md --verbose\n\n# Batch-evaluate all skills in a local folder\nnpx skilleval ./skills/\n\n# Batch-evaluate all skills in a GitHub folder\nnpx skilleval https://github.com/user/repo/tree/main/skills\n\n# Batch-evaluate an entire GitHub repo for SKILL.md files\nnpx skilleval user/repo\n\n# Visualise the dependency graph between skills (no API key needed)\nnpx skilleval ./skills/ --graph\nnpx skilleval https://github.com/user/repo/tree/main/skills --graph --json\n\n# Full example: evaluate Vercel's most popular skill on skills.sh\nnpx skilleval https://github.com/vercel-labs/skills --skill find-skills \\\n  --models anthropic/claude-opus-4.6 \\\n  --generator-model meta-llama/llama-3.3-70b-instruct:free \\\n  --judge-model anthropic/claude-sonnet-4.6 \\\n  -n 5 --verbose\n```\n\n### Custom Test Prompts\n\nCreate a JSON file with your own test prompts:\n\n```json\n[\n  {\"text\": \"Help me extract text from this PDF\", \"type\": \"positive\"},\n  {\"text\": \"Merge these two PDF files together\", \"type\": \"positive\"},\n  {\"text\": \"Convert this PDF to Word\", \"type\": \"positive\"},\n  {\"text\": \"Fill out this PDF form\", \"type\": \"positive\"},\n  {\"text\": \"Extract tables from the PDF report\", \"type\": \"positive\"},\n  {\"text\": \"What's the weather today?\", \"type\": \"negative\"},\n  {\"text\": \"Write me a Python script\", \"type\": \"negative\"},\n  {\"text\": \"Help me debug this CSS\", \"type\": \"negative\"},\n  {\"text\": \"Create a git commit message\", \"type\": \"negative\"},\n  {\"text\": \"Summarize this article for me\", \"type\": \"negative\"}\n]\n```\n\n## Scoring\n\nEach model is scored on two dimensions:\n\n- **Trigger accuracy** (50% of overall): Did the model correctly identify when to use the skill (positive prompts) and when to ignore it (negative prompts)?\n- **Compliance** (50% of overall): For positive prompts where the skill was triggered, did the model follow the skill's instructions? Split into pass/fail (30%) and quality score 0-100 (20%).\n\nExit code is `0` if all models score >= 50%, `1` otherwise  - useful for CI.\n\n## Development\n\n```bash\ngit clone https://github.com/zlatkov/skilleval.git\ncd skilleval\nnpm install\nnpm run dev -- ./path/to/SKILL.md\n```\n\n## License\n\nMIT\n","readmeFilename":"README.md"}