{"_id":"@dungle-scrubs/skillval","_rev":"4-f01beb0caa1d419435518a6651135f40","name":"@dungle-scrubs/skillval","dist-tags":{"latest":"0.4.0"},"versions":{"0.1.0":{"name":"@dungle-scrubs/skillval","version":"0.1.0","keywords":["agent-skills","agents","cli","codex","evaluation","skill-md","testing"],"license":"MIT","_id":"@dungle-scrubs/skillval@0.1.0","maintainers":[{"name":"dungle-scrubs-org","email":"kevinf@the7and8.com"}],"homepage":"https://github.com/dungle-scrubs/skillval#readme","bugs":{"url":"https://github.com/dungle-scrubs/skillval/issues"},"bin":{"skillval":"dist/cli.js"},"dist":{"shasum":"343323952b89c52c77c7babcfc15a6d7d4b5470d","tarball":"https://registry.npmjs.org/@dungle-scrubs/skillval/-/skillval-0.1.0.tgz","fileCount":7,"integrity":"sha512-7iG+3SP1K5+7VS3V8EuDIZd2sEKydAhIXU7IbG76cT2cgQviwTl/Eb//1bNK3TJ61YYAsM0flgWLwn5dbTg9KQ==","signatures":[{"sig":"MEYCIQDCMa1qcVpSw1uc1VUdOqHSSIEztsV30iqClrHUX+xOPQIhAJ5eKU4Q7knyaPhjpRsVWDCo/PJG+0N4QgViRIVJk9/3","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":126635},"pnpm":{"overrides":{"esbuild":">=0.28.1"}},"type":"module","engines":{"node":">=22"},"gitHead":"04a8a5e01763fd77f08ce560c4aebdb7ad03f195","private":false,"scripts":{"lint":"biome check .","test":"vitest run","build":"tsup src/cli.ts --clean --format esm --platform node --sourcemap --target node22","format":"biome format --write .","schema":"tsx scripts/generate-schemas.ts","prepare":"pnpm build","typecheck":"tsc --noEmit","schema:check":"tsx scripts/generate-schemas.ts --check"},"_npmUser":{"name":"dungle-scrubs-org","email":"kevinf@the7and8.com"},"repository":{"url":"git+https://github.com/dungle-scrubs/skillval.git","type":"git"},"_npmVersion":"11.17.0","description":"Deterministic evaluation for agent skills","directories":{},"_nodeVersion":"26.4.0","dependencies":{"yaml":"^2.9.0","typebox":"^1.3.6","commander":"^15.0.0","typescript":"^6.0.3","@types/node":"^26.0.1"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"packageManager":"pnpm@10.28.2","devDependencies":{"tsx":"^4.23.1","tsup":"^8.5.0","vitest":"^4.1.9","lefthook":"^2.1.10","@biomejs/biome":"2.5.5"},"_npmOperationalInternal":{"tmp":"tmp/skillval_0.1.0_1784706907169_0.05464604497120007","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@dungle-scrubs/skillval","version":"0.2.0","keywords":["agent-skills","agents","cli","codex","evaluation","skill-md","testing"],"license":"MIT","_id":"@dungle-scrubs/skillval@0.2.0","maintainers":[{"name":"dungle-scrubs-org","email":"kevinf@the7and8.com"}],"homepage":"https://github.com/dungle-scrubs/skillval#readme","bugs":{"url":"https://github.com/dungle-scrubs/skillval/issues"},"bin":{"skillval":"dist/cli.js"},"dist":{"shasum":"1cd1c9e6ba9ea078d2687c2e0197ce6f2d297797","tarball":"https://registry.npmjs.org/@dungle-scrubs/skillval/-/skillval-0.2.0.tgz","fileCount":7,"integrity":"sha512-5KOu+3Hg6ZGOXFuf/jczP3RQlXElk5o8Y/VXSNlzU3pXlsWwB1P30b+EnNW7NPkRYXi0+b6HcS93Sm0jNO2xxg==","signatures":[{"sig":"MEUCIQCguC/ama5r4c9VF71PggJEqvZIKWRoZR0y5+FNPemhFgIgVQ2TIU+RcbMI/JplYyk4dlG/+n5qndA89PJR3W+4oL8=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@dungle-scrubs%2fskillval@0.2.0","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":126635},"pnpm":{"overrides":{"esbuild":">=0.28.1"}},"type":"module","engines":{"node":">=22"},"gitHead":"a17f54a8c7bbd10f6b27f66a94ea44945e477bdb","private":false,"scripts":{"lint":"biome check .","test":"vitest run","build":"tsup src/cli.ts --clean --format esm --platform node --sourcemap --target node22","format":"biome format --write .","schema":"tsx scripts/generate-schemas.ts","prepare":"pnpm build","typecheck":"tsc --noEmit","schema:check":"tsx scripts/generate-schemas.ts --check"},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:81bf87b2-120c-4cb8-91f4-4135f72dad70"}},"repository":{"url":"git+https://github.com/dungle-scrubs/skillval.git","type":"git"},"_npmVersion":"12.0.1","description":"Deterministic evaluation for agent skills","directories":{},"_nodeVersion":"22.23.1","dependencies":{"yaml":"^2.9.0","typebox":"^1.3.6","commander":"^15.0.0","typescript":"^6.0.3","@types/node":"^26.0.1"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"packageManager":"pnpm@10.28.2","devDependencies":{"tsx":"^4.23.1","tsup":"^8.5.0","vitest":"^4.1.9","lefthook":"^2.1.10","@biomejs/biome":"2.5.5"},"_npmOperationalInternal":{"tmp":"tmp/skillval_0.2.0_1784708550014_0.939295820167263","host":"s3://npm-registry-packages-npm-production"}},"0.3.0":{"name":"@dungle-scrubs/skillval","version":"0.3.0","keywords":["agent-skills","agents","cli","codex","evaluation","skill-md","testing"],"license":"MIT","_id":"@dungle-scrubs/skillval@0.3.0","maintainers":[{"name":"dungle-scrubs-org","email":"kevinf@the7and8.com"}],"homepage":"https://github.com/dungle-scrubs/skillval#readme","bugs":{"url":"https://github.com/dungle-scrubs/skillval/issues"},"bin":{"skillval":"dist/cli.js"},"dist":{"shasum":"7ed10b814b99fd93e33710b7fa0b38ee6dc8e821","tarball":"https://registry.npmjs.org/@dungle-scrubs/skillval/-/skillval-0.3.0.tgz","fileCount":7,"integrity":"sha512-QxMUKUEXK5Jic9ZLOMtlpUK0LulOi4MdpRQOT+xF8wEpw7HPSnHsbuZElwY8M56WfI8QlEehTClLlC/I0mDjIw==","signatures":[{"sig":"MEUCIER4vTmH+6ODT1ZtOV4xRZP25p104emqpV3x6+xE5+9TAiEAgC36xLk139llvexBYD8gOBV1r8fEaekFVh0qsEFYt04=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@dungle-scrubs%2fskillval@0.3.0","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":199639},"pnpm":{"overrides":{"esbuild":">=0.28.1"}},"type":"module","engines":{"node":">=22"},"gitHead":"ab8e214f237cd8edccb998e24484a110b468408b","private":false,"scripts":{"lint":"biome check .","test":"vitest run","build":"tsup src/cli.ts --clean --format esm --platform node --sourcemap --target node22","format":"biome format --write .","schema":"tsx scripts/generate-schemas.ts","prepare":"pnpm build","typecheck":"tsc --noEmit","schema:check":"tsx scripts/generate-schemas.ts --check"},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:81bf87b2-120c-4cb8-91f4-4135f72dad70"}},"repository":{"url":"git+https://github.com/dungle-scrubs/skillval.git","type":"git"},"_npmVersion":"12.0.1","description":"Deterministic evaluation for agent skills","directories":{},"_nodeVersion":"22.23.1","dependencies":{"ajv":"^8.20.0","yaml":"^2.9.0","typebox":"^1.3.6","commander":"^15.0.0","typescript":"^7.0.2","@types/node":"^22.0.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"packageManager":"pnpm@10.28.2","devDependencies":{"tsx":"^4.23.1","tsup":"^8.5.0","vitest":"^4.1.9","lefthook":"^2.1.10","@biomejs/biome":"2.5.5"},"_npmOperationalInternal":{"tmp":"tmp/skillval_0.3.0_1784776519295_0.8897216003630228","host":"s3://npm-registry-packages-npm-production"}},"0.4.0":{"name":"@dungle-scrubs/skillval","version":"0.4.0","description":"Deterministic evaluation for agent skills","keywords":["agent-skills","agents","cli","codex","evaluation","skill-md","testing"],"type":"module","bin":{"skillval":"dist/cli.js"},"scripts":{"build":"pnpm build:ui && tsup src/cli.ts --clean --format esm --platform node --sourcemap --target node22","format":"biome format --write .","lint":"biome check .","prepare":"pnpm build","schema":"tsx scripts/generate-schemas.ts","schema:check":"tsx scripts/generate-schemas.ts --check","test":"vitest run","typecheck":"tsc --noEmit && tsc -p report-ui","build:ui":"vite build --config report-ui/vite.config.ts && node scripts/embed-report-ui.mjs","assets:check":"git diff --stat --exit-code -- src/generated/report-assets.ts"},"repository":{"type":"git","url":"git+https://github.com/dungle-scrubs/skillval.git"},"bugs":{"url":"https://github.com/dungle-scrubs/skillval/issues"},"homepage":"https://github.com/dungle-scrubs/skillval#readme","license":"MIT","private":false,"publishConfig":{"access":"public"},"engines":{"node":">=22"},"packageManager":"pnpm@10.28.2","dependencies":{"@ast-grep/napi":"^0.45.0","@types/node":"^22.0.0","ajv":"^8.20.0","commander":"^15.0.0","typebox":"^1.3.6","typescript":"^7.0.2","yaml":"^2.9.0"},"devDependencies":{"@biomejs/biome":"2.5.5","@radix-ui/react-collapsible":"^1.1.20","@radix-ui/react-dialog":"^1.1.23","@radix-ui/react-popover":"^1.1.23","@radix-ui/react-tooltip":"^1.2.16","@tailwindcss/vite":"^4.3.3","@testing-library/react":"^16.3.2","@testing-library/user-event":"^14.6.1","@types/jsdom":"^28.0.3","@types/react":"^19.2.17","@types/react-dom":"^19.2.3","@vitejs/plugin-react":"^6.0.4","class-variance-authority":"^0.7.1","clsx":"^2.1.1","jsdom":"^29.1.1","lefthook":"^2.1.10","lucide-react":"^1.26.0","react":"^19.2.8","react-dom":"^19.2.8","react-wrap-balancer":"^1.1.1","tailwind-merge":"^3.6.0","tailwindcss":"^4.3.3","tsup":"^8.5.0","tsx":"^4.23.1","tw-animate-css":"^1.4.0","vite":"^8.1.5","vitest":"^4.1.9"},"pnpm":{"overrides":{"esbuild":">=0.28.1"}},"gitHead":"07303e59666cb5e99e947c05e9dea764aeb676d9","_id":"@dungle-scrubs/skillval@0.4.0","_nodeVersion":"22.23.1","_npmVersion":"12.0.1","dist":{"integrity":"sha512-fIeUifXecRm8OYCXsCyhlbEfBTSJ7f3BoaKvrIrDSPoEqr0iRyFhI/lOPn0mKP9jVoHulGAIQg1durnc52nQEQ==","shasum":"e36160cfd98c0f598a1395134d7b7cad4d7feadf","tarball":"https://registry.npmjs.org/@dungle-scrubs/skillval/-/skillval-0.4.0.tgz","fileCount":10,"unpackedSize":1392105,"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@dungle-scrubs%2fskillval@0.4.0","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQD/i4w01O3feiBjwBCNk4e1iS4sisWsORURdgWq1amAowIgNqHr6rLNocoxyk9OoBefTHpjH61dG2M3SML4yTgGY4w="}]},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:81bf87b2-120c-4cb8-91f4-4135f72dad70"}},"directories":{},"maintainers":[{"name":"dungle-scrubs-org","email":"kevinf@the7and8.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/skillval_0.4.0_1785213515070_0.7236837181505473"},"_hasShrinkwrap":false}},"time":{"created":"2026-07-22T07:55:06.907Z","modified":"2026-07-28T04:38:35.539Z","0.1.0":"2026-07-22T07:55:07.329Z","0.2.0":"2026-07-22T08:22:30.156Z","0.3.0":"2026-07-23T03:15:19.514Z","0.4.0":"2026-07-28T04:38:35.210Z"},"bugs":{"url":"https://github.com/dungle-scrubs/skillval/issues"},"license":"MIT","homepage":"https://github.com/dungle-scrubs/skillval#readme","keywords":["agent-skills","agents","cli","codex","evaluation","skill-md","testing"],"repository":{"type":"git","url":"git+https://github.com/dungle-scrubs/skillval.git"},"description":"Deterministic evaluation for agent skills","maintainers":[{"name":"dungle-scrubs-org","email":"kevinf@the7and8.com"}],"readme":"# skillval\n\n`skillval` evaluates [Agent Skills](https://agentskills.io/) and agent instruction files\n(`CLAUDE.md`, `AGENTS.md`) with deterministic graders and no model judges. Each case can run a\n`solo` arm (the skill alone) and a `baseline` arm (no skill), measuring whether a skill changes\nagent behavior instead of merely checking whether the final answer looks acceptable. When the\nbaseline also passes, the rule is flagged as a no-op and a possible prune candidate. You can only\ntrust what you test.\n\nBoth arms run in a clean environment - your globally installed skills are hidden - so the only\nvariable is whether the skill under test is present. `solo` seeds just that skill; `baseline` seeds\nnothing.\n\n## Capability and preference rules\n\nA skill carries two kinds of rules. **Capability** rules teach a model something it does not yet\nreliably do. **Preference** rules express a choice - style, convention, house taste - a model would not reach on its own. Most skills mix both. The distinction matters because capabilities expire: as models are trained on the same information, a capability rule stops changing behavior and turns into dead weight. skillval finds those.\n\nEach case runs with the skill (`solo`) and again without it (`baseline`). Solo-pass with\nbaseline-fail means the rule is load-bearing. Solo-pass with baseline-pass means the model already\ndoes this on its own - the rule is a prune candidate. Preferences stay; stale capabilities go.\nCases can record which kind they exercise with the `type` field (`capability` or `preference`).\n\nBased on my own usage, skillval's long-run effect on a skill library is not shrinkage. Most rules\neither earn their keep or have not been disproven yet; what accumulates instead is a ledger of\nwhich rules still change behavior, on which models. Pruning happens rule by rule, model by model,\nas frontier models absorb what the skills teach - and until the evidence is in, the rule stays.\n\n### Reading a result\n\n- **`solo` pass, `baseline` fail** - the skill is doing the work; it changed behavior. Load-bearing.\n- **`solo` pass, `baseline` pass** - the case passes with or without the skill; the model already\n  does this. A no-op and a prune candidate.\n- **`solo` fail** - the skill did not produce the required behavior. A failing case to investigate.\n- **`solo` infra** - every trial of the arm hit an infrastructure failure (agent output too large to\n  capture, or a timeout), so the arm was never graded. The case is reported **inconclusive**: not a\n  failure, not a pass, and never a no-op - a passing `baseline` beside an ungraded `solo` proves\n  nothing about the skill. Infrastructure arms are never cached, so a rerun grades them fresh.\n\n`should_trigger`, when set, is checked only on arms where the skill under test is present (`solo`),\nnever on `baseline`, where it is absent by design.\n\n## Group mode\n\nSolo mode measures a skill in isolation - the skill alone versus nothing. Group mode measures its\nmarginal effect **inside a set of other skills**, which is closer to how skills are used in\npractice, and it surfaces interference that isolation cannot see.\n\nDefine named loadouts in the configuration, then pass `--loadout <name>`:\n\n```yaml\n# config.yml\nloadouts:\n  everyday: [commit-style, naming, imports]\n```\n\n```sh\nskillval run typescript-style --loadout everyday\n```\n\nGroup mode runs three arms per case (ignoring the case's `arms` field, since the verdict needs all\nthree): `solo` (the target alone), `group` (the loadout plus the target), and `peers` (the loadout\nminus the target). Every arm runs clean, differing only by its seeded set. Loadout members must be\ndiscovered skills; they only need a `SKILL.md`, not a `skillval.yml`. If a member name matches more\nthan one discovered skill (the same name under two roots), the first match wins and the run prints a\n`warning:` line naming what was used and what was shadowed. The verdict per case:\n\n| Arms | Verdict |\n| --- | --- |\n| `solo` pass, `group` **fail**, `peers` pass | **interferes with your other skills** |\n| `group` pass, `peers` fail | **works and is needed here** (load-bearing) |\n| `group` pass, `peers` pass | **redundant** - another skill already does it |\n| `solo` fail, `peers` pass | **not needed at all** |\n\nThe raw three arm results stay in the report; the verdict is a derived `loadout` block. Any other\ncombination is reported as `inconclusive`, and so is any combination in which a consulted arm was\nnever graded (an infrastructure failure) - an ungraded arm supports no verdict. `should_trigger` is checked on `solo` and `group` (the\ntarget is present) but never on `peers`. The run summary calls out interference, the way it calls\nout no-ops.\n\nInterference is only attributed to the target when the target's presence is what breaks the case:\n`solo` passes, `group` fails, and `peers` (the loadout minus the target) still passes. If `peers`\nalso fails, the loadout breaks the case without the target at all, so the finding is about the other\nskills, not this one - that is left `inconclusive` rather than blamed on the target. (A pure\ntrigger-only case has no behavioral check on `peers`, so `solo`-vs-`group` still isolates the target\nand interference stands.)\n\nThe `redundant` / `load-bearing` / `not needed` verdicts compare `group` against `peers`, so they\nneed an assertion that grades behavior on the `peers` arm (a `must_match`, grader, and so on). A\npure trigger-only case - `should_trigger` and nothing else - has no such check on `peers` (the\ntrigger check is target-specific), so it is reported `inconclusive` unless it shows interference.\n\n## Cost preview\n\nGroup mode multiplies trials: three arms per case, each up to five trials on disagreement, across\nevery case. Before spending, run `--dry-run` to see exactly what a run would cost against the current\ncache - it resolves the same skills, loadout, executor identity, and cache a real run would, then\nreports the trials that would run without spawning a single one:\n\n```console\n$ skillval run --loadout everyday --dry-run\nexecutor: codex codex-cli 0.145.0 (model gpt-5.6-sol, thinking medium, invocation detection heuristic)\ndry run: no trials will be spawned\ncommit-style:\n  wraps-body [solo] run (3-5 trials)\n  wraps-body [group] cached\n  wraps-body [peers] run (3-5 trials)\nplan: 2 arm(s) to run, 1 cached, 0 reused\ntrials to run: 6 (up to 10 if arms escalate on disagreement)\n```\n\nEach arm is `cached` (a hit, no trials), `reused from solo` (a group arm with no peers to add), or\n`run` with its trial count - the minimum it will spend, and the ceiling if trials disagree and\nescalate to five. `--dry-run` writes no report and applies no execution gates, so it previews cost\neven for a suite a real run would refuse. `--json` returns the full plan.\n## Instruction files\n\nAn agent instruction file (`CLAUDE.md`, `AGENTS.md`) is a bag of rules nobody re-tests. skillval\naudits it the same way it audits a skill, one rule at a time, using **single-rule ablation**: hold\nthe whole file fixed and remove exactly one rule. That is group mode pointed inward, where the rest\nof the file is the rule's loadout:\n\n| Arm | Content |\n| --- | --- |\n| `solo` | the rule alone |\n| `group` | the whole file (the rule plus its sibling rules) |\n| `peers` | the whole file minus that one rule |\n\nThe verdict table is the same as [group mode](#group-mode), read per rule: `group` pass with `peers`\nfail means the rule is load-bearing; `group` pass with `peers` **pass** means another rule in the\nsame file already covers it, so the rule is **redundant** and a deletion candidate. That intra-file\nredundancy is the finding whole-file testing cannot produce.\n\nEach instruction file carries a sibling `skillval.yml` with `target: instructions`, and each case\nnames the rule it ablates with `rule_text` - the verbatim span, matched exactly (authored\nindentation is part of the address). A span that is missing or appears more than once is a\nvalidation error rather than a silent mis-ablation.\n\n```yaml\n# AGENTS.md sits beside this file\ntarget: instructions\nclass: preference\ncases:\n  - id: greeting-token\n    mode: generation\n    rule: greeting-token\n    rule_text: \"- When asked to create a greeting file, its first line must be exactly HELLO.\"\n    prompt: Create greeting.txt containing a short greeting.\n    assert:\n      must_match: [\"HELLO\"]\n```\n\n`should_trigger` is a validation error on instruction targets: instruction files are ambient, so\nnothing is ever \"invoked\".\n\n### Which executor sees which file\n\nInstruction files are read natively, and the executors differ. Measured behavior:\n\n| Executor | Reads ambiently |\n| --- | --- |\n| `codex` | `AGENTS.md` |\n| `claude` | `CLAUDE.md` (a bare `AGENTS.md` is **not** read; verified) |\n| `pi` | `AGENTS.md`, then `CLAUDE.md` |\n\nA rule only reaches an executor that reads the file it lives in. A rule in a `CLAUDE.md` is\ntherefore **not applicable** to codex, and is reported `n/a` - never a pass and never a fail, so an\naudit never claims a result for instructions that executor would not see in the real project.\n\nKnown v1 limitation: a rule that reaches claude only through a `CLAUDE.md` `@import` of `AGENTS.md`\nis `n/a` for claude, because v1 ablates the file the executor reads natively. Cross-file import\nablation is planned.\n\n### Reading the report\n\nThe report is a remediation manifest, not a pass/fail dump. Each finding carries the file, the\nverbatim span, the verdict, and an `action` an agent can execute directly:\n\n| Verdict | Action | Meaning |\n| --- | --- | --- |\n| load-bearing | `keep` | the rule is doing work |\n| redundant | `delete` | another rule already covers it |\n| prune | `delete` | not needed at all |\n| interference | `review` | the rule fights the others |\n| inconclusive | `investigate` | see the raw arm results |\n\nEvery finding keeps its raw arm results, so the reasoning stays inspectable.\n\n### The HTML report\n\nAfter each run skillval writes a self-contained HTML report beside the JSON one and opens it. It\nleads with **what to change**: every rule flagged `delete` or `review`, with the exact span to act\non, why it was flagged (tied to the arm that proved it), and the arm evidence beside the\nrecommendation. The page is a single file embedding its own React app and stylesheet - it\nreferences nothing on the network, works from `file://`, and follows the system light/dark theme.\nReport content is rendered by React (escaped by construction), never templated into markup.\n\nReports carry a two-tab nav (Latest run | Coverage): `latest.html` is a stable alias refreshed\nafter every HTML-enabled run (with `htmlReport: false` it is left untouched and may lag or not\nexist), while the hash-named reports remain the immutable archive - an archived page labels\nitself \"This run (archived)\" and links to the alias rather than claiming to be the latest. A\nskills **what to change** panel derives an action per case: a no-op maps to *prune candidate*\n(surface, verify cross-model, then decide) and loadout redundancy to *review* - deliberately\nsofter than the instruction mapping, where a redundant rule is a deletable line. Every\nload-bearing term is a dotted quick-view that opens a right-side explainer, and each page opens\nwith a collapsed 20-second primer - the reports assume a reader who has forgotten how skillval\nworks and re-teach at point of use.\n\nThe report is written but not opened: pass `--open` to launch it, or open the printed path. A\nsweep is many runs, and hijacking the browser once per run makes batch work unusable. Turn the\nHTML off entirely with `htmlReport: false` in the configuration - useful in CI or scripted runs.\nFailing to open a browser is never a run failure; the path is always printed.\n\n## Install\n\n```sh\npnpm add -g @dungle-scrubs/skillval\n```\n\nNode.js 22 or newer is required. The Codex CLI must be installed and authenticated for evaluation\nruns. Discovery with `skillval list` does not invoke Codex.\n\n## Quickstart\n\nCreate `~/.config/skillval/config.yml`:\n\n```yaml\nroots:\n  - ~/dev/agent-skills\nexecutor: codex\n```\n\nPin the model and effort too, so a verdict is attributable to a named identity\nrather than to whatever your agent CLI happens to be configured for that day:\n\n```yaml\nexecutor: claude\nmodel: sonnet\neffort: low\n```\n\nPrecedence is `--model` / `--effort` flag > config > the agent CLI's own default.\nBoth fields are optional; unpinned, the executor's default applies and is\nrecorded. Leaving them unset is how a study can silently split across two ledger\ncolumns when you switch models for unrelated reasons.\n\nGiven `~/dev/agent-skills/typescript-style/SKILL.md`, add\n`~/dev/agent-skills/typescript-style/skillval.yml`:\n\n```yaml\nskill: typescript-style\nclass: preference\ncases:\n  - id: prefer-const-object\n    mode: generation\n    type: preference\n    rule: enums-as-const\n    arms: [solo, baseline]\n    prompt: >-\n      Create sizes.ts with a fixed set of small, medium, and large values.\n    assert:\n      must_match: [\"as const\"]\n      must_not_match: [\"\\\\benum\\\\s\"]\n      graders: [tsc]\n    trials: 1\n```\n\nRun the case:\n\n```console\n$ skillval run typescript-style\ntypescript-style (preference, e8342aa91a17):\n  prefer-const-object [skill] ...\n  prefer-const-object [skill] pass\n  prefer-const-object [baseline] ...\n  prefer-const-object [baseline] FAIL\nreport: /Users/example/.local/state/skillval/reports/0f47c8d4....json\nall cases passed\n```\n\nRun every discovered skill that has a `skillval.yml` by omitting the skill names. Use `--case <id>`\nto select one case, `--no-cache` to ignore cached arm results, `--skip-baseline` to omit baseline\narms, `--dry-run` to preview the trials a run would cost without spawning any (see\n[Cost preview](#cost-preview)), and `--json` for the complete report. The command exits with status 1\nwhen any selected case fails, and with status 2 when nothing failed but at least one case was\ninconclusive (its deciding arm hit only infrastructure failures and was never graded) - not a\ncontent failure, but not a clean pass a script should act on either.\n\nUse `--model <model>` and `--effort <level>` to pin the executor's model and effort for the run,\nso you can evaluate one skill under, for example, `--model sonnet --effort medium`. Both pass\nthrough to the configured executor and are recorded in the report and the cache identity, so runs\nat different levels are cached and compared separately. Effort levels are executor-specific and\nvalidated before the run: `codex` accepts `none, minimal, low, medium, high, xhigh, max`; `claude`\naccepts `low, medium, high, xhigh, max`; `pi` accepts `off, minimal, low, medium, high, xhigh`.\nModel support for a given effort is a subset of these, enforced by the harness itself.\n\n## Configuration\n\nThe configuration follows the [configuration JSON Schema](schemas/config.schema.json):\n\n```yaml\nroots:\n  - ~/dev/skills/skills/standards\n  - $HOME/dev/shared/skills/backend\nexecutor: codex\nhtmlReport: true\n```\n\n`htmlReport` (optional, enabled when omitted) writes a self-contained\n[HTML report](#the-html-report) beside the JSON one after each run and opens it. Set it to `false`\nfor headless or CI runs.\n\n`roots` contains directories whose immediate children have the form `<skill>/SKILL.md`. Both `~`\nand `$HOME` are expanded. `executor` selects the trial adapter: `codex`, `claude`, or `pi`. Missing roots are skipped during `run`; `list` returns them in\n`missingRoots` with JSON output and prints each as `missing root: <path>` in human output.\n\n`exclude` (optional) omits skills from discovery by **name** - useful for third-party skills\ninstalled under a root you also own, so you cannot simply drop the root. Patterns match the skill\nname with `*` and `?` glob wildcards; an excluded skill is never discovered, so it is absent from\n`list`, cannot be a `run` target, and is not seeded as a loadout member. Instruction targets are\naddressed by project path, not skill name, so `exclude` does not affect them.\n\n```yaml\nexclude:\n  - impeccable      # a vendored skill that is not mine\n  - vendor-*        # everything from a third-party pack\n```\n\n`projects` (optional) contains **project trees** scanned recursively for\n[instruction files](#instruction-files) and project-scoped skills, each gated by a sibling\n`skillval.yml`:\n\n```yaml\nprojects:\n  - ~/dev/myapp\n```\n\nA scan finds `CLAUDE.md`/`AGENTS.md` at any depth plus skills under `.claude/skills/*` and\n`.agents/skills/*`, always skipping `.git` and `node_modules`. Targets are identified by tree\nposition (`myapp:.`, `myapp:packages/api`).\n\nA `projects` entry is **one project, pointed at deliberately**. A directory holding several\nindependent git repos (each with its own `.git`, often gitignored by the parent) is not supported:\nevaluate each real repo from its own root. Deep recursion inside one repo - internal packages with\ntheir own `AGENTS.md` - is the intended use.\n\n`loadouts` (optional) defines named skill sets for [group mode](#group-mode): a map from a loadout\nname to the discovered skill names it contains. Select one with `--loadout <name>`.\n\nConfiguration path precedence is:\n\n1. `--config <path>`\n2. `SKILLVAL_CONFIG`\n3. `$XDG_CONFIG_HOME/skillval/config.yml`\n4. `~/.config/skillval/config.yml`\n\nThere is no legacy `~/.skillval` lookup. State uses `$XDG_STATE_HOME/skillval`, or\n`~/.local/state/skillval` when `XDG_STATE_HOME` is unset:\n\n- `cache/` stores arm results.\n- `reports/` stores run reports named by a hash of the participating targets, their content hashes,\n  the executor identity, and the `--case` filter - results are executor-specific and\n  slice-specific, so running the same targets under a second executor, or a single case out of a\n  suite, writes a separate report instead of overwriting the first. Each report also\n  includes every participating skill's content hash and the executor's name, version, model,\n  thinking-level identity, and invocation-detection method.\n\n`skillval list` returns the skill name, configured root, class, case count, whether `skillval.yml`\nexists, and a `missing`, `invalid`, or `ready` status in JSON output. Invalid case files include a\nvalidation error. Discovery only requires `SKILL.md`; evaluation requires a valid `skillval.yml`.\n\n## Coverage matrix\n\n`skillval coverage` renders every ready skill's eval coverage as one self-contained HTML page\n(written to `reports/coverage.html` under the state directory and opened, replacing the previous\nrender - it is a view of the current suites, not a run artifact). Each case is classified onto a\ngrader rung: **trigger-only** (proves the skill loads, says nothing about what it changes),\n**regex** (lexical presence in output), or **execution** (deterministic proof of the artifact\nbeyond lexical matching - runtime behavior via `command_exit`, validation via `json_schema` or a\nregistered grader, or structural `ast` rules; a case with several graders counts on its strongest\nevidence). The page shows per-rung totals, a composition bar whose\nsegments carry hover tooltips explaining each rung, and a per-root matrix - skills sorted\nweakest-coverage-first - expandable to case-level graders, arms, and trials. Gap stats call out\nskills with zero behavioral cases, skills without a negative trigger case, and how many skills\ncompare against a baseline arm. The page shares the run report's two-tab nav, linking to\n`latest.html` and back. `--json` returns the full coverage report as data instead. This is\nthe mechanical half of the bundled skill's audit (its \"read what is graded\" step); the judgment\nhalf - what is worth writing next - stays with [the skill](#bundled-skill).\n\n## The ledger\n\nA report answers \"what happened in this run\". `skillval ledger` answers the question the suite\nexists for: **which rules still earn their keep, on which model, at which effort.** It reads every\nreport already on disk - no trials are spent - and renders one row per case, one column per\nexecutor identity (name/model/thinking, exactly the fields the cache keys on, so the columns are\nthe units whose verdicts are comparable).\n\n```console\n$ skillval ledger --transitions\ncase                                claude/sonnet/low  claude/sonnet/high  codex/gpt-5.6-sol/medium\nstandards-python/ty-over-mypy       ----               LOAD                noop\nobservability/boundary-tracing      noop               LOAD                LOAD\n```\n\n### Your profile decides what counts as dead weight\n\nWhether a rule is worth keeping depends on how *you* work. A rule that is load-bearing at low\nreasoning effort and a no-op at high is worth keeping if you live at low effort, and is context\ntax if you live at high. Name the identities you actually run:\n\n```yaml\n# config.yml\nprofile:\n  targets:\n    - claude/sonnet/low\n```\n\nThe ledger's `verdict` column then reads **keep** when a rule is load-bearing on any tier you run,\nand **PRUNE** only when it is a no-op across all of them; `--prune-candidates` filters to those.\nWith no profile, every identity on record counts.\n\nNote the half that surprises people: this also demotes rules that only earn their keep *above*\nyour tier. A rule that is load-bearing at high and a no-op at low is dead weight under a low-only\nprofile - correctly, since you never work where it helps. List every tier you actually use, not\njust your favourite one.\n\nThat second shape follows from what a verdict is: a comparison between two arms, where raising\neffort moves both. Usually the baseline **converges** on the rule's answer - the model reaches for\nit unaided once it thinks harder - so the rule is outgrown. In principle a baseline can also\n**diverge**: at low effort a model gives a short conventional answer that happens to match the\nhouse pick, and at high effort it deliberates, weighs the alternatives, and lands somewhere\ndefensible that is not your convention, making the rule *more* necessary as the model improves.\n\nBe sceptical of the second shape when you see it. Every candidate for it in the corpus this tool\nwas built against evaporated at `trials: 3` - each was either a single-roll coin flip or a\nmis-keyed assert. It is a shape the ledger can report, not one you should expect. Raise `trials`\nbefore believing it.\n\nA verdict of `not-invoked` or `inconclusive` is silence, not evidence: it can neither argue for\nkeeping a rule nor for pruning it, and a row with nothing but silence reads as\ninsufficient-evidence.\n\n`--transitions` shows only the rows whose verdict differs across identities, which is where the\ninformation is: a rule that is load-bearing at one tier and a no-op at another has a **scope**, not\na defect, and the matrix is what tells you which tiers still need it.\n\nTwo verdicts exist so that a run's non-results cannot masquerade as findings. `----` means the\nskill was never invoked - a floor on *loading* it, not a judgment about the rule, and a no-op\nrecorded below that floor means nothing because the `solo` arm never read the skill. `inco` means\nthe trial was never graded at all: an executor crash, a timeout, or a provider outage. Any failing\n`run` check is read this way, which also repairs history written before provider failures were\ntyped as infrastructure.\n\n## Trust model\n\nA `skillval.yml` is executable input, not passive configuration. Two fields run case-authored\nshell commands directly on the machine that grades the suite:\n\n- fixture `setup` commands, before the trial's agent runs;\n- `assert.command_exit`, at grading time.\n\nBoth run with a minimal environment - only `PATH` is inherited, and `HOME` points at a throwaway\ntrial directory - and are killed on timeout, but that is scoping, not a sandbox: nothing prevents\na command from reading or writing anything your user account can reach. Evaluating a skill\ntherefore means trusting its `skillval.yml` exactly as you would trust running its Makefile or\nnpm scripts.\n\nBecause of that, case-authored shell is **off by default**. A run refuses any selected case that\ncarries fixture `setup` commands or a `command_exit` grader, failing before any trial spawns with a\nmessage naming the skill, case, and surface. Pass `--allow-shell` to opt in once you have reviewed\nthe case file. Keeping it off by default means pointing skillval at a skill from a repository you do\nnot control never runs that skill's shell unless you explicitly allow it - the safe default for CI\nand for auditing third-party skills.\n\nThe agent trials themselves are a separate boundary, sandboxed per executor (see\n[Executors](#executors)): codex trials get an OS sandbox, claude trials get permission modes, and\npi generation trials have no sandbox at all and must be acknowledged with\n`--allow-unsandboxed-pi`.\n\nAn instruction file is inert Markdown - evaluating one executes nothing from the file itself - but\nits sibling `skillval.yml` carries the same executable fields as any other case file, so it is\ntrusted at exactly the same level.\n\n## Case files\n\nOnly a file named `skillval.yml` next to `SKILL.md` is recognized. There is no `evals.yml`\nfallback. The complete format is described by the\n[case-file JSON Schema](schemas/skillval.schema.json).\nThe published configuration and case-file schemas are generated from the same executable TypeBox\ncontracts used for runtime validation. Contributors can regenerate them with `pnpm schema` and\ncheck freshness with `pnpm schema:check`.\n\nTop-level fields:\n\n- `target`: `skill` (the default when omitted) or `instructions`. An `instructions` file sits beside\n  a `CLAUDE.md`/`AGENTS.md` and declares no `skill`; see [Instruction files](#instruction-files).\n- `skill`: the directory and skill name. Required for skill targets, and rejected on instruction\n  targets, whose identity comes from their position in the project tree.\n- `class`: `preference` or `capability`.\n- `cases`: an array of deterministic evaluation cases.\n- `fixture`: optional suite-wide workspace fixture applied to every case that does not declare\n  its own. See [Fixtures](#fixtures).\n\nCase fields:\n\n- `id`: unique case identifier.\n- `mode`: `trigger` grades the final agent message; `generation` grades files produced in the\n  temporary workspace.\n- `type`: optional `preference` or `capability` classification.\n- `rule`: optional stable rule identifier included in reports.\n- `rule_text`: the verbatim rule span this case ablates, required on instruction targets. It is\n  content-addressed and matched exactly - authored indentation is part of the address - and must\n  appear exactly once in the file, so a stale or ambiguous span is a clean error instead of a silent\n  mis-ablation. The `peers` arm is the file with exactly this span removed.\n- `should_trigger`: optional expected invocation verdict. It is checked only on arms where the skill under test is present (`solo`).\n- `arms`: `solo`, or `solo` and `baseline`. The default is `[solo]`.\n- `prompt`: the complete trial prompt.\n- `assert.must_match`: JavaScript regular expressions that must match, with the `m` flag.\n- `assert.must_not_match`: JavaScript regular expressions that must not match, with the `m` flag.\n- `assert.graders`: parameterless deterministic graders. `tsc` is supported for generation cases.\n  Unknown graders and graders used with an unsupported mode are validation errors.\n- `assert.json_schema`: validates a produced file against a JSON Schema (draft 2020-12), for\n  generation cases. Takes `file` (relative to the workspace) and `schema` (the JSON Schema, an\n  object or boolean). The file must exist inside the workspace, be a regular file, and parse as\n  JSON; a schema mismatch reports the failing instance path. Omit `$schema` or set it to 2020-12;\n  other declared dialects, an escaping `file` path, or a schema that does not compile are validation\n  errors.\n- `assert.ast`: parses a produced file (TypeScript, TSX, JavaScript, CSS, HTML by extension) and\n  grades its STRUCTURE with [ast-grep](https://ast-grep.github.io/) rule objects: every\n  `must_match` rule needs at least one match, any `must_not_match` match fails with the offending\n  line. This decides placement facts regex cannot see and execution cannot always separate - the\n  canonical case is the validation-vs-invariant lookalike, where an `assert(param > 0)` in a\n  constructor is input validation but a `this.`-referencing guard in an operation is an internal\n  invariant. Structural matching also cannot be satisfied by a comment. Pure parsing on the\n  grading machine - no shell runs, so `--allow-shell` is not required. In the coverage matrix an\n  ast case counts on the execution rung: deterministic proof of the artifact beyond lexical\n  presence.\n- `assert.command_exit`: runs a shell command in the workspace and passes when it exits with the\n  expected code, for generation cases. Takes `command` and optional `expect` (default `0`). The\n  command is case-authored arbitrary shell, the same trust level as fixture `setup`, and is off by\n  default: a case using it is refused unless the run passes `--allow-shell` (see\n  [Trust model](#trust-model)). It runs with a minimal environment and is killed after\n  120 seconds. This is the language-agnostic grader: run a\n  compiler, test runner, or validator over produced files in any language. Used in a non-generation\n  case it is a validation error.\n- `trials`: an integer from 1 through 5. Results use a strict majority. If configured trials\n  disagree, the arm escalates to 5 trials.\n- `fixture`: optional workspace fixture for this case. It replaces the suite-level `fixture`\n  entirely; `path` and `setup` never merge across levels.\n\nEvery trial must also contain a complete executor trace. For generation cases, regex assertions\nsee only produced files, prefixed with `=== filename ===`; prose cannot satisfy a file assertion.\nThe `tsc` grader injects a module package file when needed and a strict bundler-resolution\nTypeScript configuration, then runs the TypeScript installation shipped with `skillval`.\n\n## Fixtures\n\nBy default every trial starts in an empty temporary workspace. A fixture populates that workspace\nbefore the trial runs, for cases that need a realistic repository or document tree. A fixture has\ntwo fields, and at least one is required:\n\n- `path`: a directory relative to `skillval.yml`, copied recursively into the workspace before\n  the trial. `.git` and `node_modules` directories are never copied. The path must exist and be a\n  directory, and it may not contain symbolic links (create links with `setup` commands instead);\n  anything else is a validation error at load time.\n- `setup`: shell commands run sequentially inside the workspace after the copy, with a minimal\n  environment (`PATH` plus a throwaway `HOME`). These are case-authored arbitrary shell commands\n  executed on the grading machine and are off by default: a case whose fixture carries `setup` is\n  refused unless the run passes `--allow-shell` (see [Trust model](#trust-model)). A non-zero exit\n  fails the trial with a `fixture-setup` error before the agent runs; it is never a grading failure.\n  Each command's stdout and stderr are captured into the trial record.\n\nA suite-level `fixture` applies to every case; a case-level `fixture` replaces it entirely.\nFixture directory contents and setup commands are part of the arm cache identity, so editing a\nfixture file or a setup command invalidates cached results for the cases that use it.\n\nGeneration-mode regex assertions read every workspace file except `.git` and `node_modules`\ncontents and the injected `package.json`/`tsconfig.json`, so fixture files are graded alongside\nanything the agent produced. Graders access the workspace directly (`tsc` compiles what it finds\nthere). Write `must_match` patterns against the state you expect after the agent acts, not only\nagainst new files.\n\nNested `.git` directories inside a fixture are not supported. Express git state with `setup`\ncommands instead - this example stages a merge conflict for the agent to resolve:\n\n```yaml\nskill: resolve-conflicts\nclass: capability\ncases:\n  - id: merge-conflict\n    mode: generation\n    prompt: Resolve the merge conflict in notes.md, keeping both sections.\n    assert:\n      must_not_match: [\"^<{7} \", \"^={7}$\", \"^>{7} \"]\n    fixture:\n      path: fixtures/notes-repo\n      setup:\n        - git init -q -b main\n        - git config user.name fixture && git config user.email fixture@skillval.invalid\n        - git add -A && git commit -qm base\n        - git switch -qc feature\n        - printf 'feature section\\n' >> notes.md && git commit -qam feature\n        - git switch -q main\n        - printf 'main section\\n' >> notes.md && git commit -qam main\n        - git merge feature || true\n```\n\nThe final `|| true` matters: `git merge` exits non-zero on conflict, which would otherwise fail\nthe trial as a fixture-setup error - here the conflict is the point.\n\n## Executors\n\nExecutors are adapters with three responsibilities: report stable metadata for cache keys, prepare\nprovider-specific skill and environment state, and run one trial request to return a normalized\n`Trace`. The runner owns temporary workspace lifecycle, grading, caching, majority voting, and\nreports. Three adapters exist: `codex`, `claude`, and `pi`.\n\nThe Codex adapter runs:\n\n```text\ncodex exec --json --skip-git-repo-check --ephemeral -s <sandbox> -C <workspace> <prompt>\n```\n\nTrigger cases use a read-only sandbox. Generation cases use `workspace-write`. Codex has no\ndedicated skill-invocation event, so its adapter detects invocation when a **completed**\n`command_execution` command contains `<skill>/SKILL.md`. A started-but-unfinished command never\ncounts, and a failed *simple* command (a lone `cat` of a missing path) does not either - it\nprovably never loaded the skill. A compound command's aggregate exit status cannot attribute\nfailure to the read itself (`cat SKILL.md && rg no-match` exits 1 with the skill already in\ncontext), so a compound command counts on completion regardless of exit code. Each arm seeds its own skills as workspace-local\n`.agents/skills/<name>` copies - the `solo` arm the evaluated skill, the `baseline` arm none.\n\nEvery arm runs clean: `HOME` points to an empty temporary directory so `~/.agents/skills` is\ninvisible, and `CODEX_HOME` points to a per-trial home that symlinks only `config.toml` and\n`auth.json` from the real `~/.codex`. Skills, plugins, and plugin-activation state are omitted, so\nno globally installed skill leaks in through `CODEX_HOME`. Authentication and model configuration\nare unchanged; the only skills the model sees are the ones seeded into the workspace.\n\nThe Claude adapter runs Claude Code headlessly:\n\n```text\nclaude -p <prompt> --output-format stream-json --verbose --no-session-persistence <permissions>\n```\n\nwith the workspace as the working directory. Trigger cases run\n`--permission-mode dontAsk --allowedTools \"Read,Glob,Grep,Skill\"` - read-only, but the Skill tool\nmust be allowed or invocation would be blocked before it can be observed. Generation cases run\n`--permission-mode acceptEdits`. Invocation is detected from `Skill` tool_use blocks in the\nstream-json trace that name the evaluated skill. Every arm points `CLAUDE_CONFIG_DIR` at a clean\ndirectory holding the credentials file and a minimal `settings.json` rebuilt from only the model,\neffort, and auth-routing keys - hooks, permissions, plugins, and user skills are omitted, so no\nuser configuration acts on one arm differently (on macOS credentials live in the Keychain, so\nauthentication survives; elsewhere the credentials file is copied across). Each arm seeds its own\nskills as workspace-local `.claude/skills/<name>` copies - the `solo` arm the target, the\n`baseline` arm none. The reported model and effort come from the real configuration's\n`settings.json` (`model`/`effortLevel`), or `default`.\n\n**User-invoked skills.** A skill whose frontmatter carries\n`disable-model-invocation: true` is withheld from the model entirely by Claude Code - it appears in\nno listing, no Skill tool call can name it, and its body never enters the context. A seeded arm\nwould therefore be identical to its own baseline, and every case on such a skill would be\nunfalsifiable. Staging removes that key from the staged copy (never from your file), so the body\ncan reach the model.\n\nThat is an approximation, and its limits are worth stating. In production the user invokes the\nskill deliberately and the body loads unconditionally; in a trial the model still has to choose to\ninvoke it. So a `should_trigger: true` case on such a skill is not testing what production does -\nit is establishing the precondition that the body reached the model at all. Read a `not-invoked`\nresult there as \"the precondition was not met\", never as \"the rule is dead\": the ledger already\ntreats `not-invoked` as silence rather than evidence, and `--prune-candidates` will not act on it.\n\nThe pi adapter runs [pi](https://github.com/badlogic/pi-mono) headlessly:\n\n```text\npi -p --mode json --no-session <arm flags> <tool flags> <prompt>\n```\n\nwith the workspace as the working directory. Every arm passes `--no-skills` to hide the user's\nglobal skill library, plus a repeatable `--skill <directory>` per seeded skill (pi loads explicit\n`--skill` paths even under `--no-skills`); the `solo` arm seeds the target, the `baseline` arm\nseeds nothing - no HOME or config redirection is involved.\nTrigger cases restrict tools with `-t read` (read also loads SKILL.md, so invocation stays\nobservable); generation cases keep pi's default tool set. pi implements the Agent Skills\nprogressive-disclosure standard by having the model `read` a listed skill's SKILL.md, so\ninvocation is detected structurally: a `read` toolCall whose `path` argument's final segments are\nexactly `<skill>/SKILL.md`, and whose correlated toolResult did not error - a read that failed\n(missing file, denied) never put the skill into context. Another tool merely mentioning the path\n(a grep pattern, a write body) does not count, and neither does a shell-based read in a\ngeneration arm - the `read` tool is the skill-loading mechanism.\nThe reported model is `defaultProvider/defaultModel` from `~/.pi/settings.json`. pi resolves\nprovider API keys from its auth file or environment variables (e.g. `ZAI_API_KEY`) - the key\nmust be available in the environment running skillval.\n\nUnlike codex (which gets a read-only or `workspace-write` sandbox) and claude (permission modes),\npi has no OS sandbox: generation trials rely on the temporary-workspace convention alone, with no\nenforced isolation, so an agent's writes are only conventionally scoped to the workspace. Because\nof this, skillval refuses to run pi generation cases unless you pass `--allow-unsandboxed-pi` to\nacknowledge the missing sandbox. Trigger cases are read-only and unaffected. Prefer codex or claude\nfor untrusted generation cases.\n\nEach adapter reports its detection method as `invocationDetection` in report metadata. The\n`invoked` signal still has asymmetric confidence: claude (a `Skill` tool_use block) and pi (a\n`read` toolCall's path argument) are `structured` - they parse the executor's actual\nskill-loading event - while codex is `heuristic`: it has no such event, so its adapter\nstring-matches successful command text for `<skill>/SKILL.md`. Trigger rates should not be\ncompared across executors as if they measured the same thing.\n\nBy default, executors do not set a model or thinking/effort level; trials inherit the harness\ndefaults the user has configured, and each adapter captures both into its identity so results are\nalways associated with what actually ran: codex reads `model` and `model_reasoning_effort` from\n`~/.codex/config.toml`, claude reads `model` and `effort` from `settings.json`, and pi reads\n`defaultProvider/defaultModel` and `defaultThinkingLevel` from `~/.pi/settings.json`. A missing\nvalue is recorded as `default` (the provider's own default applies). Passing `--model`/`--effort`\noverrides the default for the run, and the override is what gets captured. Changing any of these -\nin the provider configuration or via the flags - therefore keys distinct cached results.\n\nCached arm results are keyed by runner version, skill content hash, serialized case, arm, executor\nname, executor version, configured model, and configured thinking level. Instruction arms add a\ncontent hash of the resolved instruction file the arm seeds, because two instruction cases can share\nidentical case JSON while their surrounding rules differ - the seeded content, not just the case,\nmust key the arm. A trial has a 15-minute timeout and a 256 MB output cap; exceeding either is\nrecorded as an infrastructure failure, not a content result.\n\nFor instruction targets, each adapter writes the arm's resolved file under the name it reads\nnatively, with no filename translation, and pi additionally redirects `PI_CODING_AGENT_DIR` and\n`HOME` at a clean per-trial directory carrying only auth and model selection - otherwise a\nuser-global `AGENTS.md`/`CLAUDE.md` would enter every arm and could make the `peers` arm pass,\nmisreporting the target rule as redundant.\n\n## Bundled skill\n\nskillval ships one agent skill of its own, under `skill/skillval-coverage/`. It is the judgment\nlayer the deterministic tooling cannot provide: pointed at the skills skillval discovers, it audits\nwhich rules are under-tested, classifies each rule capability-vs-preference, ranks the real gaps by\ndecay risk rather than by case count, and coaches the keep / write / stop decision - including the\nsingle filter that governs it, *can you name a future in which this case flips to fail?* It\ndiagnoses and advises; it does not write or run cases (that is the assisted-authoring skill below).\n\nIt is not auto-installed. After installing skillval, make the bundled directory discoverable to your\nagent by symlinking it onto a skill path both skillval's CLI and your harness can see, e.g.:\n\n```sh\nln -s \"$(npm root -g)/@dungle-scrubs/skillval/skill/skillval-coverage\" ~/.agents/skills/skillval-coverage\n```\n\nThe skill carries its own `skillval.yml`, so skillval can evaluate the skill it ships.\n\n## Roadmap\n\n- Add an opt-in model-as-judge grader for the quality dimension a regex cannot reach (is a present\n  behavior *correct*, not just present). Deterministic stays the default; the judge runs only when a\n  case declares it. Design: [`design/model-as-judge.md`](design/model-as-judge.md).\n- Add a `skillval skill install` command that symlinks the bundled skill onto a discoverable path,\n  replacing the manual step above.\n- Support multi-executor runs through the same normalized trace interface, now that `codex` and\n  `claude` adapters share it.\n- Run multiple models and emit per-model reports. A passing binding or trigger result on a weaker\n  tier is a conservative bound for stronger tiers. Baseline no-op results remain model-specific,\n  and a rule is a prune candidate only when every model in normal use passes at baseline.\n- Add contested-boundary cases with `expect_invoked` and `expect_not_invoked` outcomes.\n- Include the discovered skill-listing hash in trigger-case invalidation so description changes in\n  neighboring skills invalidate affected results.\n- Add cheap trigger simulations for broad description coverage before expensive executor trials.\n- Add a `lint` subcommand for Agent Skills format, references, case coverage, and regular\n  expressions.\n- Ablate a rule across files, not only within one: prove whether an ancestor or imported rule makes\n  a nested rule redundant. This also covers the rule that reaches claude only via a `CLAUDE.md`\n  `@import` of `AGENTS.md`, which single-file ablation reports as `n/a` today.\n- Ship an assisted-authoring skill that drafts `skillval.yml` cases from an existing skill or\n  instruction file, proposing one case per rule with its `rule_text` span for a human to ratify.\n  Case authoring is the real adoption barrier, especially for a long `CLAUDE.md`.\n- Harvest missed triggers, false invocations, and behavioral regressions from real session\n  transcripts as new cases.\n- Support multi-model interpretation and no-op pruning in report summaries, not only raw reports.\n\n## Contributing\n\nSee [CONTRIBUTING.md](CONTRIBUTING.md). Commits follow Conventional Commits, PRs are\nsquash-merged, and every change must pass\n`pnpm typecheck && pnpm lint && pnpm schema:check && pnpm test && pnpm build`.\n\n## License\n\n[MIT](LICENSE)\n","readmeFilename":"README.md"}