{"_id":"@alis-build/harness-eval","_rev":"5-5d4e8f858b3b952ff01869151e108dc0","name":"@alis-build/harness-eval","dist-tags":{"latest":"0.1.4"},"versions":{"0.1.0":{"name":"@alis-build/harness-eval","version":"0.1.0","author":{"name":"www.alisx.com"},"license":"Apache-2.0","_id":"@alis-build/harness-eval@0.1.0","maintainers":[{"name":"newtonnthiga","email":"newton@alisx.com"},{"name":"jankrynauw","email":"jan@alisx.com"},{"name":"daniel-alis-build","email":"daniel.van.niekerk@alisx.com"},{"name":"hi_ruan","email":"ruan@alisx.com"}],"homepage":"https://github.com/alis-build/harness-eval-ts#readme","bugs":{"url":"https://github.com/alis-build/harness-eval-ts/issues"},"bin":{"harness-eval":"dist/cli/bin.js"},"dist":{"shasum":"2a08d6b4aa17c5aef22fe56d4b978ee21746e038","tarball":"https://registry.npmjs.org/@alis-build/harness-eval/-/harness-eval-0.1.0.tgz","fileCount":35,"integrity":"sha512-cFIdEP4fRV4kqUJ2y8vdUhz7GMG51xBr71hcuTXxW5jYAFEY+R8jd6ujHUP68tMiZYXtkLT4FS2Ki5nqfMSemA==","signatures":[{"sig":"MEQCIEy4SHqxLGTzFIC19rFThEW4ALoPJejOx5hc2+tqwmKoAiBT2cRU7GgojS0xVDVWyDAU03ylkFK6x287khl+87duWA==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":642697},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=22.12.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./config":{"types":"./dist/config/loader.d.ts","import":"./dist/config/loader.js"},"./runner":{"types":"./dist/runner/suite.d.ts","import":"./dist/runner/suite.js"},"./adapters/claude-code":{"types":"./dist/adapters/claude-code/index.d.ts","import":"./dist/adapters/claude-code/index.js"}},"gitHead":"0ba9c671b140a48fbf44288d84de55d1ccb15b83","scripts":{"test":"vitest run","build":"pnpm run generate-schemas && tsdown","clean":"rm -rf dist","watch":"tsdown --watch","prepack":"pnpm run build","typecheck":"tsc --noEmit","test:watch":"vitest","prepublishOnly":"pnpm run build","generate-schemas":"tsx src/schemas/generate.ts"},"_npmUser":{"name":"newtonnthiga","email":"newton@alisx.com"},"repository":{"url":"git+https://github.com/alis-build/harness-eval-ts.git","type":"git"},"_npmVersion":"11.14.1","description":"Harness-level eval framework for measuring AI coding agent tool-selection behavior","directories":{},"_nodeVersion":"26.0.0","dependencies":{"zod":"^4.4.3","yaml":"^2.6.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"packageManager":"pnpm@11.3.0","devDependencies":{"tsx":"^4.22.4","tsdown":"^0.22.3","vitest":"^2.1.0","typescript":"^5.6.0","@types/node":"^22.12.0"},"_npmOperationalInternal":{"tmp":"tmp/harness-eval_0.1.0_1782218761715_0.6547997884858971","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@alis-build/harness-eval","version":"0.1.1","author":{"name":"www.alisx.com"},"license":"Apache-2.0","_id":"@alis-build/harness-eval@0.1.1","maintainers":[{"name":"newtonnthiga","email":"newton@alisx.com"},{"name":"jankrynauw","email":"jan@alisx.com"},{"name":"daniel-alis-build","email":"daniel.van.niekerk@alisx.com"},{"name":"hi_ruan","email":"ruan@alisx.com"}],"homepage":"https://github.com/alis-build/harness-eval-ts#readme","bugs":{"url":"https://github.com/alis-build/harness-eval-ts/issues"},"bin":{"harness-eval":"dist/cli/bin.js"},"dist":{"shasum":"f1cec3c3f86d9e044a622f11df3de0b1a138a704","tarball":"https://registry.npmjs.org/@alis-build/harness-eval/-/harness-eval-0.1.1.tgz","fileCount":35,"integrity":"sha512-mdSS2xULN5X6QkdDXslNNy+yxdrfq3bDC2qj5GDIkBQIHAygOSBe7rwM9kKEmOD8uu5SxbLGxgJ08NRQ28as2w==","signatures":[{"sig":"MEUCIQC0dkb/JsxoyRUbCUEwzmXVk3Zcb2WCAsFMDRt5Xnr/aAIgZkRJpjjfIxEew9RqaBclJRq0wBG4DTyXnJ8LxbHEWgc=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@alis-build%2fharness-eval@0.1.1","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":643035},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=22.12.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./config":{"types":"./dist/config/loader.d.ts","import":"./dist/config/loader.js"},"./runner":{"types":"./dist/runner/suite.d.ts","import":"./dist/runner/suite.js"},"./adapters/claude-code":{"types":"./dist/adapters/claude-code/index.d.ts","import":"./dist/adapters/claude-code/index.js"}},"gitHead":"42d96392df7a533fc431f7d06d883415a861aec7","scripts":{"test":"vitest run","build":"pnpm run generate-schemas && tsdown","clean":"rm -rf dist","watch":"tsdown --watch","prepack":"pnpm run build","postbuild":"node scripts/link-local-bin.mjs","typecheck":"tsc --noEmit","test:watch":"vitest","prepublishOnly":"pnpm run build","generate-schemas":"tsx src/schemas/generate.ts"},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:59c01781-11b6-4034-ac3c-6ae6a213b928"}},"repository":{"url":"git+https://github.com/alis-build/harness-eval-ts.git","type":"git"},"_npmVersion":"11.13.0","description":"Harness-level eval framework for measuring AI coding agent tool-selection behavior","directories":{},"_nodeVersion":"24.16.0","dependencies":{"zod":"^4.4.3","yaml":"^2.6.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"packageManager":"pnpm@11.3.0","devDependencies":{"tsx":"^4.22.4","tsdown":"^0.22.3","vitest":"^2.1.0","typescript":"^5.6.0","@types/node":"^22.12.0"},"_npmOperationalInternal":{"tmp":"tmp/harness-eval_0.1.1_1782220392183_0.7525363172435044","host":"s3://npm-registry-packages-npm-production"}},"0.1.2":{"name":"@alis-build/harness-eval","version":"0.1.2","author":{"name":"www.alisx.com"},"license":"Apache-2.0","_id":"@alis-build/harness-eval@0.1.2","maintainers":[{"name":"newtonnthiga","email":"newton@alisx.com"},{"name":"jankrynauw","email":"jan@alisx.com"},{"name":"daniel-alis-build","email":"daniel.van.niekerk@alisx.com"},{"name":"hi_ruan","email":"ruan@alisx.com"}],"homepage":"https://github.com/alis-build/harness-eval-ts#readme","bugs":{"url":"https://github.com/alis-build/harness-eval-ts/issues"},"bin":{"harness-eval":"dist/cli/bin.js"},"dist":{"shasum":"82400b987bdc6422d05e3435ec803509bf034ead","tarball":"https://registry.npmjs.org/@alis-build/harness-eval/-/harness-eval-0.1.2.tgz","fileCount":35,"integrity":"sha512-anLMVbOuA3wFyHAvtA3pQocMv8Ic2QHQSP2Lj8WJ/F4fECpn9v82NqGfV3PkvCsOKYMnQu0/+dCxCwLQSNPESA==","signatures":[{"sig":"MEUCIC4ITwpUuwowP+pgSTxsfocpyvZBsN15PL6GrLTyylsdAiEAraNdfI37apfM7RSJIBK6cjkph2TnVt7/9pPVkk0HcKE=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@alis-build%2fharness-eval@0.1.2","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":722651},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=22.12.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./config":{"types":"./dist/config/loader.d.ts","import":"./dist/config/loader.js"},"./runner":{"types":"./dist/runner/suite.d.ts","import":"./dist/runner/suite.js"},"./adapters/claude-code":{"types":"./dist/adapters/claude-code/index.d.ts","import":"./dist/adapters/claude-code/index.js"}},"gitHead":"ac993420c3a31f7048e1404e60e5ab7ea76d9e68","scripts":{"test":"vitest run","build":"pnpm run generate-schemas && tsdown","clean":"rm -rf dist","watch":"tsdown --watch","prepack":"pnpm run build","postbuild":"node scripts/link-local-bin.mjs","typecheck":"tsc --noEmit","test:watch":"vitest","prepublishOnly":"pnpm run build","generate-schemas":"tsx src/schemas/generate.ts"},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:59c01781-11b6-4034-ac3c-6ae6a213b928"}},"repository":{"url":"git+https://github.com/alis-build/harness-eval-ts.git","type":"git"},"_npmVersion":"11.13.0","description":"Harness-level eval framework for measuring AI coding agent tool-selection behavior","directories":{},"_nodeVersion":"24.16.0","dependencies":{"zod":"^4.4.3","yaml":"^2.6.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"packageManager":"pnpm@11.3.0","devDependencies":{"tsx":"^4.22.4","tsdown":"^0.22.3","vitest":"^2.1.0","typescript":"^5.6.0","@types/node":"^22.12.0","@google-cloud/aiplatform":"^6.8.1"},"_npmOperationalInternal":{"tmp":"tmp/harness-eval_0.1.2_1782227916257_0.46321394272702676","host":"s3://npm-registry-packages-npm-production"}},"0.1.3":{"name":"@alis-build/harness-eval","version":"0.1.3","author":{"name":"www.alisx.com"},"license":"Apache-2.0","_id":"@alis-build/harness-eval@0.1.3","maintainers":[{"name":"newtonnthiga","email":"newton@alisx.com"},{"name":"jankrynauw","email":"jan@alisx.com"},{"name":"daniel-alis-build","email":"daniel.van.niekerk@alisx.com"},{"name":"hi_ruan","email":"ruan@alisx.com"}],"homepage":"https://github.com/alis-build/harness-eval-ts#readme","bugs":{"url":"https://github.com/alis-build/harness-eval-ts/issues"},"bin":{"harness-eval":"dist/cli/bin.js"},"dist":{"shasum":"200706ca61382e5254ab83337d0f7d36df9d54bf","tarball":"https://registry.npmjs.org/@alis-build/harness-eval/-/harness-eval-0.1.3.tgz","fileCount":42,"integrity":"sha512-AL7b8RucFxJnUBD2dSdrMx33PoUPqZ01LkWp/wLgWMIMTHeQL7Okbex4FvHnmkOjMHFiLykq7sotQIGvcWLVWA==","signatures":[{"sig":"MEUCIEX2RgQHeqHItC/tH/pOVt4R378Q27yyP2u4VMhiN36rAiEAp85wcjSoPuR0BxH7ltfTIAdJ3QlzytA30XK70hGdwXc=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@alis-build%2fharness-eval@0.1.3","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":886644},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=22.12.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./config":{"types":"./dist/config/loader.d.ts","import":"./dist/config/loader.js"},"./runner":{"types":"./dist/runner/suite.d.ts","import":"./dist/runner/suite.js"},"./adapters/codex":{"types":"./dist/adapters/codex/index.d.ts","import":"./dist/adapters/codex/index.js"},"./adapters/claude-code":{"types":"./dist/adapters/claude-code/index.d.ts","import":"./dist/adapters/claude-code/index.js"}},"gitHead":"7f2a2389b6d7aa823d65b43bcffc8f6b0184cd5d","scripts":{"test":"vitest run","build":"pnpm run generate-schemas && tsdown","clean":"rm -rf dist","watch":"tsdown --watch","prepack":"pnpm run build","postbuild":"node scripts/link-local-bin.mjs","typecheck":"tsc --noEmit","test:watch":"vitest","prepublishOnly":"pnpm run build","generate-schemas":"tsx src/schemas/generate.ts"},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:59c01781-11b6-4034-ac3c-6ae6a213b928"}},"repository":{"url":"git+https://github.com/alis-build/harness-eval-ts.git","type":"git"},"_npmVersion":"11.13.0","description":"Harness-level eval framework for measuring AI coding agent tool-selection behavior","directories":{},"_nodeVersion":"24.17.0","dependencies":{"zod":"^4.4.3","yaml":"^2.6.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"packageManager":"pnpm@11.8.0","devDependencies":{"tsx":"^4.22.4","tsdown":"^0.22.3","vitest":"^2.1.0","typescript":"^5.6.0","@types/node":"^22.12.0","@google-cloud/aiplatform":"^6.8.1"},"_npmOperationalInternal":{"tmp":"tmp/harness-eval_0.1.3_1782369294154_0.3647341737952672","host":"s3://npm-registry-packages-npm-production"}},"0.1.4":{"name":"@alis-build/harness-eval","version":"0.1.4","description":"Harness-level eval framework for measuring AI coding agent tool-selection behavior","type":"module","main":"./dist/index.js","types":"./dist/index.d.ts","author":{"name":"www.alisx.com"},"license":"Apache-2.0","engines":{"node":">=22.12.0"},"repository":{"type":"git","url":"git+https://github.com/alis-build/harness-eval-ts.git"},"homepage":"https://github.com/alis-build/harness-eval-ts#readme","bugs":{"url":"https://github.com/alis-build/harness-eval-ts/issues"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./adapters/claude-code":{"types":"./dist/adapters/claude-code/index.d.ts","import":"./dist/adapters/claude-code/index.js"},"./adapters/codex":{"types":"./dist/adapters/codex/index.d.ts","import":"./dist/adapters/codex/index.js"},"./adapters/gemini-cli":{"types":"./dist/adapters/gemini-cli/index.d.ts","import":"./dist/adapters/gemini-cli/index.js"},"./runner":{"types":"./dist/runner/suite.d.ts","import":"./dist/runner/suite.js"},"./config":{"types":"./dist/config/loader.d.ts","import":"./dist/config/loader.js"}},"scripts":{"generate-schemas":"tsx src/schemas/generate.ts","build":"pnpm run generate-schemas && tsdown","prebuild":"pnpm run typecheck","postbuild":"node scripts/link-local-bin.mjs","prepack":"pnpm run build","prepublishOnly":"pnpm run build","watch":"tsdown --watch","clean":"rm -rf dist","test":"vitest run","test:watch":"vitest","typecheck":"tsc --noEmit"},"bin":{"harness-eval":"dist/cli/bin.js"},"dependencies":{"yaml":"^2.6.0","zod":"^4.4.3"},"devDependencies":{"@google-cloud/aiplatform":"^6.8.1","@types/node":"^22.12.0","tsdown":"^0.22.3","tsx":"^4.22.4","typescript":"^5.6.0","vitest":"^2.1.0"},"publishConfig":{"access":"public"},"packageManager":"pnpm@11.8.0","gitHead":"f92ec29b365ee22040594920db1b66cdbc02327a","_id":"@alis-build/harness-eval@0.1.4","_nodeVersion":"24.17.0","_npmVersion":"11.13.0","dist":{"integrity":"sha512-vna6wGo463IG0IIqq832CA/iR4m3We4DrRZJ378FM3CqGCIzEAoEDeRYHUrn89TFIGl4VQCuaOeTSLe5ixdQKA==","shasum":"14a7e5b728f5c42e5d32b53ecdd251462ced0d71","tarball":"https://registry.npmjs.org/@alis-build/harness-eval/-/harness-eval-0.1.4.tgz","fileCount":42,"unpackedSize":957720,"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@alis-build%2fharness-eval@0.1.4","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCICR0A4Z8743/MM3XL/RQ31gAXM08WWwWLXdVNs110X9iAiEAjaqx+wziSknf4Dr+6SUF1TAgGULSvz0iVZFwDCfyd/w="}]},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:59c01781-11b6-4034-ac3c-6ae6a213b928"}},"directories":{},"maintainers":[{"name":"newtonnthiga","email":"newton@alisx.com"},{"name":"jankrynauw","email":"jan@alisx.com"},{"name":"daniel-alis-build","email":"daniel.van.niekerk@alisx.com"},{"name":"hi_ruan","email":"ruan@alisx.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/harness-eval_0.1.4_1782383352953_0.7560244367816038"},"_hasShrinkwrap":false}},"time":{"created":"2026-06-23T12:46:01.552Z","modified":"2026-06-25T10:29:13.469Z","0.1.0":"2026-06-23T12:46:01.901Z","0.1.1":"2026-06-23T13:13:12.375Z","0.1.2":"2026-06-23T15:18:36.457Z","0.1.3":"2026-06-25T06:34:54.321Z","0.1.4":"2026-06-25T10:29:13.135Z"},"bugs":{"url":"https://github.com/alis-build/harness-eval-ts/issues"},"author":{"name":"www.alisx.com"},"license":"Apache-2.0","homepage":"https://github.com/alis-build/harness-eval-ts#readme","repository":{"type":"git","url":"git+https://github.com/alis-build/harness-eval-ts.git"},"description":"Harness-level eval framework for measuring AI coding agent tool-selection behavior","maintainers":[{"name":"newtonnthiga","email":"newton@alisx.com"},{"name":"jankrynauw","email":"jan@alisx.com"},{"name":"daniel-alis-build","email":"daniel.van.niekerk@alisx.com"},{"name":"hi_ruan","email":"ruan@alisx.com"}],"readme":"# @alis-build/harness-eval\n\nStatistical eval framework for **AI coding agent harnesses**. Run real headless harness sessions, capture tool trajectories, and score behavior and outcomes across many repetitions and configurations.\n\n**Built-in harness adapters:** `claude-code`, `codex`, and `gemini-cli`. Set `adapter:` in suite YAML; the runner, assertions, and eval interchange stay the same regardless of vendor.\n\n**Use it to answer:** “When users ask X, does this harness actually call our MCP tools — reliably, in this plugin/model setup?”\n\n---\n\n## Requirements\n\n- Node.js ≥ 22.12 required; Node 24 LTS recommended for development and CI\n- A harness CLI on `PATH` for the adapter you use (see [Adding harness adapters](#adding-harness-adapters)):\n  - **`claude-code`** — `claude` ([Claude Code CLI](https://code.claude.com/docs/en/cli-reference))\n  - **`codex`** — `codex` ([Codex CLI](https://developers.openai.com/codex/cli/reference))\n  - **`gemini-cli`** — `gemini` ([Gemini CLI](https://geminicli.com/docs/cli/cli-reference/))\n\n### Authentication (by adapter)\n\n| Adapter | Typical auth |\n| ------- | ------------ |\n| **Claude Code** | `claude login` with `isolateConfig: false`, or `ANTHROPIC_API_KEY` with isolated config (default harness behavior) |\n| **Codex** | Logged-in `~/.codex`, or `OPENAI_API_KEY` when `codex.isolateConfig: true` |\n| **Gemini CLI** | Logged-in Gemini CLI config with `geminiCli.isolateConfig: false`, or Vertex/API key env vars (`GOOGLE_APPLICATION_CREDENTIALS`, `GEMINI_API_KEY`, etc.) when isolated |\n\nEach adapter section below documents `isolateConfig`, MCP setup, and headless flags in detail.\n\n---\n\n## Install\n\n**Consumers** — run via npx (no global install required):\n\n```bash\nnpx @alis-build/harness-eval --help\n```\n\nOr install as a project dependency:\n\n```bash\nnpm install @alis-build/harness-eval\nnpx @alis-build/harness-eval run examples/basic.yaml --output report.json\n```\n\nThe npm package name is `@alis-build/harness-eval`; the CLI binary is `harness-eval`. With a single bin entry, `npx @alis-build/harness-eval <command>` invokes it directly.\n\nIn a git checkout of this repo, npm resolves `npx @alis-build/harness-eval` to the local package (not the registry). Run `pnpm run build` first so `dist/cli/bin.js` exists; the build links `harness-eval` into `node_modules/.bin` for local use.\n\n### Development (clone & build)\n\nContributors working from a git checkout:\n\n```bash\npnpm install\npnpm run build\npnpm exec harness-eval --help\n# or: node dist/cli/bin.js --help\n```\n\n---\n\n## Quick start\n\n### 1. Write a suite\n\nSuites are YAML files. Committed examples:\n\n- [`examples/pipeline/`](examples/pipeline/) — **recommended** unified layout with inline `judge:` + `pipeline:` orchestration\n- [`examples/basic.yaml`](examples/basic.yaml) — Claude Code smoke test (`Read` on this repo's README)\n- [`examples/codex-basic.yaml`](examples/codex-basic.yaml) — Codex CLI smoke test\n- [`examples/gemini-cli-basic.yaml`](examples/gemini-cli-basic.yaml) — Gemini CLI smoke test\n- [`examples/matrix.yaml`](examples/matrix.yaml) — Claude Code with a model matrix (sonnet vs opus)\n- [`examples/multi-file/`](examples/multi-file/) — directory layout with `suite.yaml` plus cases under `cases/`\n- [`examples/grading.yaml`](examples/grading.yaml) — Claude Code judge config (standalone)\n- [`examples/codex-grading.yaml`](examples/codex-grading.yaml) — Codex judge config\n- [`examples/gemini-grading.yaml`](examples/gemini-grading.yaml) — Gemini CLI judge config\n\n```yaml\nadapter: claude-code   # or: codex | gemini-cli\n\ndefaultConfig:\n  model: claude-sonnet-4-6\n  timeoutMs: 120000\n  cwd: ..\n  claudeCode:\n    isolateConfig: false # use your logged-in Claude Code config\n    permissionMode: bypassPermissions\n    allowedTools:\n      - Read\n\nmatrix:\n  - label: sonnet\n    config: {}\n\ncases:\n  - id: summarize-readme\n    prompt: \"Read README.md and summarize what harness-eval does in one or two sentences.\"\n    repetitions: 3\n\n    # Behavioral checks (deterministic, on tool trajectory)\n    assertions:\n      - called: Read\n        threshold: 0.8\n      - not:\n          responded_without_tool_calls: true\n\n    # Outcome checks (LLM judge via `harness-eval grade`)\n    expectations:\n      - \"The response describes an eval framework for AI coding agent harnesses\"\n      - \"The summary is grounded in README content, not a generic refusal\"\n```\n\nGeneric fields (`model`, `cwd`, `timeoutMs`, `env`) sit at the top level. Harness-specific options nest under `claudeCode`, `codex`, or `geminiCli` depending on `adapter`.\n\n**Full suite & grading YAML reference:** [docs/suite-config.md](docs/suite-config.md) — all case/matrix fields, inline `judge:` / `pipeline:`, multi-file layout, and standalone `grading.yaml`.\n\n### 2. Run behavioral eval\n\n```bash\n# Unified pipeline (run + optional grade + envelope when pipeline: is defined)\nnpx @alis-build/harness-eval pipeline examples/pipeline/\n\n# Or run harness only\nnpx @alis-build/harness-eval run examples/basic.yaml --output report.json --max-concurrent 1 --format console\n```\n\nThis spawns the configured harness CLI headless for each (case × matrix cell × repetition), evaluates **assertions** on the captured trajectory, and prints pass rates.\n\n**Progress (stderr):** one line per repetition with ETA by default; use `--quiet` for dots or `--verbose` for tool/assertion detail.\n\nExit code `0` = all cells passed all assertion thresholds.\n\n### 3. Grade outcomes (optional)\n\n**Unified suite:** add a top-level `judge:` block in `suite.yaml` (see [`examples/pipeline/suite.yaml`](examples/pipeline/suite.yaml)), then:\n\n```bash\nnpx @alis-build/harness-eval grade report.json --suite examples/pipeline/suite.yaml --output grading.json --max-concurrent 1 --format console\n# or: npx @alis-build/harness-eval pipeline examples/pipeline/ --steps grade\n```\n\n**Standalone grading file:** judge config in a separate **`grading.yaml`** (still supported). See [`examples/grading.yaml`](examples/grading.yaml).\n\n```bash\nnpx @alis-build/harness-eval grade report.json --config examples/grading.yaml --output grading.json --max-concurrent 1 --format console\n```\n\nRuns a separate harness subprocess as **judge** (`judge.adapter`: `claude-code`, `codex`, or `gemini-cli`) against the `expectations` in your suite (copied into `report.json`). Produces per-expectation PASS/FAIL with cited evidence.\n\nExit codes: `0` = all graded expectations passed; `1` = at least one failed; `2` = no expectations or no gradable repetitions.\n\n---\n\n## Data contracts & schemas\n\nharness-eval separates **vendor output** from **eval interchange**. Use the types below when wiring CI, a database, or an external judge — not raw adapter NDJSON or OTLP as your primary record.\n\n### Layering\n\n| Layer           | Type                  | Where                     | Use for                                            |\n| --------------- | --------------------- | ------------------------- | -------------------------------------------------- |\n| Vendor stream   | `StreamEvent`         | `src/types/stream.ts`     | Adapter debug only (Claude/Codex/Gemini NDJSON)    |\n| Harness session | **`TrajectoryView`**  | `src/types/trajectory.ts` | Assertions, trajectory queries, judge input        |\n| Run report      | **`SuiteReport`**     | `report.json` from `run`  | Runner output; full trajectories + assertion stats |\n| Eval record     | **`EvalRunEnvelope`** | `buildEvalRunEnvelope()`  | CI gates, APIs, DB storage                         |\n| Observability   | OTLP                  | `--otel-output`           | Tempo / Jaeger side export                         |\n\n```\nSuite YAML → run → TrajectoryView → SuiteReport (report.json)\n                              ↓ optional grade / external judge\n                         EvalRunEnvelope → DB / API / CI gate\n```\n\n### `TrajectoryView`\n\nCross-harness normalized session. Every adapter maps vendor output into this shape.\n\n| Field           | Meaning                                                                                |\n| --------------- | -------------------------------------------------------------------------------------- |\n| `meta`          | Session id, model, cwd, available tools, MCP server status                             |\n| `toolCalls`     | Every tool call in emission order (`name`, `args`, `result`, `turnIndex`, `callIndex`) |\n| `turns`         | Per-turn assistant text and tool calls                                                 |\n| `finalResponse` | Concatenated assistant text (for `response_contains` and judges)                       |\n| `usage`         | Tokens, cost, duration, turn count                                                     |\n| `success`       | Whether the harness reported success                                                   |\n\nTool names follow the harness format (e.g. `mcp__plugin_alis-build_api__SearchSkills`). Assertions use `turnIndex` / `callIndex` for ordering — not wall-clock time.\n\n### `SuiteReport` (`report.json`)\n\nProduced by `harness-eval run`. Contains everything from the run:\n\n- `cells[]` — one row per (test case × matrix cell)\n- `cells[].repetitions[]` — each harness invocation\n- `cells[].repetitions[].adapterResult.view` — **`TrajectoryView`** when the harness succeeded\n- `cells[].repetitions[].assertionResults` — per-rep behavioral assertion tree\n- `cells[].assertionStats` — pass rates across repetitions\n- `cells[].expectations` — natural-language outcome checks (copied from suite for judges)\n\nGate behavioral eval on `cells[].passed` or on assertion stats. This file is enough to hand off to a custom judge without re-running the harness.\n\n### `EvalRunEnvelope`\n\nVersioned document for **storage and interchange** (`schemaVersion` `1.0`). Build it from a report (and optional grading):\n\n```typescript\nimport {\n  buildEvalRunEnvelope,\n  buildEvalRunEnvelopeFromFiles,\n} from \"@alis-build/harness-eval\";\n\nconst envelope = buildEvalRunEnvelope(report, {\n  grading, // optional: from gradeReport()\n  suite: { uri: \"./examples/basic.yaml\" },\n  provenance: { git: { commit: process.env.GITHUB_SHA } },\n});\n\n// Or from disk after CLI run:\nconst envelope = await buildEvalRunEnvelopeFromFiles(\"report.json\", {\n  gradingPath: \"grading.json\",\n  suitePath: \"examples/basic.yaml\",\n});\n```\n\n| Field                                        | Meaning                                                                           |\n| -------------------------------------------- | --------------------------------------------------------------------------------- |\n| `summary.behavioralPass`                     | All cells passed assertion thresholds                                             |\n| `summary.outcomePass`                        | All graded expectations passed (when outcome layer present)                       |\n| `cells[].repetitions[]`                      | Unit of work for judges — trajectory, assertion results, optional `outcomeGrades` |\n| `cells[].repetitions[].artifacts.transcript` | Text for LLM judges (`trajectoryToTranscript`)                                    |\n| `cells[].repetitions[].externalScores`       | Attach scores from LangSmith, Braintrust, etc.                                    |\n\n**Full reference:** [docs/eval-record.md](docs/eval-record.md)\n\n### TypeScript types & Zod schemas\n\n| Artifact                         | Location                                                                               |\n| -------------------------------- | -------------------------------------------------------------------------------------- |\n| TypeScript interfaces            | `@alis-build/harness-eval` — `TrajectoryView`, `EvalRunEnvelope`, `AssertionResult`, … |\n| Zod schemas (runtime validation) | `src/schemas/` in repo only — not published on npm                                     |\n| JSON Schema (DB / OpenAPI)       | `schemas/*.schema.json` — shipped in the npm package                                   |\n\nZod is the **source of truth** for JSON Schema. Each field has `title` and `description` for downstream tooling.\n\n```bash\npnpm run generate-schemas   # Zod → schemas/*.schema.json\n```\n\nPublished JSON Schema files (Draft 2020-12):\n\n- `schemas/trajectory-view.schema.json` — `TrajectoryView` + `schemaVersion`\n- `schemas/eval-run-envelope.schema.json` — full run envelope\n\nCanonical `$id` URLs (for validators and `$ref`):\n\n- `https://raw.githubusercontent.com/alis-build/harness-eval-ts/main/schemas/trajectory-view.schema.json`\n- `https://raw.githubusercontent.com/alis-build/harness-eval-ts/main/schemas/eval-run-envelope.schema.json`\n\nSource: [github.com/alis-build/harness-eval-ts](https://github.com/alis-build/harness-eval-ts)\n\nRuntime validation (repo development or clone):\n\n```typescript\nimport { evalRunEnvelopeSchema } from \"./src/schemas/eval-run-envelope\";\nevalRunEnvelopeSchema.parse(envelope);\n```\n\nnpm consumers validate with the published JSON Schema files or by cloning the repo for Zod imports.\n\nUses [Zod 4 `z.toJSONSchema()`](https://zod.dev/json-schema).\n\n---\n\n## External eval frameworks & custom judges\n\nharness-eval is intentionally split: **run the harness and score behavior deterministically**; **outcome quality can live anywhere**.\n\nYou do not need `harness-eval grade` if you already have LangSmith, Braintrust, OpenAI Evals, a Python judge, or an internal rubric service.\n\n### What harness-eval provides vs what you can replace\n\n| Concern                  | harness-eval                   | External framework / custom judge          |\n| ------------------------ | ------------------------------ | ------------------------------------------ |\n| Headless harness runs    | `run` / `runSuite`             | —                                          |\n| Tool-call behavior       | Assertions on `TrajectoryView` | Optional: re-implement on `toolCalls`      |\n| Outcome / rubric scoring | `grade` (built-in judges)      | Your judge, eval platform, or human review |\n| Storage contract         | `EvalRunEnvelope`              | Same envelope; attach `externalScores`     |\n\n### Pattern 1 — Behavioral only (no LLM judge)\n\nRun the suite, gate CI on behavioral pass rates, skip outcome grading entirely.\n\n```bash\nnpx @alis-build/harness-eval run examples/basic.yaml --output report.json\n# exit 0 ⇒ all assertion thresholds met\n```\n\nOmit `expectations` from the suite, or ignore them. Your pipeline only checks `report.json` assertion stats.\n\n### Pattern 2 — Custom judge in TypeScript (`gradeFn`)\n\nKeep the harness-eval grading **workflow** (concurrency, report shape) but swap the judge implementation:\n\n```typescript\nimport {\n  gradeReport,\n  trajectoryToTranscript,\n  type GraderFn,\n} from \"@alis-build/harness-eval\";\n\nconst myJudge: GraderFn = async ({ prompt, transcript, expectations }) => {\n  // Call your API, rubric service, or local model\n  const results = await myRubricService.score(transcript, expectations);\n  return {\n    expectations: results,\n    summary: { passed: 2, failed: 0, total: 2, passRate: 1 },\n  };\n};\n\nconst grading = await gradeReport(report, { gradeFn: myJudge });\n```\n\nOutput is the same `SuiteGradingReport` shape as the built-in judges — merge into `EvalRunEnvelope` via `buildEvalRunEnvelope(report, { grading })`.\n\n### Pattern 3 — Separate judge pipeline (any language)\n\n1. `npx @alis-build/harness-eval run … --output report.json`\n2. Your service reads each repetition:\n\n```typescript\n// Minimal handoff fields from report.json\nconst cell = report.cells[0];\nconst rep = cell.repetitions[0];\nconst view = rep.adapterResult?.view;\nconst prompt = cell.prompt;\nconst expectations = cell.expectations ?? [];\n\n// Prefer transcript for LLM judges\nimport { trajectoryToTranscript } from \"@alis-build/harness-eval\";\nconst transcript = view ? trajectoryToTranscript(view, prompt ?? \"\") : null;\n\n// Or use structured toolCalls for deterministic checks\nconst toolNames = view?.toolCalls.map((t) => t.name) ?? [];\n```\n\n3. Write scores to your DB or a sidecar JSON.\n4. Optionally merge into an envelope for a unified eval store:\n\n```typescript\nconst envelope = buildEvalRunEnvelope(report, { grading });\n// Attach platform scores per repetition (not a buildEvalRunEnvelope option today):\nenvelope.cells[0].repetitions[0].externalScores = [\n  { source: \"langsmith\", metric: \"correctness\", value: 0.92 },\n];\n```\n\n**Judges should use `trajectoryToTranscript(view, prompt)` or structured `toolCalls`** — not raw vendor NDJSON (adapter-specific and verbose).\n\n### Pattern 4 — LangSmith, Braintrust, OpenAI Evals, etc.\n\nTypical flow:\n\n1. **Generate trajectories** with harness-eval (real harness, real MCP tools, statistical repetitions).\n2. **Upload or reference** each repetition in your platform:\n   - **Input:** `prompt`, `artifacts.transcript` (from envelope), or `TrajectoryView`\n   - **Metadata:** `caseId`, `cellLabel`, `axes`, `runId`, git/CI provenance from `EvalRunEnvelope`\n3. **Run the platform's evaluators** (LLM judges, human review, custom scorers).\n4. **Attach scores** back via `externalScores` on `EvalRepetition` when building the envelope, or store platform run IDs in `provenance`.\n\nharness-eval does not need to own scoring — it owns **faithful harness reproduction** and a **stable trajectory contract**.\n\n### Pattern 5 — Behavioral here, outcome elsewhere (recommended split)\n\n```bash\n# CI job 1: behavioral gate (fast, deterministic)\nnpx @alis-build/harness-eval run suite.yaml --output report.json\n\n# CI job 2: your outcome eval (async, platform-specific)\nnode scripts/push-to-langsmith.mjs report.json\n# or: python scripts/run_braintrust_eval.py report.json\n```\n\n- Job 1 fails on tool-selection regressions immediately.\n- Job 2 scores answer quality without blocking on harness spawn time.\n\nBoth can converge on one `EvalRunEnvelope` in your database for dashboards.\n\n### Injecting a custom `GraderInput`\n\nBuilt-in grader input shape:\n\n```typescript\ninterface GraderInput {\n  prompt: string;\n  transcript: string; // from trajectoryToTranscript()\n  expectations: string[]; // from suite / report\n}\n```\n\nBuilt-in output shape (`outcomeGrades` in the envelope):\n\n```typescript\ninterface GradedExpectation {\n  text: string;\n  passed: boolean;\n  evidence: string;\n}\n```\n\nMap your framework's output into these shapes (or use `externalScores`) so CI and DB layers stay consistent.\n\n---\n\n## Two layers of evaluation\n\n| Layer        | Command | What it checks                          | Mechanism                                    |\n| ------------ | ------- | --------------------------------------- | -------------------------------------------- |\n| **Behavior** | `run`   | Tool calls, order, args, efficiency     | Deterministic assertions on `TrajectoryView` |\n| **Outcome**  | `grade` | Answer quality, grounding, completeness | LLM judge (`claude-code`, `codex`, or `gemini-cli`) on transcript + `finalResponse` |\n\nBoth layers use statistical thresholds: a case runs `repetitions` times per matrix cell, and each assertion/expectation has a pass-rate threshold (default `1.0`).\n\n---\n\n## CLI reference\n\n```bash\nnpx @alis-build/harness-eval run <suite.yaml> [options]\nnpx @alis-build/harness-eval grade <report.json> [options]\nnpx @alis-build/harness-eval envelope <report.json> [options]\nnpx @alis-build/harness-eval pipeline <suite.yaml|dir> [options]\nnpx @alis-build/harness-eval format <report.json> [options]\nnpx @alis-build/harness-eval --help\n```\n\n### `run`\n\n| Option                             | Description                                                                              |\n| ---------------------------------- | ---------------------------------------------------------------------------------------- |\n| `--output <path>`                  | Write full `SuiteReport` JSON                                                            |\n| `--otel-output <dir>`              | Write OTLP trace JSON per repetition (optional)                                          |\n| `--format console\\|markdown\\|json` | Report format (default: `console`)                                                       |\n| `--baseline <path>`                | Compare against a previous report                                                        |\n| `--max-concurrent <n>`             | Parallel harness processes (default: 4)                                                  |\n| `--adapter <id>`                   | Harness adapter (default: `claude-code`)                                                 |\n| `--quiet`                          | Progress: dots only (`.` ok, `x` fail)                                                   |\n| `--verbose`                        | Progress: per-rep tool counts and assertion summary                                      |\n| `--progress <mode>`                | `default` \\| `quiet` \\| `verbose` \\| `json` (ndjson on stderr; disables color)           |\n| `--color` / `--no-color`           | Force or disable ANSI colors (auto when stderr is a TTY; `NO_COLOR` / `FORCE_COLOR` env) |\n\n### `grade`\n\nUses **`grading.yaml`**, an inline **`judge:`** block in `suite.yaml` (`--suite`), or adapter-specific grading files under `examples/`.\n\n**Field reference:** [docs/suite-config.md — Grading config](docs/suite-config.md#grading-config-gradingyaml)\n\n```yaml\n# examples/grading.yaml (Claude Code judge)\njudge:\n  adapter: claude-code\n  model: claude-sonnet-4-6\n  timeoutMs: 300000\n  maxConcurrent: 1\n  claudeCode:\n    permissionMode: bypassPermissions\n```\n\nOther committed judge configs: [`examples/codex-grading.yaml`](examples/codex-grading.yaml) (`adapter: codex`), [`examples/gemini-grading.yaml`](examples/gemini-grading.yaml) (`adapter: gemini-cli`).\n\n```bash\nnpx @alis-build/harness-eval grade report.json --config examples/grading.yaml --output grading.json\nnpx @alis-build/harness-eval grade report.json --config examples/codex-grading.yaml --output grading.json\nnpx @alis-build/harness-eval grade report.json --config examples/gemini-grading.yaml --output grading.json\n```\n\n| Option                                 | Description                                                       |\n| -------------------------------------- | ----------------------------------------------------------------- |\n| `--config <path>`                      | Grading YAML (`judge` block) — model, env, timeout, adapter options |\n| `--suite <path>`                       | Unified `suite.yaml` with inline `judge:` (alternative to `--config`) |\n| `--output <path>`                      | Write grading JSON                                                |\n| `--expectations <path>`                | Sidecar YAML/JSON if report lacks expectations                    |\n| `--format console\\|json`               | Output format                                                     |\n| `--model <id>`                         | Overrides `judge.model` in config                                 |\n| `--binary <path>`                      | Overrides judge binary for the selected adapter                   |\n| `--timeout-ms <n>`                     | Overrides `judge.timeoutMs`                                       |\n| `--max-concurrent <n>`                 | Overrides `judge.maxConcurrent` (default: 2 if unset)             |\n| `--quiet` / `--verbose` / `--progress` | Same progress modes as `run` (including `--color` / `--no-color`) |\n\nCLI flags override the YAML file. Expectations still come from `report.json` (copied from the suite at `run` time) unless `--expectations` is set. The grading report may include `gradingConfigPath` when `--config` was used.\n\n**Built-in judge defaults** (override under `judge.claudeCode`, `judge.codex`, or `judge.geminiCli`):\n\n| Adapter | Defaults (summary) |\n| ------- | ------------------ |\n| `claude-code` | `maxTurns: 1`, `bare: true`, `disableSlashCommands: true`, `noSessionPersistence: true`, `permissionMode: bypassPermissions`; JSON output |\n| `codex` | `ephemeral: true`, `ignoreUserConfig: true`, `skipGitRepoCheck: true`, `askForApproval: never` |\n| `gemini-cli` | `approvalMode: yolo`, `isolateConfig: true`, `skipTrust: true`; `--output-format json` |\n\nSee [docs/suite-config.md](docs/suite-config.md) and each adapter section below for full flag tables.\n\nExit codes: `0` = all expectations passed; `1` = failures; `2` = no expectations or no gradable repetitions (harness failures without trajectories are skipped).\n\nOptional — use [External eval frameworks & custom judges](#external-eval-frameworks--custom-judges) instead of this command.\n\n### `envelope`\n\nBuild the versioned **`EvalRunEnvelope`** (primary eval interchange document) from a harness `report.json`. Optionally merge outcome grades and emit platform-compatible projections.\n\n```bash\nnpx @alis-build/harness-eval envelope report.json --suite examples/basic.yaml --grading grading.json --output envelope.json\n\n# Interchange projections\nnpx @alis-build/harness-eval envelope report.json --projection trajectory --output trajectory.jsonl\nnpx @alis-build/harness-eval envelope report.json --projection instances --output instances.json\nnpx @alis-build/harness-eval envelope report.json --projection instances --output instances.jsonl\n```\n\n| Option                                                      | Description                                               |\n| ----------------------------------------------------------- | --------------------------------------------------------- |\n| `--output <path>`                                           | Write output (stdout if omitted)                          |\n| `--grading <path>`                                          | Merge `grading.json` outcome scores into the envelope     |\n| `--suite <path>`                                            | Suite YAML for provenance (`uri`, `contentHash`)          |\n| `--projection envelope\\|trajectory\\|instances` | Output shape (default: `envelope`)                        |\n| `--include-raw-stream-events`                               | Include adapter raw stream events in repetition artifacts |\n| `--no-transcript`                                           | Omit judge transcript artifacts                           |\n\nExit codes: `0` = envelope built and behavioral pass; `1` = built but behavioral failures; `2` = usage or file errors.\n\n### `pipeline`\n\nOrchestrate **run → grade → envelope** from a unified `suite.yaml` when a `pipeline:` block is present. See [docs/suite-config.md — Pipeline orchestration](docs/suite-config.md#pipeline-orchestration-pipeline).\n\n```bash\nnpx @alis-build/harness-eval pipeline examples/pipeline/\nnpx @alis-build/harness-eval pipeline my-suite/ --steps run,grade\n```\n\n| Option | Description |\n| ------ | ----------- |\n| `--steps run,grade,envelope` | Subset of configured steps (default: all configured) |\n| `--output <path>` | Override `pipeline.run.output` |\n| `--report <path>` | Override report input for grade/envelope |\n| `--grading <path>` | Override grading input for envelope |\n| `--grading-output <path>` | Override `pipeline.grade.output` |\n| `--envelope-output <path>` | Override `pipeline.envelope.output` |\n| `--projection envelope\\|trajectory\\|instances` | Envelope projection |\n| `--max-concurrent <n>` | Parallel harness/judge workers |\n\nExit codes match the first failing step (`run`, `grade`, or `envelope`). Returns `2` when no `pipeline:` block exists.\n\n### `format`\n\nRe-render an existing `report.json` without re-running the harness.\n\n---\n\n## Output artifacts\n\nAfter a typical run:\n\n| File                          | Produced by                        | Purpose                                                        |\n| ----------------------------- | ---------------------------------- | -------------------------------------------------------------- |\n| **`suite.yaml`**              | You                                | Test spec: prompts, matrix, assertions, expectations           |\n| **`report.json`**             | `run --output`                     | `SuiteReport` — trajectories, assertion stats, per-rep details |\n| **`grading.json`**            | `grade --output`                   | Outcome scores with evidence (optional; or use external judge) |\n| **`envelope.json`**           | `envelope --output`                | Versioned `EvalRunEnvelope` for DB / API / eval platforms      |\n| **`trajectory.jsonl`**        | `envelope --projection trajectory` | Tabular interchange rows (JSONL)                               |\n| **`schemas/*.schema.json`**   | `pnpm run generate-schemas`        | JSON Schema for validators and OpenAPI                         |\n| **`otel-traces/*.otlp.json`** | `run --otel-output`                | OTLP for trace UIs (optional; not the eval contract)           |\n\nWrite artifact paths with `--output` (and `--otel-output` for traces) wherever your pipeline or CI expects them.\n\nSee [Data contracts & schemas](#data-contracts--schemas) for type details.\n\n---\n\n## Suite concepts\n\n**Authoring reference:** [docs/suite-config.md](docs/suite-config.md) — complete field list for suite YAML, matrix cells, test cases, reference trajectories, and grading config.\n\n### Test case\n\nOne prompt + assertions + optional expectations, run N times per matrix cell.\n\n### Matrix cell\n\nOne configuration point (plugin version, model, tool allowlist, etc.). Each (case × cell) is one row in the report.\n\n### Config merge order\n\nLater wins: `defaultConfig` → `case.config` → `cell.config`.\n\nList fields like `allowedTools` and `pluginDirs` are **replaced**, not merged.\n\n### Thresholds\n\n```yaml\nassertions:\n  - called: mcp__api__search_skills\n    threshold: 0.8 # pass if ≥80% of reps call the tool\n```\n\nDefault threshold is `1.0` (every evaluated rep must pass). Reps where the harness crashes are excluded from the denominator and counted as `adapterErrors`.\n\n### Reference trajectory (optional)\n\nDefine expected tool calls for Vertex trajectory metrics on the eval envelope. Use `tool_name_mode: bare` when reference steps use short tool names but the harness records MCP-prefixed names. See [docs/suite-config.md — Reference trajectory](docs/suite-config.md#reference-trajectory).\n\n**Full reference:** [docs/assertions.md](docs/assertions.md) — all assertion kinds, predicates, statistical model, and how to add new assertion types or harness adapters.\n\n---\n\n## Harness adapters\n\nBuilt-in adapters register at module load. Each has a dedicated section below with CLI flag mapping, examples, and judge configuration.\n\n| Adapter | Suite key | Example suite | Example judge |\n| ------- | --------- | ------------- | ------------- |\n| Claude Code | `claudeCode` | [`examples/basic.yaml`](examples/basic.yaml) | [`examples/grading.yaml`](examples/grading.yaml) |\n| Codex CLI | `codex` | [`examples/codex-basic.yaml`](examples/codex-basic.yaml) | [`examples/codex-grading.yaml`](examples/codex-grading.yaml) |\n| Gemini CLI | `geminiCli` | [`examples/gemini-cli-basic.yaml`](examples/gemini-cli-basic.yaml) | [`examples/gemini-grading.yaml`](examples/gemini-grading.yaml) |\n\nAdditional harnesses (e.g. Antigravity CLI) plug in via the same pattern:\n\n1. Implement `HarnessAdapter` under `src/adapters/<id>/` with a `run(config)` that returns a `TrajectoryView`.\n2. Add a nested config key on `SuiteConfig` (e.g. `codex: { ... }`) for harness-specific options.\n3. Call `registerAdapter(\"<id>\", adapter)` at startup (built-in registration in `src/adapters/registry.ts`, or from plugin bootstrap code).\n4. Set `adapter: <id>` in suite YAML; the runner resolves via `getAdapter(id)`.\n\n```typescript\nimport {\n  registerAdapter,\n  listAdapters,\n  getAdapter,\n} from \"@alis-build/harness-eval\";\n\nregisterAdapter(\"my-harness\", myAdapter);\nconsole.log(listAdapters()); // [\"claude-code\", \"codex\", \"gemini-cli\", …]\n```\n\nDuplicate registration throws so accidental overrides fail fast during startup or tests.\n\n---\n\n## Claude Code adapter\n\nNested under `claudeCode` in YAML (or flat in programmatic config). Maps to [Claude Code CLI flags](https://code.claude.com/docs/en/cli-reference#cli-flags).\n\nThe adapter always passes `-p`, `--output-format stream-json`, and `--verbose`.\n\n| Field                             | CLI flag                               | Notes                                                                    |\n| --------------------------------- | -------------------------------------- | ------------------------------------------------------------------------ |\n| `binary`                          | —                                      | Default `claude`                                                         |\n| `pluginDirs`                      | `--plugin-dir`                         | Repeatable                                                               |\n| `pluginUrls`                      | `--plugin-url`                         | Repeatable                                                               |\n| `addDirs`                         | `--add-dir`                            | Extra readable dirs (repeatable)                                         |\n| `mcpConfig`                       | `--mcp-config`                         | MCP config file path                                                     |\n| `strictMcpConfig`                 | `--strict-mcp-config`                  | Only MCP servers from `mcpConfig`                                        |\n| `model`                           | `--model`                              | Also settable at top level                                               |\n| `permissionMode`                  | `--permission-mode`                    | `default`, `acceptEdits`, `plan`, `auto`, `dontAsk`, `bypassPermissions` |\n| `effort`                          | `--effort`                             | `low` … `max`                                                            |\n| `agent`                           | `--agent`                              | Subagent for session                                                     |\n| `fallbackModel`                   | `--fallback-model`                     | Comma-separated fallback chain                                           |\n| `tools`                           | `--tools`                              | Restrict built-in tools (`Bash,Edit,Read` or `default`)                  |\n| `allowedTools`                    | `--allowedTools`                       | Auto-approve tool patterns                                               |\n| `disallowedTools`                 | `--disallowedTools`                    | Deny tool patterns                                                       |\n| `maxTurns`                        | `--max-turns`                          | Print-mode turn cap                                                      |\n| `maxBudgetUsd`                    | `--max-budget-usd`                     | Print-mode spend cap                                                     |\n| `settings`                        | `--settings`                           | Settings JSON file path or inline JSON string                            |\n| `settingSources`                  | `--setting-sources`                    | e.g. `user,project`                                                      |\n| `systemPrompt`                    | `--system-prompt`                      | Replace default system prompt                                            |\n| `systemPromptFile`                | `--system-prompt-file`                 | Replace from file                                                        |\n| `appendSystemPrompt`              | `--append-system-prompt`               | Append to default prompt                                                 |\n| `appendSystemPromptFile`          | `--append-system-prompt-file`          | Append from file                                                         |\n| `debug`                           | `--debug`                              | `true` or category filter string                                         |\n| `debugFile`                       | `--debug-file`                         | Debug log path                                                           |\n| `includeHookEvents`               | `--include-hook-events`                | Hook events in stream-json                                               |\n| `noSessionPersistence`            | `--no-session-persistence`             | Don't save session to disk                                               |\n| `disableSlashCommands`            | `--disable-slash-commands`             | Disable skills/commands for session                                      |\n| `bare`                            | `--bare`                               | Skip auto-discovery (hooks, skills, plugins, MCP)                        |\n| `safeMode`                        | `--safe-mode`                          | Disable customizations for troubleshooting                               |\n| `dangerouslySkipPermissions`      | `--dangerously-skip-permissions`       | Same as `bypassPermissions` mode                                         |\n| `allowDangerouslySkipPermissions` | `--allow-dangerously-skip-permissions` | Add bypass to mode cycle                                                 |\n| `isolateConfig`                   | —                                      | `false` = use your login/plugins; `true` (default) = fresh temp config   |\n\nGeneric `cwd` sets the child process working directory (not a Claude flag). Relative paths in `mcpConfig`, `pluginDirs`, `addDirs`, and settings/prompt files resolve against the suite YAML directory.\n\nNot wired (eval usually starts fresh sessions): `--resume`, `--continue`, `--session-id`, `--worktree`, interactive-only flags.\n\nThe adapter captures Claude’s stream-json output and builds a `TrajectoryView`. Unknown stream events are ignored so schema evolution does not break CI.\n\n---\n\n## Codex CLI adapter\n\nNested under `codex` in YAML (or flat in programmatic config). Maps to [Codex CLI reference](https://developers.openai.com/codex/cli/reference) (`codex exec` flags).\n\nThe harness adapter invokes:\n\n```bash\ncodex --ask-for-approval never exec --json [exec flags…] \"<prompt>\"\n```\n\n`--ask-for-approval` is a **global** flag (before `exec`); other options attach to the `exec` subcommand.\n\n| Field | CLI flag | Notes |\n| ----- | -------- | ----- |\n| `binary` | — | Default `codex` |\n| `model` | `--model` | Also settable at top level |\n| `profile` | `--profile` | Layer `$CODEX_HOME/<profile>.config.toml` |\n| `sandbox` | `--sandbox` | `read-only`, `workspace-write`, `danger-full-access` |\n| `addDirs` | `--add-dir` | Extra writable dirs (repeatable) |\n| `configOverrides` | `-c key=value` | Inline TOML overrides (repeatable) |\n| `askForApproval` | `--ask-for-approval` | Default `never` for non-interactive eval |\n| `dangerouslyBypassApprovalsAndSandbox` | `--yolo` | Hardened CI only |\n| `dangerouslyBypassHookTrust` | `--dangerously-bypass-hook-trust` | Automation with vetted hooks |\n| `ephemeral` | `--ephemeral` | No session rollout files |\n| `ignoreUserConfig` | `--ignore-user-config` | Skip `$CODEX_HOME/config.toml` |\n| `skipGitRepoCheck` | `--skip-git-repo-check` | Allow runs outside git repos |\n| `outputSchema` | `--output-schema` | JSON Schema for structured final output |\n| `outputLastMessage` | `--output-last-message` | Write final assistant message to file (auto temp path when `captureLastMessage` is true) |\n| `captureLastMessage` | — | Default `true`: auto `--output-last-message` and read into `finalResponse` if JSONL has no assistant text |\n| `isolateConfig` | — | `false` (default) = inherit `~/.codex`; `true` = temp `$CODEX_HOME` per run |\n\nGeneric `cwd` sets the child process working directory (`--cd`). MCP tool calls in Codex `--json` output map to harness names `mcp__<server>__<tool>`; shell commands map to `Bash`.\n\nThe adapter maps Codex JSONL events into the shared `StreamEvent` shape and feeds `TrajectoryBuilder`. Fixture-driven tests use committed recordings under `tests/fixtures/codex/` — CI does not require `codex` on `PATH`.\n\n**Example suite:** [examples/codex-basic.yaml](examples/codex-basic.yaml)\n\n**Codex judge:** set `judge.adapter: codex` and nest options under `judge.codex` in grading YAML (see [docs/suite-config.md](docs/suite-config.md)).\n\n**Package export:** `@alis-build/harness-eval/adapters/codex`\n\n---\n\n## Gemini CLI adapter\n\nNested under `geminiCli` in YAML (or flat in programmatic config). Maps to [Gemini CLI reference](https://geminicli.com/docs/cli/cli-reference/).\n\nThe harness adapter invokes:\n\n```bash\ngemini -p \"<prompt>\" --output-format stream-json --approval-mode yolo [flags…]\n```\n\n| Field | CLI flag | Notes |\n| ----- | -------- | ----- |\n| `binary` | — | Default `gemini` |\n| `model` | `--model` | Also settable at top level |\n| `approvalMode` | `--approval-mode` | Default `yolo`; overridable: `default`, `auto_edit`, `plan` |\n| `sandbox` | `--sandbox` | Sandboxed execution |\n| `skipTrust` | `--skip-trust` | Default `true` for harness and judge — skips folder trust in headless runs |\n| `includeDirectories` | `--include-directories` | Extra workspace dirs (repeatable) |\n| `allowedMcpServerNames` | `--allowed-mcp-server-names` | MCP server allowlist |\n| `extensions` | `--extensions` | Extension allowlist |\n| `debug` | `--debug` | Verbose logging |\n| `isolateConfig` | — | `false` (default) = inherit caller config; `true` = temp config dir per run |\n\nMCP tool calls map to harness names `mcp__<server>__<tool>`; built-in Gemini tools keep native names (e.g. `Bash`, `read_file`).\n\nThe adapter maps Gemini stream-json events into the shared `StreamEvent` shape and feeds `TrajectoryBuilder`. Fixture-driven tests use committed recordings under `tests/fixtures/gemini-cli/` — CI does not require `gemini` on `PATH`.\n\n**Example suite:** [examples/gemini-cli-basic.yaml](examples/gemini-cli-basic.yaml)\n\n**Gemini CLI judge:** set `judge.adapter: gemini-cli` and nest options under `judge.geminiCli` in grading YAML (see [docs/suite-config.md](docs/suite-config.md)). Example: [examples/gemini-grading.yaml](examples/gemini-grading.yaml).\n\n**Package export:** `@alis-build/harness-eval/adapters/gemini-cli`\n\n---\n\n## Library API\n\n```typescript\nimport {\n  loadSuite,\n  loadSuiteDocument,\n  runSuite,\n  runPipeline,\n  gradeReport,\n  buildEvalRunEnvelope,\n  trajectoryToTranscript,\n  trajectoryToOtlp,\n  resolveGradeOptions,\n  gradingReportPassed,\n} from \"@alis-build/harness-eval\";\nimport { loadGradingConfig } from \"@alis-build/harness-eval/config\";\n\n// Unified pipeline\nconst doc = await loadSuiteDocument(\"./examples/pipeline/suite.yaml\");\nconst { exitCode } = await runPipeline(doc, { maxConcurrent: 2 });\n\n// Or step-by-step\nconst suite = await loadSuite(\"./examples/basic.yaml\");\nconst report = await runSuite(suite, { maxConcurrent: 2 });\n\nconst gradingConfig = await loadGradingConfig(\"./examples/grading.yaml\");\nconst gradeOpts = resolveGradeOptions(gradingConfig, { maxConcurrent: 2 });\nconst grading = await gradeReport(report, gradeOpts);\n\n// Export trajectory for custom tooling\nconst view = report.cells[0].repetitions[0].adapterResult?.view;\nif (view) {\n  const transcript = trajectoryToTranscript(\n    view,\n    \"Read README.md and summarize harness-eval.\",\n  );\n  const otlp = trajectoryToOtlp(view, { prompt: \"...\" });\n}\n\n// Build versioned envelope for DB / CI (see docs/eval-record.md)\nconst envelope = buildEvalRunEnvelope(report, {\n  grading,\n  suite: { uri: \"./examples/basic.yaml\" },\n});\n```\n\nSubpath exports: `@alis-build/harness-eval/runner`, `@alis-build/harness-eval/config`, `@alis-build/harness-eval/adapters/claude-code`, `@alis-build/harness-eval/adapters/codex`, `@alis-build/harness-eval/adapters/gemini-cli`.\n\n---\n\n## Architecture (brief)\n\n```\nSuite YAML  →  runSuite  →  Harness adapter  →  TrajectoryView\n                                    ↓\n                          assertions (run, in harness-eval)\n                                    ↓\n                          SuiteReport (report.json)\n                                    ↓\n              ┌─────────────────────┴─────────────────────┐\n              ↓                                           ↓\n    harness-eval grade              External judge / eval platform\n    (optional built-in)             (LangSmith, Braintrust, custom)\n              ↓                                           ↓\n              └─────────────────────┬─────────────────────┘\n                                    ↓\n                          EvalRunEnvelope  →  DB / CI / API\n```\n\n- **Pluggable harness adapters** — `claude-code`, `codex`, and `gemini-cli` today; runner and assertions depend only on `TrajectoryView`.\n- **Pluggable outcome layer** — built-in `grade`, custom `gradeFn`, or any external workflow.\n- **OTLP** — observability side export; not required for scoring.\n\nDetails: [Data contracts & schemas](#data-contracts--schemas) · [External eval frameworks](#external-eval-frameworks--custom-judges) · [docs/eval-record.md](docs/eval-record.md)\n\n---\n\n## Development\n\n```bash\npnpm install\npnpm run build\npnpm test              # vitest\npnpm run typecheck\npnpm run generate-schemas   # Zod → schemas/*.schema.json only\n```\n\n**Docs:** [Suite & grading YAML](docs/suite-config.md) · [Assertion DSL & adapter extension](docs/assertions.md) · [Eval record contract (DB / CI)](docs/eval-record.md)\n\n---\n\n## Related work\n\n- [lastmile-ai/mcp-eval](https://github.com/lastmile-ai/mcp-eval) — model + MCP eval (not harness-specific)\n- [alpic-ai/mcp-eval](https://github.com/alpic-ai/mcp-eval) — YAML-driven MCP eval\n- [OpenTelemetry GenAI semantic conventions](https://opentelemetry.io/docs/specs/semconv/registry/attributes/gen-ai/) — OTLP export shape\n\n---\n","readmeFilename":"README.md"}