{"_id":"@cogitator-ai/evals","_rev":"6-7e244ff7940668b357622471e95358ad","name":"@cogitator-ai/evals","dist-tags":{"latest":"0.1.12"},"versions":{"0.1.0":{"name":"@cogitator-ai/evals","version":"0.1.0","license":"MIT","_id":"@cogitator-ai/evals@0.1.0","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"dist":{"shasum":"779331729f9797fd6d5c97b8b31ac63c7800186c","tarball":"https://registry.npmjs.org/@cogitator-ai/evals/-/evals-0.1.0.tgz","fileCount":118,"integrity":"sha512-k+b3CPYtfCauUgSfa8PCI/HtOejYTImXIQ88DxsCGBqBnlQ8lB8vQoMBN//jLLSGire2RcifnbrZb7T5J14S4g==","signatures":[{"sig":"MEYCIQDnGcuCwErnd/KXLzGbr888jQmi7Ear4WKEGau2vtjmRwIhAKzDAZKO4Hf2lmS0c8RmikfoxChCIfstCfBEgBjitAQo","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":157696},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"0513dfb76f47bdb1b5329cc16d5605908d8ba69c","scripts":{"dev":"tsc --watch","test":"vitest run","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"repository":{"url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","type":"git","directory":"packages/evals"},"_npmVersion":"10.9.0","description":"Evaluation framework for Cogitator AI agents","directories":{},"_nodeVersion":"22.11.0","dependencies":{"zod":"^4.3.6","nanoid":"^5.0.4","@cogitator-ai/types":"workspace:*"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^4.0.18","typescript":"^5.7.2","@types/papaparse":"^5.3.15","@cogitator-ai/core":"workspace:*"},"peerDependencies":{"@cogitator-ai/core":"workspace:*"},"optionalDependencies":{"papaparse":"^5.5.0"},"peerDependenciesMeta":{"@cogitator-ai/core":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/evals_0.1.0_1771620891428_0.3965721913706379","host":"s3://npm-registry-packages-npm-production"}},"0.1.2":{"name":"@cogitator-ai/evals","version":"0.1.2","license":"MIT","_id":"@cogitator-ai/evals@0.1.2","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"dist":{"shasum":"10e6824a4010d5aaf7dd425c766912f3a50d1952","tarball":"https://registry.npmjs.org/@cogitator-ai/evals/-/evals-0.1.2.tgz","fileCount":118,"integrity":"sha512-989FZ7ZXI2Ju7IZffGTh5ghoVBlMXbB5HNWAQ2Kqv4pc60ar/63HuxOoqJ/c3aPFGAV+Fo+5NlXjHFQk9TB55w==","signatures":[{"sig":"MEUCIQDzC6Jutz0MoCZdfMmWKtapP3qATfy5cViOPv0fGfL9lwIgQVRt/Xh5q9E34V4480fWoQsz8gooYwJs9Ndo08A8V/Q=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":158890},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"b08564072cd90fe8d50904ef575a5873553d80d2","scripts":{"dev":"tsc --watch","test":"vitest run","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"repository":{"url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","type":"git","directory":"packages/evals"},"_npmVersion":"10.9.0","description":"Evaluation framework for Cogitator AI agents","directories":{},"_nodeVersion":"22.11.0","dependencies":{"zod":"^4.3.6"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^4.0.18","typescript":"^5.7.2","@types/papaparse":"^5.3.15","@cogitator-ai/core":"workspace:*"},"peerDependencies":{"@cogitator-ai/core":"workspace:*"},"optionalDependencies":{"papaparse":"^5.5.0"},"peerDependenciesMeta":{"@cogitator-ai/core":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/evals_0.1.2_1771976955294_0.18825655626338356","host":"s3://npm-registry-packages-npm-production"}},"0.1.4":{"name":"@cogitator-ai/evals","version":"0.1.4","license":"MIT","_id":"@cogitator-ai/evals@0.1.4","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"dist":{"shasum":"a30cbf5e059bdf908ea4decf64bc00e192a6254d","tarball":"https://registry.npmjs.org/@cogitator-ai/evals/-/evals-0.1.4.tgz","fileCount":118,"integrity":"sha512-+vMvc5D7rRN7V3rDMSqPN6StLIzX6ELReAnbqRVBkM/jpQgXR+Ulf4zWyYD9kjXUZXvMcsIQynz1U2owaItDUA==","signatures":[{"sig":"MEQCICap4sLx7LIBl8vpQvRRfdiwX5RUIQx+8whFrqFsPKYxAiB2LDqjDiOgvpl2vXJhKpUeTMDjtuTSglIN0nGfENPm1w==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":158891},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"f508be4a1a30c7bd136f0fa8b2e6cee5521416ab","scripts":{"dev":"tsc --watch","test":"vitest run","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"repository":{"url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","type":"git","directory":"packages/evals"},"_npmVersion":"10.9.0","description":"Evaluation framework for Cogitator AI agents","directories":{},"_nodeVersion":"22.11.0","dependencies":{"zod":"^4.3.6"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^4.0.18","typescript":"^5.7.2","@types/papaparse":"^5.3.15","@cogitator-ai/core":"workspace:*"},"peerDependencies":{"@cogitator-ai/core":"workspace:*"},"optionalDependencies":{"papaparse":"^5.5.0"},"peerDependenciesMeta":{"@cogitator-ai/core":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/evals_0.1.4_1772035536795_0.4303717229238233","host":"s3://npm-registry-packages-npm-production"}},"0.1.10":{"name":"@cogitator-ai/evals","version":"0.1.10","license":"MIT","_id":"@cogitator-ai/evals@0.1.10","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"dist":{"shasum":"fff2094bf3793f4bbf1fa2f649f092ebc724bd6d","tarball":"https://registry.npmjs.org/@cogitator-ai/evals/-/evals-0.1.10.tgz","fileCount":118,"integrity":"sha512-5MVvco2XV+D88c8lYMyfdPMXIFocW4UJt4M4YRSOe03rA3PHLQnhEvW2EO4LwcIA443Rfff28Mz+RMTiYnSI6A==","signatures":[{"sig":"MEQCIEVOAyiQkQ1EHZ9tBJKFRFJgSzPLmB+qxbq4iYR/zEi/AiAGPltzzrLvmAnhSAC6UgZUISs8Jm2oOFBKtsV0trWE9w==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":159462},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"33b6a8cb29bfdd7d2255acfa240b2df5d558354e","scripts":{"dev":"tsc --watch","test":"vitest run","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"repository":{"url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","type":"git","directory":"packages/evals"},"_npmVersion":"10.9.0","description":"Evaluation framework for Cogitator AI agents","directories":{},"_nodeVersion":"22.11.0","dependencies":{"zod":"^4.3.6"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^4.0.18","typescript":"^5.7.2","@types/papaparse":"^5.3.15","@cogitator-ai/core":"workspace:*"},"peerDependencies":{"@cogitator-ai/core":"workspace:*"},"optionalDependencies":{"papaparse":"^5.5.0"},"peerDependenciesMeta":{"@cogitator-ai/core":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/evals_0.1.10_1778881334126_0.6702784177207957","host":"s3://npm-registry-packages-npm-production"}},"0.1.11":{"name":"@cogitator-ai/evals","version":"0.1.11","license":"MIT","_id":"@cogitator-ai/evals@0.1.11","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"dist":{"shasum":"5cfb9659b462b7fd3177b7f041d33972b3908af3","tarball":"https://registry.npmjs.org/@cogitator-ai/evals/-/evals-0.1.11.tgz","fileCount":119,"integrity":"sha512-oLpXm7TGaH1eimlX7ynvdLb+XC5yAP7rMXCwK5V3mQuBIuYHGGa+OglPgGg6VJAiSF524lBNHq+Z6I6tgIBrIg==","signatures":[{"sig":"MEUCIQC1z0V4WCv4jRuW9kjFXjNHa1TBpBO+isqcJvpgshSZJwIgcDDrmE/qWmWH3rOh7o8eV7t7CG0LT86WGzPP+cao5Pg=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":160530},"main":"./dist/index.js","type":"module","_from":"file:cogitator-ai-evals-0.1.11.tgz","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"dev":"tsc --watch","test":"vitest run","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"_resolved":"/private/var/folders/20/4d95hdvj45j337gtqpbx3v1m0000gn/T/76dd2a9bbe152f28388fd69d97651108/cogitator-ai-evals-0.1.11.tgz","_integrity":"sha512-oLpXm7TGaH1eimlX7ynvdLb+XC5yAP7rMXCwK5V3mQuBIuYHGGa+OglPgGg6VJAiSF524lBNHq+Z6I6tgIBrIg==","repository":{"url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","type":"git","directory":"packages/evals"},"_npmVersion":"10.9.0","description":"Evaluation framework for Cogitator AI agents","directories":{},"_nodeVersion":"22.11.0","dependencies":{"zod":"^4.3.6"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^4.0.18","typescript":"^5.7.2","@types/papaparse":"^5.3.15","@cogitator-ai/core":"0.19.3"},"peerDependencies":{"@cogitator-ai/core":"0.19.3"},"optionalDependencies":{"papaparse":"^5.5.0"},"peerDependenciesMeta":{"@cogitator-ai/core":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/evals_0.1.11_1778882256231_0.18240587160171318","host":"s3://npm-registry-packages-npm-production"}},"0.1.12":{"name":"@cogitator-ai/evals","version":"0.1.12","description":"Evaluation framework for Cogitator AI agents","type":"module","main":"./dist/index.js","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"dependencies":{"zod":"^4.3.6"},"peerDependencies":{"@cogitator-ai/core":"0.19.4"},"peerDependenciesMeta":{"@cogitator-ai/core":{"optional":true}},"optionalDependencies":{"papaparse":"^5.5.0"},"devDependencies":{"@types/papaparse":"^5.3.15","typescript":"^5.7.2","vitest":"^4.0.18","@cogitator-ai/core":"0.19.4"},"repository":{"type":"git","url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","directory":"packages/evals"},"publishConfig":{"access":"public","registry":"https://npm.pkg.github.com"},"license":"MIT","scripts":{"build":"tsc","dev":"tsc --watch","clean":"rm -rf dist","typecheck":"tsc --noEmit","test":"vitest run","test:watch":"vitest"},"_id":"@cogitator-ai/evals@0.1.12","bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","_integrity":"sha512-gHRK6z/DT7QLmAUkSk/Rj2383lPuyf4GO8N0ftIOLenzbRf+A79GIC5Rn2PFF/fcknYSNgfGs8oVLJRQfYEQGw==","_resolved":"/private/var/folders/20/4d95hdvj45j337gtqpbx3v1m0000gn/T/a015e1bc0b2ab6cd36896e9697f5fd9a/cogitator-ai-evals-0.1.12.tgz","_from":"file:cogitator-ai-evals-0.1.12.tgz","_nodeVersion":"22.23.1","_npmVersion":"10.9.8","dist":{"integrity":"sha512-gHRK6z/DT7QLmAUkSk/Rj2383lPuyf4GO8N0ftIOLenzbRf+A79GIC5Rn2PFF/fcknYSNgfGs8oVLJRQfYEQGw==","shasum":"e2b1cf907eb423bc03ab1e8e91996502b9aa6288","tarball":"https://registry.npmjs.org/@cogitator-ai/evals/-/evals-0.1.12.tgz","fileCount":119,"unpackedSize":160530,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEYCIQDPgV3eqGGyDltaHraPgO6kXAPD9K0PYgkBPDrHLSH4UQIhAMRyOJbjifdOBpWDOwT5UPB4WERVx48L/5+Oz/lMvkb+"}]},"_npmUser":{"name":"el1fe","email":"piuro.pavel@gmail.com"},"directories":{},"maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/evals_0.1.12_1785059665779_0.45357254533783586"},"_hasShrinkwrap":false}},"time":{"created":"2026-02-20T20:54:51.300Z","modified":"2026-07-26T09:54:26.098Z","0.1.0":"2026-02-20T20:54:51.580Z","0.1.2":"2026-02-24T23:49:15.434Z","0.1.4":"2026-02-25T16:05:36.966Z","0.1.10":"2026-05-15T21:42:14.259Z","0.1.11":"2026-05-15T21:57:36.387Z","0.1.12":"2026-07-26T09:54:25.936Z"},"bugs":{"url":"https://github.com/cogitator-ai/Cogitator-AI/issues"},"license":"MIT","homepage":"https://github.com/cogitator-ai/Cogitator-AI#readme","repository":{"type":"git","url":"git+https://github.com/cogitator-ai/Cogitator-AI.git","directory":"packages/evals"},"description":"Evaluation framework for Cogitator AI agents","maintainers":[{"name":"el1fe","email":"piuro.pavel@gmail.com"}],"readme":"# @cogitator-ai/evals\n\nEvaluation framework for Cogitator AI agents. Run eval suites, compare models with A/B tests, enforce quality thresholds, and track regressions — all with built-in statistical significance testing.\n\n## Installation\n\n```bash\npnpm add @cogitator-ai/evals\n\n# Optional dependencies\npnpm add papaparse  # CSV dataset loading\n```\n\n## Features\n\n- **EvalSuite** — Run datasets against agents or plain functions with configurable concurrency, timeouts, and retries\n- **4 Deterministic Metrics** — exactMatch, contains, regex, jsonSchema (Zod)\n- **5 LLM-as-Judge Metrics** — faithfulness, relevance, coherence, helpfulness, custom llmMetric\n- **3 Statistical Metrics** — latency, cost, tokenUsage with full percentile breakdowns\n- **Custom Metrics** — `metric()` factory for anything domain-specific\n- **Assertions** — threshold, noRegression, custom assertion with auto-detection of lower-is-better metrics\n- **A/B Testing** — EvalComparison with paired t-test and McNemar's test for statistical significance\n- **4 Reporters** — console (colored table), JSON, CSV, CI (exit code on failure)\n- **Builder API** — Fluent `EvalBuilder` for composable eval pipelines\n- **Baseline Workflow** — Save baselines, compare against them, catch regressions in CI\n- **Zod Validation** — Type-safe configuration with runtime checks\n\n---\n\n## Quick Start\n\n```typescript\nimport { EvalSuite, Dataset, exactMatch, contains, threshold, latency } from '@cogitator-ai/evals';\n\nconst dataset = Dataset.from([\n  { input: 'What is 2+2?', expected: '4' },\n  { input: 'Capital of France?', expected: 'Paris' },\n  { input: 'Largest planet?', expected: 'Jupiter' },\n]);\n\nconst suite = new EvalSuite({\n  dataset,\n  target: {\n    fn: async (input) => {\n      // replace with your agent or LLM call\n      return `The answer is ${input}`;\n    },\n  },\n  metrics: [exactMatch(), contains()],\n  statisticalMetrics: [latency()],\n  assertions: [threshold('exactMatch', 0.8)],\n  concurrency: 5,\n  timeout: 30_000,\n});\n\nconst result = await suite.run();\n\nresult.report('console');\nresult.saveBaseline('./baseline.json');\n```\n\n---\n\n## Datasets\n\nDatasets are immutable collections of eval cases. Each case has an `input`, optional `expected`, optional `context`, and optional `metadata`.\n\n### From inline data\n\n```typescript\nimport { Dataset } from '@cogitator-ai/evals';\n\nconst dataset = Dataset.from([\n  { input: 'Translate hello to French', expected: 'Bonjour' },\n  { input: 'Summarize this article', context: { article: '...' } },\n]);\n```\n\n### From JSONL\n\n```typescript\nconst dataset = await Dataset.fromJsonl('./evals/qa.jsonl');\n```\n\nEach line must be a JSON object with at least an `input` field:\n\n```jsonl\n{\"input\": \"What is TypeScript?\", \"expected\": \"A typed superset of JavaScript\"}\n{\"input\": \"What is Zod?\", \"expected\": \"A TypeScript-first schema validation library\"}\n```\n\n### From CSV\n\nRequires `papaparse` as an optional dependency.\n\n```typescript\nconst dataset = await Dataset.fromCsv('./evals/qa.csv');\n```\n\nCSV must have an `input` column. Optional columns: `expected`, `metadata.*`, `context.*`.\n\n### Transformations\n\n```typescript\nconst filtered = dataset.filter((c) => c.expected !== undefined);\nconst sampled = dataset.sample(50);\nconst shuffled = dataset.shuffle();\n```\n\nAll transformations return new `Dataset` instances — the original is never mutated.\n\n---\n\n## Metrics\n\n### Deterministic\n\nBinary (0 or 1) metrics that compare output against expected values.\n\n| Metric       | Description                                | Requires `expected` |\n| ------------ | ------------------------------------------ | ------------------- |\n| `exactMatch` | Exact string match (case optional)         | Yes                 |\n| `contains`   | Output contains expected substring         | Yes                 |\n| `regex`      | Output matches a regex pattern             | No                  |\n| `jsonSchema` | Output is valid JSON matching a Zod schema | No                  |\n\n```typescript\nimport { exactMatch, contains, regex, jsonSchema } from '@cogitator-ai/evals';\nimport { z } from 'zod';\n\nconst metrics = [\n  exactMatch({ caseSensitive: true }),\n  contains(),\n  regex(/\\d{4}-\\d{2}-\\d{2}/),\n  jsonSchema(z.object({ answer: z.string(), confidence: z.number() })),\n];\n```\n\n### LLM-as-Judge\n\nMetrics scored by an LLM judge (0.0 to 1.0). Require a `judge` config on the suite.\n\n| Metric         | Evaluates                               |\n| -------------- | --------------------------------------- |\n| `faithfulness` | Factual accuracy relative to input      |\n| `relevance`    | How on-topic the response is            |\n| `coherence`    | Logical structure and readability       |\n| `helpfulness`  | Practical usefulness to the user        |\n| `llmMetric`    | Custom prompt — you define the criteria |\n\n```typescript\nimport { faithfulness, relevance, llmMetric } from '@cogitator-ai/evals';\n\nconst suite = new EvalSuite({\n  dataset,\n  target: { fn: myFunction },\n  metrics: [\n    faithfulness(),\n    relevance(),\n    llmMetric({\n      name: 'technicalAccuracy',\n      prompt: 'Rate how technically accurate the response is for a software engineering audience.',\n    }),\n  ],\n  judge: { model: 'gpt-4o', temperature: 0 },\n});\n```\n\n### Statistical\n\nAggregate metrics computed across all results. These report percentile breakdowns (p50, p95, p99) rather than per-case scores.\n\n```typescript\nimport { latency, cost, tokenUsage } from '@cogitator-ai/evals';\n\nconst suite = new EvalSuite({\n  dataset,\n  target: { agent, cogitator },\n  metrics: [exactMatch()],\n  statisticalMetrics: [latency(), cost(), tokenUsage()],\n});\n```\n\n### Custom\n\nBuild domain-specific metrics with the `metric()` factory.\n\n```typescript\nimport { metric } from '@cogitator-ai/evals';\n\nconst wordCount = metric({\n  name: 'wordCount',\n  evaluate: ({ output }) => {\n    const count = output.split(/\\s+/).length;\n    return { score: Math.min(count / 100, 1), details: `${count} words` };\n  },\n});\n\nconst suite = new EvalSuite({\n  dataset,\n  target: { fn: myFunction },\n  metrics: [wordCount],\n});\n```\n\nScores are automatically clamped to [0, 1].\n\n---\n\n## Assertions\n\nAssertions check aggregated metrics after a suite run and produce pass/fail results.\n\n### threshold\n\nEnforces a minimum (or maximum for latency/cost) value on a metric's mean.\n\n```typescript\nimport { threshold } from '@cogitator-ai/evals';\n\nconst assertions = [\n  threshold('exactMatch', 0.9),\n  threshold('latency', 5000),\n  threshold('relevance', 0.7),\n];\n```\n\nLatency and cost metrics are automatically detected as lower-is-better.\n\n### noRegression\n\nCompares current results against a saved baseline file.\n\n```typescript\nimport { noRegression } from '@cogitator-ai/evals';\n\nconst assertions = [noRegression('./baseline.json', { tolerance: 0.05 })];\n```\n\n### Custom assertion\n\n```typescript\nimport { assertion } from '@cogitator-ai/evals';\n\nconst assertions = [\n  assertion({\n    name: 'totalCostBudget',\n    check: (_aggregated, stats) => stats.cost < 1.0,\n    message: 'Total eval cost exceeded $1.00 budget',\n  }),\n];\n```\n\n---\n\n## A/B Testing\n\n`EvalComparison` runs two targets on the same dataset and determines a winner using statistical significance tests (paired t-test for continuous metrics, McNemar's test for binary metrics).\n\n```typescript\nimport { EvalComparison, Dataset, exactMatch, contains } from '@cogitator-ai/evals';\n\nconst dataset = Dataset.from([\n  { input: 'What is 2+2?', expected: '4' },\n  { input: 'Capital of Japan?', expected: 'Tokyo' },\n  { input: 'Boiling point of water?', expected: '100°C' },\n]);\n\nconst comparison = new EvalComparison({\n  dataset,\n  targets: {\n    baseline: { fn: async (input) => baselineModel(input) },\n    challenger: { fn: async (input) => challengerModel(input) },\n  },\n  metrics: [exactMatch(), contains()],\n  concurrency: 5,\n});\n\nconst result = await comparison.run();\n\nconsole.log(`Winner: ${result.summary.winner}`);\nfor (const [name, mc] of Object.entries(result.summary.metrics)) {\n  console.log(\n    `  ${name}: baseline=${mc.baseline.toFixed(3)} challenger=${mc.challenger.toFixed(3)} p=${mc.pValue.toFixed(4)} ${mc.significant ? '*' : ''}`\n  );\n}\n```\n\nAccess full suite results via `result.baseline` and `result.challenger`.\n\n---\n\n## Reporters\n\nCall `result.report()` after a suite run to output results.\n\n| Reporter  | Output                                          |\n| --------- | ----------------------------------------------- |\n| `console` | Colored table with metrics, assertions, summary |\n| `json`    | Writes `eval-report.json` (configurable path)   |\n| `csv`     | Writes `eval-report.csv` (configurable path)    |\n| `ci`      | Compact output, `process.exit(1)` on failure    |\n\n```typescript\nconst result = await suite.run();\n\nresult.report('console');\nresult.report('json', { path: './reports/eval.json' });\nresult.report(['console', 'json', 'csv']);\nresult.report('ci');\n```\n\n---\n\n## Builder API\n\n`EvalBuilder` provides a fluent interface for constructing eval suites.\n\n```typescript\nimport {\n  EvalBuilder,\n  Dataset,\n  exactMatch,\n  contains,\n  faithfulness,\n  latency,\n  threshold,\n  noRegression,\n} from '@cogitator-ai/evals';\n\nconst suite = new EvalBuilder()\n  .withDataset(await Dataset.fromJsonl('./evals/qa.jsonl'))\n  .withTarget({ fn: async (input) => myModel(input) })\n  .withMetrics([exactMatch(), contains(), faithfulness()])\n  .withStatisticalMetrics([latency()])\n  .withJudge({ model: 'gpt-4o', temperature: 0 })\n  .withAssertions([threshold('exactMatch', 0.85), noRegression('./baseline.json')])\n  .withConcurrency(10)\n  .withTimeout(60_000)\n  .withRetries(2)\n  .onProgress(({ completed, total }) => {\n    console.log(`${completed}/${total}`);\n  })\n  .build();\n\nconst result = await suite.run();\nresult.report('console');\n```\n\n---\n\n## Baseline Workflow\n\nSave a baseline after a successful run, then use `noRegression` to guard against regressions in CI.\n\n```typescript\nconst result = await suite.run();\n\nresult.saveBaseline('./baseline.json');\n```\n\nThe baseline file is a simple JSON map of metric names to mean scores:\n\n```json\n{\n  \"exactMatch\": 0.92,\n  \"contains\": 0.97,\n  \"latency\": 1234\n}\n```\n\nIn subsequent runs, use `noRegression` to compare:\n\n```typescript\nconst suite = new EvalSuite({\n  dataset,\n  target: { fn: myFunction },\n  metrics: [exactMatch(), contains()],\n  assertions: [noRegression('./baseline.json', { tolerance: 0.05 })],\n});\n\nconst result = await suite.run();\nresult.report('ci');\n```\n\n---\n\n## API Reference\n\n### Core\n\n| Export           | Description                                                         |\n| ---------------- | ------------------------------------------------------------------- |\n| `EvalSuite`      | Main evaluation runner                                              |\n| `EvalComparison` | A/B testing runner with statistical significance                    |\n| `EvalBuilder`    | Fluent builder for EvalSuite                                        |\n| `Dataset`        | Immutable dataset with from/fromJsonl/fromCsv/filter/sample/shuffle |\n| `loadJsonl`      | Low-level JSONL file loader                                         |\n| `loadCsv`        | Low-level CSV file loader                                           |\n| `isLLMMetric`    | Type guard: checks if a `MetricFn` is an `LLMMetricFn`              |\n\n### Metrics\n\n| Export         | Type          | Description                         |\n| -------------- | ------------- | ----------------------------------- |\n| `exactMatch`   | Deterministic | Exact string match                  |\n| `contains`     | Deterministic | Substring match                     |\n| `regex`        | Deterministic | Regex pattern match                 |\n| `jsonSchema`   | Deterministic | Zod schema validation               |\n| `faithfulness` | LLM Judge     | Factual accuracy                    |\n| `relevance`    | LLM Judge     | Topical relevance                   |\n| `coherence`    | LLM Judge     | Logical structure                   |\n| `helpfulness`  | LLM Judge     | Practical usefulness                |\n| `llmMetric`    | LLM Judge     | Custom judge prompt                 |\n| `latency`      | Statistical   | Response time percentiles           |\n| `cost`         | Statistical   | Token cost aggregation              |\n| `tokenUsage`   | Statistical   | Input/output token counts           |\n| `metric`       | Custom        | Factory for domain-specific metrics |\n\n### Assertions\n\n| Export         | Description                                    |\n| -------------- | ---------------------------------------------- |\n| `threshold`    | Enforce min/max on metric mean                 |\n| `noRegression` | Compare against saved baseline                 |\n| `assertion`    | Custom assertion with arbitrary check function |\n\n### Reporters\n\n| Export   | Description                            |\n| -------- | -------------------------------------- |\n| `report` | Dispatch to one or more reporter types |\n\n### Statistics\n\n| Export         | Description                                               |\n| -------------- | --------------------------------------------------------- |\n| `pairedTTest`  | Paired t-test for continuous metric comparison            |\n| `mcnemarsTest` | McNemar's test for binary metric comparison               |\n| `mean`         | Arithmetic mean                                           |\n| `median`       | Median value                                              |\n| `stdDev`       | Sample standard deviation                                 |\n| `percentile`   | Arbitrary percentile                                      |\n| `aggregate`    | Full stats: mean, median, min, max, stdDev, p50, p95, p99 |\n\n### Agent Tools\n\n| Export              | Description                          |\n| ------------------- | ------------------------------------ |\n| `createRunEvalTool` | Creates a `run_eval` tool for agents |\n| `evalTools`         | Returns all eval tools as an array   |\n\n---\n\n## License\n\nMIT\n","readmeFilename":"README.md"}