{"_id":"@basalt-ai/cobalt","_rev":"3-53e271e4f54494321700ac45422e2f69","name":"@basalt-ai/cobalt","dist-tags":{"latest":"0.3.0"},"versions":{"0.1.0":{"name":"@basalt-ai/cobalt","version":"0.1.0","keywords":["ai","testing","evaluation","llm","agents","benchmark","llm-judge","ai-testing","agent-testing"],"license":"MIT","_id":"@basalt-ai/cobalt@0.1.0","maintainers":[{"name":"batmot1","email":"thomas@getbasalt.ai"},{"name":"will-m","email":"tech@getbasalt.ai"},{"name":"theocousin","email":"cousin.theophile@gmail.com"},{"name":"basalt.zakaria","email":"zakaria@getbasalt.ai"}],"homepage":"https://github.com/basalt-ai/cobalt#readme","bugs":{"url":"https://github.com/basalt-ai/cobalt/issues"},"bin":{"cobalt":"dist/cli.js"},"dist":{"shasum":"a9a3037823603408420440a6bf1d6ad850c58e06","tarball":"https://registry.npmjs.org/@basalt-ai/cobalt/-/cobalt-0.1.0.tgz","fileCount":12,"integrity":"sha512-mpmX4lo6tfcaCrooHuE1cDpKceKL217TWUbxBaFZKHLgHv2PdQ3oCCWnYkHDg/7/zQwAyiiVb/uipVh6CJ9D5g==","signatures":[{"sig":"MEUCIBh7OLA/RD//xwYguDWEENT91fDawEK+O2ky+/ksaCd6AiEA7n50M5Jf8+LvQA+mYXYzABWOT5EH0s9nPI0MS/F+bSY=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":276592},"main":"./dist/index.cjs","type":"module","_from":"file:basalt-ai-cobalt-0.1.0.tgz","types":"./dist/index.d.ts","module":"./dist/index.mjs","engines":{"node":">=18"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.mjs","require":"./dist/index.cjs"}},"scripts":{"dev":"tsup --watch","lint":"biome check .","test":"vitest run","build":"pnpm build:lib","check":"biome check --write .","format":"biome format --write .","build:lib":"tsup","test:watch":"vitest","test:coverage":"vitest run --coverage"},"_npmUser":{"name":"will-m","email":"tech@getbasalt.ai"},"_resolved":"/tmp/c0e66f6aea00e7fd64de2c21a11a2ab1/basalt-ai-cobalt-0.1.0.tgz","_integrity":"sha512-mpmX4lo6tfcaCrooHuE1cDpKceKL217TWUbxBaFZKHLgHv2PdQ3oCCWnYkHDg/7/zQwAyiiVb/uipVh6CJ9D5g==","repository":{"url":"git+https://github.com/basalt-ai/cobalt.git","type":"git"},"_npmVersion":"10.8.2","description":"Jest for AI Agents — test, evaluate, and track your AI experiments","directories":{},"_nodeVersion":"20.20.0","dependencies":{"ora":"^8.1.1","zod":"^3.24.1","defu":"^6.1.4","hono":"^4.11.7","jiti":"^2.4.2","citty":"^0.1.6","p-map":"^7.0.3","pathe":"^1.1.2","openai":"^4.77.3","p-limit":"^6.1.0","autoevals":"^0.0.131","picocolors":"^1.1.1","better-sqlite3":"^12.6.2","@anthropic-ai/sdk":"^0.30.0","@hono/node-server":"^1.19.9","@modelcontextprotocol/sdk":"^1.26.0"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.3.5","vitest":"^2.1.8","typescript":"^5.7.2","@types/node":"^22.10.5","@biomejs/biome":"^1.9.4","@cobalt/tsconfig":"0.0.0","@vitest/coverage-v8":"^2.1.8","@types/better-sqlite3":"^7.6.13"},"_npmOperationalInternal":{"tmp":"tmp/cobalt_0.1.0_1770707338132_0.6471982785604917","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@basalt-ai/cobalt","version":"0.2.0","keywords":["ai","testing","evaluation","llm","agents","benchmark","llm-judge","ai-testing","agent-testing"],"license":"MIT","_id":"@basalt-ai/cobalt@0.2.0","maintainers":[{"name":"batmot1","email":"thomas@getbasalt.ai"},{"name":"will-m","email":"tech@getbasalt.ai"},{"name":"theocousin","email":"cousin.theophile@gmail.com"},{"name":"basalt.zakaria","email":"zakaria@getbasalt.ai"}],"homepage":"https://github.com/basalt-ai/cobalt#readme","bugs":{"url":"https://github.com/basalt-ai/cobalt/issues"},"bin":{"cobalt":"dist/cli.js"},"dist":{"shasum":"4991698b119ba6e13f186fa42715d1eaf9e6e025","tarball":"https://registry.npmjs.org/@basalt-ai/cobalt/-/cobalt-0.2.0.tgz","fileCount":19,"integrity":"sha512-udZZxfnR3xviDhQNjpLtx6Qa1/Ix5FuqYEB1+5mYl4+Ki9wxl2CjAvvIijHWaoYG28MmMjnc8G5ePdxZgVfMRA==","signatures":[{"sig":"MEYCIQCkvLxynLYcY6XpJfE38z2HYNBprTag+Aqgsnv8ZFojOAIhAPjUQFgCExlYrynsSaEpcTCwMytMMmOjfnBxhzG9y1Yk","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1459326},"main":"./dist/index.cjs","type":"module","_from":"file:basalt-ai-cobalt-0.2.0.tgz","types":"./dist/index.d.ts","module":"./dist/index.mjs","engines":{"node":">=18"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.mjs","require":"./dist/index.cjs"}},"scripts":{"dev":"tsup --watch","lint":"biome check .","test":"vitest run","build":"pnpm build:lib && pnpm build:dashboard","check":"biome check --write .","format":"biome format --write .","build:lib":"tsup","test:watch":"vitest","dev:dashboard":"vite -c src/dashboard/ui/vite.config.ts","test:coverage":"vitest run --coverage","build:dashboard":"vite build -c src/dashboard/ui/vite.config.ts"},"_npmUser":{"name":"will-m","email":"tech@getbasalt.ai"},"_resolved":"/tmp/66a440cf92b0f15a5be0d78ef8fbb8dd/basalt-ai-cobalt-0.2.0.tgz","_integrity":"sha512-udZZxfnR3xviDhQNjpLtx6Qa1/Ix5FuqYEB1+5mYl4+Ki9wxl2CjAvvIijHWaoYG28MmMjnc8G5ePdxZgVfMRA==","repository":{"url":"git+https://github.com/basalt-ai/cobalt.git","type":"git"},"_npmVersion":"10.8.2","description":"Unit testing for AI Agents — test, evaluate, and track your AI experiments","directories":{},"_nodeVersion":"20.20.0","dependencies":{"ora":"^8.1.1","zod":"^3.24.1","clsx":"^2.1.1","defu":"^6.1.4","hono":"^4.11.7","jiti":"^2.4.2","citty":"^0.1.6","p-map":"^7.0.3","pathe":"^1.1.2","react":"^19.0.0","dotenv":"^16.6.1","openai":"^5.0.0","p-limit":"^6.1.0","recharts":"^3.7.0","autoevals":"^0.0.131","react-dom":"^19.0.0","picocolors":"^1.1.1","tailwindcss":"^4.1.18","react-router":"^7.0.0","better-sqlite3":"^12.6.2","tailwind-merge":"^3.4.0","@anthropic-ai/sdk":"^0.74.0","@hono/node-server":"^1.19.9","@radix-ui/react-slot":"^1.2.4","@radix-ui/react-tabs":"^1.1.13","@phosphor-icons/react":"^2.1.10","@tanstack/react-table":"^8.21.3","@radix-ui/react-dialog":"^1.1.15","@radix-ui/react-select":"^2.2.6","@radix-ui/react-switch":"^1.2.6","@radix-ui/react-popover":"^1.1.15","@radix-ui/react-tooltip":"^1.2.8","class-variance-authority":"^0.7.1","@modelcontextprotocol/sdk":"^1.26.0","@radix-ui/react-separator":"^1.1.8","@radix-ui/react-scroll-area":"^1.2.10","@radix-ui/react-dropdown-menu":"^2.1.16"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.3.5","vite":"^6.0.0","vitest":"^2.1.8","typescript":"^5.7.2","@types/node":"^22.10.5","@types/react":"^19.0.0","@biomejs/biome":"^1.9.4","@cobalt/tsconfig":"0.0.0","@types/react-dom":"^19.0.0","@tailwindcss/vite":"^4.1.18","@vitest/coverage-v8":"^2.1.8","@vitejs/plugin-react":"^4.3.0","@types/better-sqlite3":"^7.6.13"},"_npmOperationalInternal":{"tmp":"tmp/cobalt_0.2.0_1770914174525_0.002430638527701534","host":"s3://npm-registry-packages-npm-production"}},"0.3.0":{"name":"@basalt-ai/cobalt","version":"0.3.0","description":"Unit testing for AI Agents — test, evaluate, and track your AI experiments","type":"module","main":"./dist/index.cjs","module":"./dist/index.mjs","types":"./dist/index.d.ts","bin":{"cobalt":"dist/cli.js"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.mjs","require":"./dist/index.cjs"}},"keywords":["ai","testing","evaluation","llm","agents","benchmark","llm-judge","ai-testing","agent-testing"],"engines":{"node":">=18"},"dependencies":{"@ai-sdk/anthropic":"^3.0.45","@ai-sdk/openai":"^3.0.29","@ai-sdk/react":"^3.0.92","@anthropic-ai/sdk":"^0.74.0","@hono/node-server":"^1.19.9","@modelcontextprotocol/sdk":"^1.26.0","@phosphor-icons/react":"^2.1.10","@radix-ui/react-dialog":"^1.1.15","@radix-ui/react-dropdown-menu":"^2.1.16","@radix-ui/react-popover":"^1.1.15","@radix-ui/react-scroll-area":"^1.2.10","@radix-ui/react-select":"^2.2.6","@radix-ui/react-separator":"^1.1.8","@radix-ui/react-slot":"^1.2.4","@radix-ui/react-switch":"^1.2.6","@radix-ui/react-tabs":"^1.1.13","@radix-ui/react-tooltip":"^1.2.8","@tanstack/react-table":"^8.21.3","ai":"^6.0.90","autoevals":"^0.0.131","better-sqlite3":"^12.6.2","citty":"^0.1.6","class-variance-authority":"^0.7.1","clsx":"^2.1.1","defu":"^6.1.4","dotenv":"^16.6.1","hono":"^4.11.7","jiti":"^2.4.2","openai":"^5.0.0","ora":"^8.1.1","p-limit":"^6.1.0","p-map":"^7.0.3","pathe":"^1.1.2","picocolors":"^1.1.1","react":"^19.0.0","react-dom":"^19.0.0","react-router":"^7.0.0","recharts":"^3.7.0","tailwind-merge":"^3.4.0","tailwindcss":"^4.1.18","zod":"^3.24.1"},"devDependencies":{"@biomejs/biome":"^1.9.4","@tailwindcss/vite":"^4.1.18","@types/better-sqlite3":"^7.6.13","@types/node":"^22.10.5","@types/react":"^19.0.0","@types/react-dom":"^19.0.0","@vitejs/plugin-react":"^4.3.0","@vitest/coverage-v8":"^2.1.8","tsup":"^8.3.5","typescript":"^5.7.2","vite":"^6.0.0","vitest":"^2.1.8","@cobalt/tsconfig":"0.0.0"},"repository":{"type":"git","url":"git+https://github.com/basalt-ai/cobalt.git"},"license":"MIT","scripts":{"build":"pnpm build:lib && pnpm build:dashboard","build:lib":"tsup","build:dashboard":"vite build -c src/dashboard/ui/vite.config.ts","dev:dashboard":"vite -c src/dashboard/ui/vite.config.ts","dev":"tsup --watch","test":"vitest run","test:watch":"vitest","test:coverage":"vitest run --coverage","lint":"biome check .","format":"biome format --write .","check":"biome check --write ."},"_id":"@basalt-ai/cobalt@0.3.0","bugs":{"url":"https://github.com/basalt-ai/cobalt/issues"},"homepage":"https://github.com/basalt-ai/cobalt#readme","_integrity":"sha512-mwtBBBAH9VM71Pqc2plu1+kKbrO8Zm2a+MNXmpl3aeAo6nolTVEvFD3XxM40XCd83ujR3XY/AXG5u9Sum4eVqA==","_resolved":"/tmp/82545bcbb9abbdaf36d027200c01828e/basalt-ai-cobalt-0.3.0.tgz","_from":"file:basalt-ai-cobalt-0.3.0.tgz","_nodeVersion":"20.20.0","_npmVersion":"10.8.2","dist":{"integrity":"sha512-mwtBBBAH9VM71Pqc2plu1+kKbrO8Zm2a+MNXmpl3aeAo6nolTVEvFD3XxM40XCd83ujR3XY/AXG5u9Sum4eVqA==","shasum":"316bb9a7389cb93414e5199459d3d81ea5cb9f0e","tarball":"https://registry.npmjs.org/@basalt-ai/cobalt/-/cobalt-0.3.0.tgz","fileCount":20,"unpackedSize":1665399,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEMCICkWrdp1K56ERijC4E0ZCjCGc6m7rpo6g9VZKAjc/n7wAh8dHm0unYKub0jX2bv0Ru8myPQ1BpzDCq4P2HWBPPcC"}]},"_npmUser":{"name":"will-m","email":"tech@getbasalt.ai"},"directories":{},"maintainers":[{"name":"batmot1","email":"thomas@getbasalt.ai"},{"name":"will-m","email":"tech@getbasalt.ai"},{"name":"theocousin","email":"cousin.theophile@gmail.com"},{"name":"basalt.zakaria","email":"zakaria@getbasalt.ai"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/cobalt_0.3.0_1771488988178_0.28520513446095097"},"_hasShrinkwrap":false}},"time":{"created":"2026-02-10T07:08:58.033Z","modified":"2026-02-19T08:16:28.542Z","0.1.0":"2026-02-10T07:08:58.281Z","0.2.0":"2026-02-12T16:36:14.698Z","0.3.0":"2026-02-19T08:16:28.390Z"},"bugs":{"url":"https://github.com/basalt-ai/cobalt/issues"},"license":"MIT","homepage":"https://github.com/basalt-ai/cobalt#readme","keywords":["ai","testing","evaluation","llm","agents","benchmark","llm-judge","ai-testing","agent-testing"],"repository":{"type":"git","url":"git+https://github.com/basalt-ai/cobalt.git"},"description":"Unit testing for AI Agents — test, evaluate, and track your AI experiments","maintainers":[{"name":"batmot1","email":"thomas@getbasalt.ai"},{"name":"will-m","email":"tech@getbasalt.ai"},{"name":"theocousin","email":"cousin.theophile@gmail.com"},{"name":"basalt.zakaria","email":"zakaria@getbasalt.ai"}],"readme":"# @basalt-ai/cobalt\n\n**Unit testing for AI Agents** — A TypeScript testing framework for evaluating AI systems.\n\n## Installation\n\n```bash\nnpm install @basalt-ai/cobalt\n# or\npnpm add @basalt-ai/cobalt\n# or\nyarn add @basalt-ai/cobalt\n```\n\n## Quick Start\n\n```bash\nnpx cobalt init\nnpx cobalt run\n```\n\nWrite your own experiment:\n\n```typescript\nimport { experiment, Evaluator, Dataset } from '@basalt-ai/cobalt'\n\nconst dataset = new Dataset({\n  items: [\n    { input: 'What is 2+2?', expectedOutput: '4' },\n    { input: 'Capital of France?', expectedOutput: 'Paris' },\n  ],\n})\n\nexperiment('qa-agent', dataset, async ({ item }) => {\n  const result = await myAgent(item.input)\n  return { output: result }\n}, {\n  evaluators: [\n    new Evaluator({\n      name: 'Correctness',\n      type: 'llm-judge',\n      prompt: 'Is the output correct?\\nExpected: {{expectedOutput}}\\nActual: {{output}}',\n    }),\n  ],\n})\n```\n\n## Features\n\n- **Experiment Runner** — Run AI agents on datasets with parallel execution\n- **Evaluators** — LLM judges (boolean/scale), custom functions, similarity, and 11 Autoevals types\n- **Datasets** — JSON, JSONL, CSV, or load from Langfuse, LangSmith, Braintrust, Basalt\n- **Cost Tracking** — Automatic token counting and cost estimation with caching\n- **CI/CD** — Quality thresholds with exit codes for pipelines\n- **MCP Server** — Model Context Protocol integration for AI coding assistants\n- **Statistics** — avg, min, max, p50, p95, p99 for all evaluators\n- **CLI** — `run`, `init`, `history`, `compare`, `serve`, `clean`, `mcp`\n\n## Evaluators\n\n### LLM Judge\n\n```typescript\nnew Evaluator({\n  name: 'relevance',\n  type: 'llm-judge',\n  prompt: 'Is the output relevant?\\nInput: {{input}}\\nOutput: {{output}}',\n  scoring: 'boolean', // or 'scale' for 0-1\n})\n```\n\n### Custom Function\n\n```typescript\nnew Evaluator({\n  name: 'length-check',\n  type: 'function',\n  fn: ({ output }) => ({\n    score: String(output).split(' ').length <= 100 ? 1 : 0,\n    reason: `${String(output).split(' ').length} words`,\n  }),\n})\n```\n\n### Similarity\n\n```typescript\nnew Evaluator({\n  name: 'semantic-match',\n  type: 'similarity',\n  field: 'expectedOutput',\n  distance: 'cosine',\n})\n```\n\n### Autoevals\n\n```typescript\nnew Evaluator({\n  name: 'factuality',\n  type: 'autoevals',\n  evaluatorType: 'Factuality',\n})\n```\n\nAvailable: `Levenshtein`, `Factuality`, `ContextRecall`, `ContextPrecision`, `AnswerRelevancy`, `Json`, `Battle`, `Humor`, `Embedding`, `ClosedQA`, `Security`.\n\n## Datasets\n\n```typescript\n// Inline\nnew Dataset({ items: [{ input: '...', expectedOutput: '...' }] })\n\n// Files\nDataset.fromJSON('./data.json')\nDataset.fromJSONL('./data.jsonl')\nDataset.fromCSV('./data.csv')\n\n// Remote platforms\nawait Dataset.fromLangfuse('dataset-name')\nawait Dataset.fromLangsmith('dataset-name')\nawait Dataset.fromBraintrust('project', 'dataset')\nawait Dataset.fromBasalt('dataset-id')\n\n// Transformations (immutable, chainable)\ndataset.filter(item => item.category === 'qa').sample(50).slice(0, 10)\n```\n\n## Configuration\n\n```typescript\n// cobalt.config.ts\nimport { defineConfig } from '@basalt-ai/cobalt'\n\nexport default defineConfig({\n  testDir: './experiments',\n  judge: { model: 'gpt-5-mini', provider: 'openai' },\n  concurrency: 5,\n  timeout: 30_000,\n  reporters: ['cli'],\n  cache: { enabled: true, ttl: '7d' },\n})\n```\n\n## Environment Variables\n\n```bash\nOPENAI_API_KEY=sk-...          # For OpenAI LLM judges\nANTHROPIC_API_KEY=sk-ant-...   # For Anthropic LLM judges\n```\n\n## Documentation\n\nFull documentation: [github.com/basalt-ai/cobalt](https://github.com/basalt-ai/cobalt)\n\n- [Datasets](https://github.com/basalt-ai/cobalt/blob/main/docs/datasets.md)\n- [Evaluators](https://github.com/basalt-ai/cobalt/blob/main/docs/evaluators.md)\n- [Experiments](https://github.com/basalt-ai/cobalt/blob/main/docs/experiments.md)\n- [Configuration](https://github.com/basalt-ai/cobalt/blob/main/docs/configuration.md)\n- [MCP Server](https://github.com/basalt-ai/cobalt/blob/main/docs/mcp.md)\n- [CI/CD](https://github.com/basalt-ai/cobalt/blob/main/docs/ci-mode.md)\n\n## Requirements\n\n- Node.js 18+\n- TypeScript 5.0+ (if using TypeScript)\n\n## License\n\nMIT — see [LICENSE](../../LICENSE) for details.\n\n## Links\n\n- [GitHub](https://github.com/basalt-ai/cobalt)\n- [Issues](https://github.com/basalt-ai/cobalt/issues)\n- [Discord](https://discord.gg/yW2RyZKY)\n- [Contributing Guide](../../CONTRIBUTING.md)\n","readmeFilename":"README.md"}