{"_id":"@artium-ai/cat-experiments","_rev":"2-d6035223104bed932ae560ce9f58d40b","name":"@artium-ai/cat-experiments","dist-tags":{"latest":"0.0.5"},"versions":{"0.0.1":{"name":"@artium-ai/cat-experiments","version":"0.0.1","keywords":["llm","experiments","evaluation","testing","ai","machine-learning"],"author":{"name":"Artium"},"license":"MIT","_id":"@artium-ai/cat-experiments@0.0.1","maintainers":[{"name":"kurtis_artium","email":"kurtisseebaldt@artium.ai"}],"homepage":"https://github.com/thisisartium/cat-experiments#readme","bugs":{"url":"https://github.com/thisisartium/cat-experiments/issues"},"bin":{"cat-experiments":"dist/bin/cli.js","cat-experiments-executor":"dist/bin/executor.js"},"dist":{"shasum":"6e81c9f5b335b169a1487d270a1a35dd0c409eac","tarball":"https://registry.npmjs.org/@artium-ai/cat-experiments/-/cat-experiments-0.0.1.tgz","fileCount":46,"integrity":"sha512-G44rQiIGrLOyJP+mSQjlZuP0SIUHC3r1zwnzodoJypvY+WXQr2y13knocwukFCz7+TdQl5uHO/wn7/MeeqqY0Q==","signatures":[{"sig":"MEUCIHj8G+PAiw3LD249mx5jP69+LWKcgL9E+O58xE97xo5jAiEA0NFcyEG1gX//ZipxwHUAEdH1PK3KxGLbo7UNtA0w3bM=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":121980},"main":"./dist/src/index.js","type":"module","types":"./dist/src/index.d.ts","engines":{"node":">=18"},"exports":{".":{"types":"./dist/src/index.d.ts","import":"./dist/src/index.js"},"./tracing":{"types":"./dist/src/tracing/index.d.ts","import":"./dist/src/tracing/index.js"}},"gitHead":"3c794bc436a1a0ddb30fd72db93102db583cc1a3","scripts":{"lint":"eslint src","test":"vitest run","build":"tsc","clean":"rm -rf dist","typecheck":"tsc --noEmit","test:watch":"vitest","prepublishOnly":"npm run build"},"_npmUser":{"name":"kurtis_artium","email":"kurtisseebaldt@artium.ai"},"repository":{"url":"git+https://github.com/thisisartium/cat-experiments.git","type":"git","directory":"node/sdk"},"_npmVersion":"11.6.2","description":"TypeScript/JavaScript SDK for running LLM experiments with cat-experiments CLI","directories":{},"_nodeVersion":"25.2.1","dependencies":{"tsx":"^4.0.0"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.0.0","@types/node":"^20.0.0","@opentelemetry/api":"^1.9.0","@opentelemetry/sdk-trace-base":"^1.30.0","@opentelemetry/sdk-trace-node":"^1.30.0"},"peerDependencies":{"@opentelemetry/api":"^1.9.0","@opentelemetry/sdk-trace-base":"^1.30.0","@opentelemetry/sdk-trace-node":"^1.30.0"},"optionalDependencies":{"@artium-ai/cat-experiments-linux-x64":"0.0.1","@artium-ai/cat-experiments-win32-x64":"0.0.1","@artium-ai/cat-experiments-darwin-x64":"0.0.1","@artium-ai/cat-experiments-linux-arm64":"0.0.1","@artium-ai/cat-experiments-darwin-arm64":"0.0.1"},"peerDependenciesMeta":{"@opentelemetry/api":{"optional":true},"@opentelemetry/sdk-trace-base":{"optional":true},"@opentelemetry/sdk-trace-node":{"optional":true}},"_npmOperationalInternal":{"tmp":"tmp/cat-experiments_0.0.1_1768000063236_0.3931311618058162","host":"s3://npm-registry-packages-npm-production"}},"0.0.5":{"name":"@artium-ai/cat-experiments","version":"0.0.5","description":"TypeScript/JavaScript SDK for running LLM experiments with cat-experiments CLI","type":"module","exports":{".":{"import":"./dist/src/index.js","types":"./dist/src/index.d.ts"},"./tracing":{"import":"./dist/src/tracing/index.js","types":"./dist/src/tracing/index.d.ts"}},"main":"./dist/src/index.js","types":"./dist/src/index.d.ts","bin":{"cat-experiments":"dist/bin/cli.js","cat-experiments-executor":"dist/bin/executor.js"},"scripts":{"build":"tsc","test":"vitest run","test:watch":"vitest","lint":"eslint src","typecheck":"tsc --noEmit","clean":"rm -rf dist","prepublishOnly":"npm run build"},"dependencies":{"tsx":"^4.0.0"},"devDependencies":{"@opentelemetry/api":"^1.9.0","@opentelemetry/sdk-trace-base":"^1.30.0","@opentelemetry/sdk-trace-node":"^1.30.0","@types/node":"^20.0.0","typescript":"^5.0.0","vitest":"^2.0.0"},"peerDependencies":{"@opentelemetry/api":"^1.9.0","@opentelemetry/sdk-trace-base":"^1.30.0","@opentelemetry/sdk-trace-node":"^1.30.0"},"peerDependenciesMeta":{"@opentelemetry/api":{"optional":true},"@opentelemetry/sdk-trace-base":{"optional":true},"@opentelemetry/sdk-trace-node":{"optional":true}},"optionalDependencies":{"@artium-ai/cat-experiments-darwin-arm64":"0.0.5","@artium-ai/cat-experiments-darwin-x64":"0.0.5","@artium-ai/cat-experiments-linux-arm64":"0.0.5","@artium-ai/cat-experiments-linux-x64":"0.0.5","@artium-ai/cat-experiments-win32-x64":"0.0.5"},"engines":{"node":">=18"},"repository":{"type":"git","url":"git+https://github.com/thisisartium/cat-experiments.git","directory":"node/sdk"},"keywords":["llm","experiments","evaluation","testing","ai","machine-learning"],"author":{"name":"Artium"},"license":"MIT","_id":"@artium-ai/cat-experiments@0.0.5","bugs":{"url":"https://github.com/thisisartium/cat-experiments/issues"},"homepage":"https://github.com/thisisartium/cat-experiments#readme","_nodeVersion":"22.22.0","_npmVersion":"11.8.0","dist":{"integrity":"sha512-LeCFjqEccipBKevMNViVyGBi6d7Luy58714Qf1zhCk6z/02okD3JoZNJrQVKjUgVfzZ2i7YW7B8gd3lMvFhFPQ==","shasum":"61a001f7d00ea4127e34df348781ea99316001b0","tarball":"https://registry.npmjs.org/@artium-ai/cat-experiments/-/cat-experiments-0.0.5.tgz","fileCount":46,"unpackedSize":141926,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIEdCCR5ayUZ7zOg45an5HhCQeiLnygszDhX4KjsFX6PhAiBc10AC7GICDYnClu1dY/vvYScb2hlwfdV1txDIUxkhNQ=="}]},"_npmUser":{"name":"GitHub Actions","email":"npm-oidc-no-reply@github.com","trustedPublisher":{"id":"github","oidcConfigId":"oidc:f10768cd-3321-44b4-b9bf-d1905e0c509a"}},"directories":{},"maintainers":[{"name":"kurtis_artium","email":"kurtisseebaldt@artium.ai"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/cat-experiments_0.0.5_1769722800481_0.541062561572291"},"_hasShrinkwrap":false}},"time":{"created":"2026-01-09T23:07:43.171Z","modified":"2026-01-29T21:40:00.739Z","0.0.1":"2026-01-09T23:07:43.395Z","0.0.5":"2026-01-29T21:40:00.627Z"},"bugs":{"url":"https://github.com/thisisartium/cat-experiments/issues"},"author":{"name":"Artium"},"license":"MIT","homepage":"https://github.com/thisisartium/cat-experiments#readme","keywords":["llm","experiments","evaluation","testing","ai","machine-learning"],"repository":{"type":"git","url":"git+https://github.com/thisisartium/cat-experiments.git","directory":"node/sdk"},"description":"TypeScript/JavaScript SDK for running LLM experiments with cat-experiments CLI","maintainers":[{"name":"kurtis_artium","email":"kurtisseebaldt@artium.ai"}],"readme":"# cat-experiments\n\nTypeScript/JavaScript SDK for running LLM experiments with evaluation.\n\n## Installation\n\n```bash\nnpm install cat-experiments\n```\n\nThis installs both the SDK and the CLI binary for your platform.\n\n## Quick Start\n\nCreate an experiment file (`experiment.ts`):\n\n```typescript\nimport { defineExperiment, type EvalInput } from \"cat-experiments\";\n\ninterface Input {\n  question: string;\n}\n\ninterface Output {\n  answer: string;\n}\n\nexport default defineExperiment<Input, Output>({\n  name: \"my-experiment\",\n  description: \"Example experiment\",\n\n  task: async (input) => {\n    // Call your LLM or system under test here\n    const answer = await myLLM(input.input.question);\n    return { output: { answer } };\n  },\n\n  evaluators: {\n    exact_match: (input: EvalInput<Input, Output>) => {\n      const expected = input.expected_output?.answer ?? \"\";\n      const actual = input.actual_output?.answer ?? \"\";\n      return {\n        score: expected === actual ? 1.0 : 0.0,\n        label: expected === actual ? \"match\" : \"mismatch\",\n      };\n    },\n  },\n});\n```\n\nCreate a dataset file (`data.jsonl`):\n\n```jsonl\n{\"id\": \"1\", \"input\": {\"question\": \"What is 2+2?\"}, \"output\": {\"answer\": \"4\"}}\n{\"id\": \"2\", \"input\": {\"question\": \"What is the capital of France?\"}, \"output\": {\"answer\": \"Paris\"}}\n```\n\nRun the experiment:\n\n```bash\nnpx cat-experiments run experiment.ts --dataset data.jsonl\n```\n\n## API Reference\n\n### `defineExperiment<TInput, TOutput>(config)`\n\nCreates a type-safe experiment definition.\n\n```typescript\nimport { defineExperiment } from \"cat-experiments\";\n\nexport default defineExperiment<Input, Output>({\n  name: \"experiment-name\",\n  description: \"Optional description\",\n\n  // The system under test\n  task: async (input) => {\n    return {\n      output: { /* your output */ },\n      metadata: { /* optional metadata */ },\n    };\n  },\n\n  // Evaluation functions\n  evaluators: {\n    evaluator_name: (input) => {\n      return {\n        score: 0.0 - 1.0,\n        label: \"optional label\",\n        metadata: { /* optional */ },\n      };\n    },\n  },\n\n  // Optional default parameters\n  params: {\n    model: \"gpt-4\",\n    temperature: 0.7,\n  },\n});\n```\n\n#### Task Input\n\nThe task function receives a `TaskInput<TInput>` object:\n\n```typescript\ninterface TaskInput<TInput> {\n  id: string;              // Example ID\n  run_id: string;          // Unique run ID (includes repetition)\n  input: TInput;           // Your typed input\n  expected_output?: any;   // Expected output from dataset\n  metadata?: any;          // Example metadata\n  params: Record<string, unknown>;  // Runtime parameters\n}\n```\n\n#### Task Output\n\nReturn a `TaskOutput<TOutput>` object:\n\n```typescript\ninterface TaskOutput<TOutput> {\n  output: TOutput;         // Your typed output\n  metadata?: any;          // Optional metadata (tokens, latency, etc.)\n  error?: string;          // Error message if task failed\n}\n```\n\n#### Evaluator Input\n\nEvaluators receive an `EvalInput<TInput, TOutput>` object:\n\n```typescript\ninterface EvalInput<TInput, TOutput> {\n  example: {\n    id: string;\n    run_id: string;\n    input: TInput;\n    output?: TOutput;      // Expected output\n    metadata?: any;\n  };\n  actual_output?: TOutput; // Output from task\n  expected_output?: TOutput;\n  task_metadata?: any;     // Metadata from task\n  params: Record<string, unknown>;\n}\n```\n\n#### Evaluator Output\n\nReturn an `EvalOutput` object:\n\n```typescript\ninterface EvalOutput {\n  score: number;           // 0.0 to 1.0\n  label?: string;          // Optional categorical label\n  metadata?: any;          // Optional metadata\n  error?: string;          // Error message if evaluation failed\n}\n```\n\n### `matchToolCalls(expected, actual, options?)`\n\nEvaluates tool/function calls against expected calls. Useful for testing agents that use tools.\n\n```typescript\nimport { matchToolCalls } from \"cat-experiments\";\n\nconst result = matchToolCalls(\n  // Expected tool calls\n  [{ name: \"search\", arguments: { query: \"weather\" } }],\n  // Actual tool calls\n  [{ name: \"search\", arguments: { query: \"weather today\" } }],\n  // Options\n  { mode: \"fuzzy\" }\n);\n\nconsole.log(result.score);      // 0.8\nconsole.log(result.precision);  // 1.0\nconsole.log(result.recall);     // 1.0\n```\n\n#### Options\n\n| Option | Description | Default |\n|--------|-------------|---------|\n| `mode` | Matching strictness: `\"exact\"`, `\"strict\"`, or `\"fuzzy\"` | `\"fuzzy\"` |\n| `ordered` | Whether call order matters | `false` |\n\n#### Modes\n\n- **`exact`** - Names and arguments must match exactly\n- **`strict`** - Names must match, actual arguments must contain all expected arguments\n- **`fuzzy`** - Partial matching with similarity scoring\n\n## CLI Reference\n\n```bash\nnpx cat-experiments run <experiment.ts> [options]\n```\n\n### Options\n\n| Option | Description | Default |\n|--------|-------------|---------|\n| `-d, --dataset` | Dataset file (JSONL) or remote name | required |\n| `-b, --backend` | Storage backend: `local`, `phoenix`, `cat-cafe` | `local` |\n| `--backend-url` | URL for remote backend | - |\n| `--max-workers` | Parallel workers | `5` |\n| `--repetitions` | Repetitions per example | `1` |\n| `--dry-run N` | Run N examples without persisting | `0` |\n| `--param KEY=VALUE` | Override parameters (repeatable) | - |\n| `--resume ID` | Resume a previous experiment | - |\n| `--no-progress` | Disable progress bar | - |\n| `--output` | Output format: `text`, `json` | `text` |\n\n### Examples\n\n```bash\n# Basic run with local storage\nnpx cat-experiments run experiment.ts --dataset data.jsonl\n\n# Run with 10 parallel workers\nnpx cat-experiments run experiment.ts --dataset data.jsonl --max-workers 10\n\n# Run 3 repetitions per example\nnpx cat-experiments run experiment.ts --dataset data.jsonl --repetitions 3\n\n# Override parameters\nnpx cat-experiments run experiment.ts --dataset data.jsonl \\\n  --param model=gpt-4 \\\n  --param temperature=0.5\n\n# Dry run first 5 examples (no storage)\nnpx cat-experiments run experiment.ts --dataset data.jsonl --dry-run 5\n\n# Resume a failed experiment\nnpx cat-experiments run experiment.ts --dataset data.jsonl \\\n  --resume my-experiment_20240101_120000\n```\n\n## Dataset Format\n\nDatasets are JSONL files with one example per line:\n\n```jsonl\n{\"id\": \"1\", \"input\": {\"question\": \"...\"}, \"output\": {\"answer\": \"...\"}}\n{\"id\": \"2\", \"input\": {\"question\": \"...\"}, \"output\": {\"answer\": \"...\"}, \"metadata\": {\"difficulty\": \"hard\"}}\n```\n\n| Field | Required | Description |\n|-------|----------|-------------|\n| `id` | Yes | Unique identifier for the example |\n| `input` | Yes | Input data matching your `TInput` type |\n| `output` | No | Expected output matching your `TOutput` type |\n| `metadata` | No | Additional metadata accessible in evaluators |\n\n## Tracing: Automatic Tool Call Capture\n\nIf your LLM library is instrumented with OpenTelemetry (e.g., via OpenInference for Phoenix/Arize), you can automatically capture tool calls during task execution.\n\n### Installation\n\nInstall the OpenTelemetry peer dependencies:\n\n```bash\nnpm install @opentelemetry/api @opentelemetry/sdk-trace-node\n```\n\n### Usage\n\n```typescript\nimport { defineExperiment } from \"cat-experiments\";\nimport { captureToolCalls, setupTracing } from \"cat-experiments/tracing\";\n\n// Call setupTracing() BEFORE initializing your LLM instrumentation\nsetupTracing();\n\n// Then set up your instrumentation (e.g., OpenInference)\n// import { OpenInferenceInstrumentation } from \"@arizeai/openinference-instrumentation\";\n// new OpenInferenceInstrumentation().instrument();\n\nexport default defineExperiment({\n  name: \"agent-with-tracing\",\n\n  task: async (input) => {\n    // captureToolCalls wraps your LLM call and extracts tool calls from spans\n    const captured = await captureToolCalls(async () => {\n      return await myAgent.run(input.input.query);\n    });\n\n    return {\n      output: {\n        response: captured.result.text,\n        tool_calls: captured.toolCalls,\n      },\n    };\n  },\n\n  evaluators: {\n    // ... your evaluators\n  },\n});\n```\n\n### Supported Instrumentations\n\n- **OpenInference** (Phoenix, Arize) - Full support\n- **OpenLLMetry** (Traceloop) - Partial support\n- **Generic tool spans** - Fallback for other instrumentations\n\n### Custom Extractors\n\nYou can provide custom extractors for other instrumentation formats:\n\n```typescript\nimport { captureToolCalls, type ToolCallExtractor } from \"cat-experiments/tracing\";\n\nconst myExtractor: ToolCallExtractor = {\n  canHandle(span, attributes) {\n    return \"my.custom.attribute\" in attributes;\n  },\n  extract(span, attributes) {\n    return [{\n      name: attributes[\"my.tool.name\"] as string,\n      args: JSON.parse(attributes[\"my.tool.args\"] as string),\n    }];\n  },\n};\n\nconst captured = await captureToolCalls(\n  async () => myAgent.run(query),\n  { extractors: [myExtractor] }\n);\n```\n\n## Example: Tool Call Evaluation\n\n```typescript\nimport { defineExperiment, matchToolCalls, type EvalInput } from \"cat-experiments\";\n\ninterface Input {\n  query: string;\n}\n\ninterface Output {\n  response: string;\n  tool_calls?: { name: string; arguments: Record<string, unknown> }[];\n}\n\nexport default defineExperiment<Input, Output>({\n  name: \"agent-tools\",\n\n  task: async (input) => {\n    const result = await myAgent.run(input.input.query);\n    return {\n      output: {\n        response: result.text,\n        tool_calls: result.toolCalls,\n      },\n    };\n  },\n\n  evaluators: {\n    tool_accuracy: (input: EvalInput<Input, Output>) => {\n      const expected = input.expected_output?.tool_calls ?? [];\n      const actual = input.actual_output?.tool_calls ?? [];\n\n      const result = matchToolCalls(expected, actual, { mode: \"strict\" });\n\n      return {\n        score: result.score,\n        label: result.score === 1.0 ? \"pass\" : \"fail\",\n        metadata: {\n          precision: result.precision,\n          recall: result.recall,\n          matches: result.matches,\n        },\n      };\n    },\n  },\n});\n```\n\n## License\n\nMIT\n","readmeFilename":"README.md"}