{"_id":"@aui.io/evals","_rev":"13-e12ea955b6739a7bf2ad4189152966c2","name":"@aui.io/evals","dist-tags":{"latest":"0.1.1"},"versions":{"0.1.0":{"name":"@aui.io/evals","version":"0.1.0","keywords":["aui","agent","evaluation","testing","llm","conversational-ai","agent-testing","llm-judge"],"author":{"name":"AUI.io Team"},"license":"ISC","_id":"@aui.io/evals@0.1.0","maintainers":[{"name":"ahmed_aui","email":"ahmeds@aui.io"},{"name":"tareqassi","email":"tareqa@aui.io"},{"name":"ilonas2","email":"ilonas+npm@aui.io"},{"name":"tareqsabra","email":"tareqs@aui.io"},{"name":"hibam","email":"hibam@aui.io"},{"name":"ahmad-romi","email":"ahmadr@aui.io"},{"name":"amjadsalhab","email":"amjads@aui.io"},{"name":"rolak","email":"rolak@aui.io"},{"name":"doralboim","email":"alboim@aui.io"},{"name":"dimab-aui","email":"dimab@aui.io"},{"name":"yuvalkg","email":"yuvalk@aui.io"},{"name":"noamaui","email":"noamy@aui.io"},{"name":"joni-aui","email":"yonatany@aui.io"},{"name":"nirmendel","email":"nirm@aui.io"},{"name":"michaltev","email":"michalt@aui.io"},{"name":"alaakmashaqi","email":"alaam@aui.io"},{"name":"aviram_aui","email":"aviramr@aui.io"},{"name":"yoavaui","email":"yoavb@aui.io"},{"name":"ahmaddaui","email":"ahmadd@aui.io"},{"name":"ahmadarafat","email":"ahmadar@aui.io"},{"name":"alaadwaikat-aui","email":"alaa@gigassistants.com"},{"name":"aseel-hussain-alali","email":"aseelh@aui.io"},{"name":"ahmadh","email":"ahmadh@aui.io"},{"name":"samerdmaidi","email":"samerd@aui.io"}],"homepage":"https://github.com/aui-io/evals#readme","bugs":{"url":"https://github.com/aui-io/evals/issues"},"bin":{"aui-evals":"dist/cli.js"},"dist":{"shasum":"88fc4889e7c8b1410bb2a69f4148e5b56ddf5c40","tarball":"https://registry.npmjs.org/@aui.io/evals/-/evals-0.1.0.tgz","fileCount":85,"integrity":"sha512-RSKWzMa/88x1doQ7NA+1ZYnWIiKEXvfxVie55Se5YcyzTJPZq3/CN0KCozyiIEa5NDv2B2qfKWBa5xk/C9pJXw==","signatures":[{"sig":"MEYCIQDk8lroAdMyNU5+7upG2nYd90HztfwUT4NWeyQ7OeILxgIhAPmM94C3gGpxsWSMyRX6zq+gdotCys3rD1+mv0pjSgio","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":4799904},"main":"dist/index.js","type":"module","types":"dist/index.d.ts","gitHead":"a14e0c45b2e9b54e312b84873fbe35d5636c2a15","scripts":{"test":"echo \"Error: no test specified\" && exit 1","build":"tsc","clean":"rm -rf dist","prepublishOnly":"npm run build"},"_npmUser":{"name":"amjadsalhab","email":"amjads@aui.io"},"repository":{"url":"git+https://github.com/aui-io/evals.git","type":"git"},"_npmVersion":"10.9.7","description":"Multi-turn conversation agent evaluator with LLM-as-judge scoring.","directories":{},"_nodeVersion":"22.22.2","dependencies":{"@anthropic-ai/sdk":"^0.86.1"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.21.0","typescript":"^6.0.2","@types/node":"^25.5.0"},"_npmOperationalInternal":{"tmp":"tmp/evals_0.1.0_1776329419846_0.8719101863381538","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@aui.io/evals","version":"0.1.1","keywords":["aui","agent","evaluation","testing","llm","conversational-ai","agent-testing","llm-judge"],"author":{"name":"AUI.io Team"},"license":"ISC","_id":"@aui.io/evals@0.1.1","maintainers":[{"name":"ahmed_aui","email":"ahmeds@aui.io"},{"name":"tareqassi","email":"tareqa@aui.io"},{"name":"ilonas2","email":"ilonas+npm@aui.io"},{"name":"tareqsabra","email":"tareqs@aui.io"},{"name":"hibam","email":"hibam@aui.io"},{"name":"ahmad-romi","email":"ahmadr@aui.io"},{"name":"amjadsalhab","email":"amjads@aui.io"},{"name":"rolak","email":"rolak@aui.io"},{"name":"doralboim","email":"alboim@aui.io"},{"name":"dimab-aui","email":"dimab@aui.io"},{"name":"yuvalkg","email":"yuvalk@aui.io"},{"name":"noamaui","email":"noamy@aui.io"},{"name":"joni-aui","email":"yonatany@aui.io"},{"name":"nirmendel","email":"nirm@aui.io"},{"name":"michaltev","email":"michalt@aui.io"},{"name":"alaakmashaqi","email":"alaam@aui.io"},{"name":"aviram_aui","email":"aviramr@aui.io"},{"name":"yoavaui","email":"yoavb@aui.io"},{"name":"ahmaddaui","email":"ahmadd@aui.io"},{"name":"ahmadarafat","email":"ahmadar@aui.io"},{"name":"alaadwaikat-aui","email":"alaa@gigassistants.com"},{"name":"aseel-hussain-alali","email":"aseelh@aui.io"},{"name":"ahmadh","email":"ahmadh@aui.io"},{"name":"samerdmaidi","email":"samerd@aui.io"}],"homepage":"https://github.com/aui-io/evals#readme","bugs":{"url":"https://github.com/aui-io/evals/issues"},"bin":{"aui-evals":"dist/cli.js"},"dist":{"shasum":"97c19df3c27426bcafb942dbf0b7e670c7e6221d","tarball":"https://registry.npmjs.org/@aui.io/evals/-/evals-0.1.1.tgz","fileCount":85,"integrity":"sha512-xfFyzjpP+U0LFA5CmZRmP1P3qiDfPmsZ/NKigLaacfTvtrDXO3X9pW4aCyB5iWRnz9WNo7jXphmBsYZwSN+CZA==","signatures":[{"sig":"MEYCIQDeBgNNLYlblDYLPfnbesSuROhJdCo01d8nSGFlzzWLLgIhAPPZS21ca5Bg9NhQQWSM+6mu6hYBc+vOtmHmdDnw+rAT","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":4810208},"main":"dist/index.js","type":"module","types":"dist/index.d.ts","gitHead":"bf24de653efac488b1510112851fa2b2f2ccde8c","scripts":{"test":"echo \"Error: no test specified\" && exit 1","build":"tsc","clean":"rm -rf dist","prepublishOnly":"npm run build"},"_npmUser":{"name":"amjadsalhab","email":"amjads@aui.io"},"repository":{"url":"git+https://github.com/aui-io/evals.git","type":"git"},"_npmVersion":"10.9.7","description":"Multi-turn conversation agent evaluator with LLM-as-judge scoring.","directories":{},"_nodeVersion":"22.22.2","dependencies":{"@anthropic-ai/sdk":"^0.86.1"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.21.0","typescript":"^6.0.2","@types/node":"^25.5.0"},"_npmOperationalInternal":{"tmp":"tmp/evals_0.1.1_1776330227410_0.16120208190799845","host":"s3://npm-registry-packages-npm-production"}}},"time":{"created":"2026-04-16T08:50:19.772Z","modified":"2026-09-06T12:09:04.157Z","0.1.0":"2026-04-16T08:50:20.057Z","0.1.1":"2026-04-16T09:03:47.704Z"},"bugs":{"url":"https://github.com/aui-io/evals/issues"},"author":{"name":"AUI.io Team"},"license":"ISC","homepage":"https://github.com/aui-io/evals#readme","keywords":["aui","agent","evaluation","testing","llm","conversational-ai","agent-testing","llm-judge"],"repository":{"url":"git+https://github.com/aui-io/evals.git","type":"git"},"description":"Multi-turn conversation agent evaluator with LLM-as-judge scoring.","maintainers":[{"email":"ahmeds@aui.io","name":"ahmed_aui"},{"email":"tareqa@aui.io","name":"tareqassi"},{"email":"ilonas+npm@aui.io","name":"ilonas2"},{"email":"hibam@aui.io","name":"hibam"},{"email":"ahmadr@aui.io","name":"ahmad-romi"},{"email":"amjads@aui.io","name":"amjadsalhab"},{"email":"rolak@aui.io","name":"rolak"},{"email":"noamy@aui.io","name":"noamaui"},{"email":"alaam@aui.io","name":"alaakmashaqi"},{"email":"ahmadd@aui.io","name":"ahmaddaui"},{"email":"alaa@gigassistants.com","name":"alaadwaikat-aui"},{"email":"ramzit@aui.io","name":"ramziaui"},{"email":"samerd@aui.io","name":"samerdmaidi"}],"readme":"# AUI Agent Evaluator\n\nMulti-turn conversation evaluation framework for AUI agents. Runs simulated conversations against any AUI agent, then scores them with both LLM-based and programmatic (strict) evaluations.\n\n## Install\n\n```bash\nnpm install @aui.io/evals\n```\n\n## Quick Start\n\n```bash\n# Validate a test suite\nnpx aui-evals validate my-suite.json\n\n# Preview what will run\nnpx aui-evals plan my-suite.json\n\n# Run with OpenAI judge\nnpx aui-evals run my-suite.json --llm-key=<YOUR_OPENAI_KEY>\n\n# Run with Anthropic judge\nnpx aui-evals run my-suite.json --llm-key=<YOUR_ANTHROPIC_KEY> --provider=anthropic\n```\n\nYou can also set `LLM_API_KEY` as an env var instead of `--llm-key`.\n\n## CLI Commands\n\n### List Bundled Test Suites\n\n```bash\n# See all available bundled test suites\nnpx aui-evals list-suites\n```\n\nThis shows all included Demo Agents and Quack Agents test suites with their names.\n\n### Run Bundled Test Suites\n\nYou can run bundled suites by name without specifying full paths:\n\n```bash\n# Run Demo Agents suites\nnpx aui-evals run demo:hr --llm-key=$OPENAI_KEY\nnpx aui-evals run demo:retail --llm-key=$OPENAI_KEY\nnpx aui-evals run demo:automotive-agent-140426 --llm-key=$KEY\n\n# Run Quack Agents suites\nnpx aui-evals run quack:artlist/payment_issues_tda --llm-key=$KEY\nnpx aui-evals run quack:notable/compass_tests --llm-key=$KEY\n```\n\n### Run Local Test Suites\n\n```bash\n# Validate a test suite\nnpx aui-evals validate suite.json\n\n# Show execution plan\nnpx aui-evals plan suite.json\n\n# Run evaluations\nnpx aui-evals run suite.json --llm-key=$OPENAI_KEY\n\n# Run with specific options\nnpx aui-evals run suite.json \\\n  --llm-key=$OPENAI_KEY \\\n  --provider=openai \\\n  --parallel=10 \\\n  --loops=3 \\\n  --test=\"refund,cancel\" \\\n  --tag=\"critical\"\n```\n\n### CLI Flags\n\n| Flag | Description |\n|------|-------------|\n| `--llm-key=<KEY>` | API key for the judge/user-sim LLM (or set `LLM_API_KEY` env var) |\n| `--provider=openai\\|anthropic` | LLM provider for judging (default: `openai`) |\n| `--parallel=<N>` | Number of tests to run in parallel (default: 1) |\n| `--loops=<N>` | Number of times to repeat each test (default: 1) |\n| `--test=<name>` | Filter tests by name (substring match, comma-separated) |\n| `--tag=<tag>` | Filter tests by tag (comma-separated) |\n| `--debug` | Include raw API responses in markdown output |\n\n---\n\n## Test Suite Structure\n\nA test suite is a JSON file with the following top-level fields:\n\n| Field | Type | Required | Description |\n|-------|------|----------|-------------|\n| `name` | string | yes | Name of the evaluation suite |\n| `agentId` | string | yes | Agent ID to test |\n| `apiKey` | string | yes | Network API key for authenticating with the agent |\n| `model` | string | no | Model override for the agent under test |\n| `judgeModel` | string | no | Model for the judge LLM (defaults to suite runner's model) |\n| `tests` | Test[] | yes | Array of test cases (min 1) |\n\n## Test\n\nEach test defines a simulated user scenario and the criteria to evaluate.\n\n| Field | Type | Required | Description |\n|-------|------|----------|-------------|\n| `name` | string | yes | Unique test name |\n| `userInfo` | object | no | Free-form key-value pairs passed as `agent_variables` to the agent API (user profile, subscription, billing, etc.) |\n| `guidelines` | string | yes | Behavioral guidelines — how the simulated user should act (personality, goals, tone) |\n| `systemPrompt` | string | no | System prompt override for the agent |\n| `turns` | Turn[] | yes | Conversation turns to execute (min 1). Padded to 7 turns with `\"auto\"` if fewer are defined |\n| `evaluations` | Evaluation[] | yes | Criteria to evaluate after the conversation (min 1) |\n| `tags` | string[] | no | Tags for filtering/grouping test results |\n\n## Turn\n\nEach turn represents one user message in the conversation.\n\n| Field | Type | Required | Description |\n|-------|------|----------|-------------|\n| `user` | string | yes | The user message to send. Use `\"auto\"` to let the LLM generate a realistic reply based on `userInfo`, `guidelines`, and conversation history |\n| `timeout` | number | no | Timeout in seconds for the agent response (default: 120) |\n| `expectation` | string | no | Expected behavior hint — not sent to the agent, used by the LLM judge for context when evaluating |\n\n### Auto-generated turns\n\nWhen `user` is `\"auto\"`, the framework uses the judge LLM to generate a realistic user message based on the persona and conversation history. The simulated user will output `###STOP###` when the conversation reaches a terminal state, ending the test early.\n\nIf fewer than 7 turns are defined, the remaining turns are automatically filled with `\"auto\"` to allow the conversation to play out naturally.\n\n---\n\n## Evaluations\n\nEvaluations are criteria applied after the conversation completes. There are two types:\n\n### 1. LLM Judge Evaluation (default)\n\nWhen no `strict` field is set, the criterion is evaluated by an LLM judge that reads the full conversation and determines pass/fail with reasoning.\n\n```json\n{\n  \"criterion\": \"The agent maintained a friendly and professional tone throughout\",\n  \"weight\": 1,\n  \"category\": \"tone\"\n}\n```\n\n### 2. Strict Evaluation (programmatic)\n\nWhen `strict` is set, the evaluation bypasses the LLM judge and is checked programmatically against the conversation trace data. This is deterministic and faster.\n\n### Common fields (all evaluation types)\n\n| Field | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| `criterion` | string | yes | — | Natural language description of what to evaluate |\n| `weight` | number | no | 1 | Importance weight. Higher = more impact on the overall score |\n| `category` | string | no | — | Category for grouping results (e.g. `\"accuracy\"`, `\"tone\"`, `\"safety\"`, `\"compliance\"`) |\n| `aui_logic` | string[] | no | — | AUI logic rule IDs that govern this evaluation. Used in the summary report's \"By AUI Logic\" table |\n\n---\n\n## Strict Evaluation Types\n\n### `escalated`\n\nPasses if `CONVERSATION_FORWARDING` was triggered (agent escalated to a human).\n\n```json\n{\n  \"criterion\": \"Agent should escalate this case to a human\",\n  \"strict\": \"escalated\"\n}\n```\n\n### `not_escalated`\n\nPasses if `CONVERSATION_FORWARDING` was NOT triggered.\n\n```json\n{\n  \"criterion\": \"Agent should handle this without escalation\",\n  \"strict\": \"not_escalated\"\n}\n```\n\n### `tool_used`\n\nPasses if the specified tool/workflow was executed during the conversation.\n\n| Extra field | Required | Description |\n|-------------|----------|-------------|\n| `toolName` | yes | The workflow name to check (from `executed_workflows`) |\n\n```json\n{\n  \"criterion\": \"Refund request workflow should be activated\",\n  \"strict\": \"tool_used\",\n  \"toolName\": \"REFUND_REQUEST\"\n}\n```\n\n### `tool_not_used`\n\nPasses if the specified tool/workflow was NOT executed. Optionally, can check that it was not used *before* a specific turn.\n\n| Extra field | Required | Description |\n|-------------|----------|-------------|\n| `toolName` | yes | The workflow name to check |\n| `beforeTurn` | no | 1-based turn number. If the tool is used at this turn or later, it passes. Only fails if used before this turn |\n\n```json\n{\n  \"criterion\": \"Should not offer a discount before collecting the reason\",\n  \"strict\": \"tool_not_used\",\n  \"toolName\": \"DISCOUNT_OFFER\",\n  \"beforeTurn\": 3\n}\n```\n\n### `param_equals`\n\nPasses if an extracted parameter from `trace_info.understanding.extracted_params` matches the expected value.\n\n| Extra field | Required | Description |\n|-------------|----------|-------------|\n| `paramName` | yes | Parameter name to check |\n| `paramValue` | yes | Expected value |\n\n```json\n{\n  \"criterion\": \"Agent should classify the reason as too_expensive\",\n  \"strict\": \"param_equals\",\n  \"paramName\": \"refund-reason\",\n  \"paramValue\": \"too_expensive\"\n}\n```\n\n### `param_exists`\n\nPasses if the parameter was extracted (with a non-empty, non-\"undefined\" value).\n\n| Extra field | Required | Description |\n|-------------|----------|-------------|\n| `paramName` | yes | Parameter name to check |\n\n```json\n{\n  \"criterion\": \"Agent should extract the reason parameter\",\n  \"strict\": \"param_exists\",\n  \"paramName\": \"refund-reason\"\n}\n```\n\n### `param_not_exists`\n\nPasses if the parameter was NOT extracted (missing, empty, or \"undefined\").\n\n| Extra field | Required | Description |\n|-------------|----------|-------------|\n| `paramName` | yes | Parameter name to check |\n\n```json\n{\n  \"criterion\": \"Agent should not extract a discount amount\",\n  \"strict\": \"param_not_exists\",\n  \"paramName\": \"discount-amount\"\n}\n```\n\n### `escalated_due_to_error`\n\nPasses if the agent returned an error response (`trace_info.response.type === \"error\"`).\n\n```json\n{\n  \"criterion\": \"Agent should return an error for this invalid request\",\n  \"strict\": \"escalated_due_to_error\"\n}\n```\n\n### `not_escalated_due_to_error`\n\nPasses if the agent did NOT return an error response.\n\n```json\n{\n  \"criterion\": \"Agent should handle this without errors\",\n  \"strict\": \"not_escalated_due_to_error\"\n}\n```\n\n### `rule_triggered`\n\nPasses if a specific rule (identified by its `code`) was triggered in `trace_info.decisions`. Rules are accumulated across all turns.\n\nThe rule code comes from `trace_info.decisions[].rule.code` in the agent's raw response.\n\n| Extra field | Required | Description |\n|-------------|----------|-------------|\n| `ruleCode` | yes | The rule code to check for |\n\n```json\n{\n  \"criterion\": \"Eligible escalation rule should fire\",\n  \"strict\": \"rule_triggered\",\n  \"ruleCode\": \"low-tier-eligible-escalate\"\n}\n```\n\n### `rule_not_triggered`\n\nPasses if a specific rule was NOT triggered in any turn.\n\n| Extra field | Required | Description |\n|-------------|----------|-------------|\n| `ruleCode` | yes | The rule code that should not appear |\n\n```json\n{\n  \"criterion\": \"Premium retention rule should not fire for this user\",\n  \"strict\": \"rule_not_triggered\",\n  \"ruleCode\": \"premium-retention-offer\"\n}\n```\n\n### `integration_called`\n\nPasses if a `call_integration` decision with the matching `integration.code` and `integration.status_code === 200` exists in `trace_info.decisions` across any turn.\n\n| Extra field | Required | Description |\n|-------------|----------|-------------|\n| `integrationCode` | yes | The integration code to check for (from `decisions[].integration.code`) |\n\n```json\n{\n  \"criterion\": \"Employee information integration should be called successfully\",\n  \"strict\": \"integration_called\",\n  \"integrationCode\": \"employee-information\",\n  \"category\": \"accuracy\"\n}\n```\n\n---\n\n## Scoring\n\n- Each evaluation produces a pass/fail result\n- The overall test score is the weighted pass rate: `sum(passed weights) / sum(all weights) * 100`\n- Tests that end with an error response (`escalated_due_to_error`) are excluded from the aggregate suite score\n- The suite aggregate score is the average of all non-error test scores\n\n## How It Works\n\n1. **Create task** — `POST /tasks` with a random `user_id` and `agent_id` returns a `task_id`\n2. **Run turns** — For each turn, `POST /message` with `task_id`, `text`, and `agent_variables` (the full `userInfo` object)\n   - **Fixed turns:** sends the exact message defined in the test\n   - **Auto turns:** LLM generates a realistic user reply based on the user profile, guidelines, and conversation history\n3. **Escalation detection** — If the agent response includes `CONVERSATION_FORWARDING` in `executed_workflows`, the conversation stops early\n4. **Error detection** — If `trace_info.response.type === \"error\"`, the conversation stops early\n5. **Judge** — Each evaluation criterion is scored:\n   - **Strict evaluations**: checked programmatically against trace data (deterministic, no LLM call)\n   - **LLM evaluations**: judge LLM reads the full conversation and scores pass/fail with reasoning\n6. **Scoring** — Weighted pass rate per test, aggregated across the suite\n\n## Output\n\nResults are written to `results/<suite-name>/<timestamp>/`:\n\n| File | Contents |\n|------|----------|\n| `_results.json` | Full raw results with conversations, evaluations, and trace data |\n| `_summary.md` | Aggregate scores, pass rates, category breakdown, AUI logic breakdown |\n| `<test-name>.md` | Individual test report with full conversation and evaluation details |\n\n## Project Structure\n\n```\nsrc/\n  cli.ts            — CLI entry point, argument parsing, file I/O\n  types.ts          — TypeScript types (TestSuite, UserInfo, AgentResponse, results)\n  schema.ts         — JSON Schema (draft-07) for test suite validation\n  validate.ts       — Suite validator (required fields, structure checks)\n  runner.ts         — Core runner engine + AgentAdapter interface\n  adapter.ts        — AUI API adapter (tasks/message endpoints) + OpenAI/Anthropic LLM calls\n  claude-adapter.ts — Claude Code adapter for running evals inside Claude Code\n  formatter.ts      — Markdown report generator (per-test + suite summary)\n  index.ts          — Public exports for programmatic usage\n```\n\n## Programmatic Usage\n\n```typescript\nimport { createAuiAdapter, runSuite, formatSuiteResult } from '@aui.io/evals';\nimport { readFileSync } from 'fs';\n\nconst adapter = createAuiAdapter(\n  '<NETWORK_API_KEY>',\n  '<LLM_API_KEY>',\n  'openai'  // or 'anthropic'\n);\n\nconst suite = JSON.parse(readFileSync('my-suite.json', 'utf-8'));\nconst result = await runSuite(adapter, suite);\nconsole.log(formatSuiteResult(result));\n```\n\n## Example Test Suites\n\nThis package includes real-world example test suites in two directories:\n\n- **`Demo Agents/`** - Generic demo agent evaluations for various domains:\n  - Automotive (used car sales)\n  - Airbnb (customer support)\n  - Retail (e-commerce)\n  - IT Support\n  - HR Assistant\n  - Credit Card Dispute\n\n- **`Quack Agents/`** - Production test suites for specific clients:\n  - Artlist (payment issues, refunds, subscriptions)\n  - Notable (account access, compass tests)\n  - Rentman (account management)\n  - Yotpo UGC (policy suites)\n\nThese examples demonstrate best practices for structuring test suites, writing evaluations, and using both LLM-based and strict (programmatic) evaluation criteria.\n\n### Accessing Bundled Test Suites Programmatically\n\nThe package exports helper functions to access the bundled test suites:\n\n```typescript\nimport { getDemoAgentsPath, getQuackAgentsPath, getPackageRoot } from '@aui.io/evals';\nimport { join } from 'path';\nimport { readFileSync } from 'fs';\n\n// Get paths to bundled test suites\nconst demoPath = getDemoAgentsPath();\nconst quackPath = getQuackAgentsPath();\n\n// Load a specific demo suite\nconst hrSuitePath = join(demoPath, 'demo-agents', 'hr.json');\nconst hrSuite = JSON.parse(readFileSync(hrSuitePath, 'utf-8'));\n\n// Load a Quack Agents suite\nconst artlistPath = join(quackPath, 'artlist', 'payment_issues_tda.json');\nconst artlistSuite = JSON.parse(readFileSync(artlistPath, 'utf-8'));\n\n// Or get the package root and navigate from there\nconst packageRoot = getPackageRoot();\nconst customPath = join(packageRoot, 'Demo Agents', 'eval_retail-agent_140426.json');\n```\n\n**Via CLI:**\n```bash\n# Reference bundled suites directly\nnpx aui-evals run \"node_modules/@aui.io/evals/Demo Agents/demo-agents/hr.json\" \\\n  --llm-key=$OPENAI_KEY\n\n# Or copy them to your project first\ncp -r node_modules/@aui.io/evals/Demo\\ Agents ./test-suites\nnpx aui-evals run \"./test-suites/demo-agents/hr.json\" --llm-key=$OPENAI_KEY\n```\n\n## Custom Adapter\n\nImplement `AgentAdapter` to use a different agent backend:\n\n```typescript\ninterface AgentAdapter {\n  spawnSession(agentId: string, apiKey: string, label: string, model?: string): Promise<string>;\n  sendMessage(sessionKey: string, message: string, userInfo?: UserInfo, timeout?: number): Promise<AgentResponse>;\n  callJudge(prompt: string, model?: string): Promise<string>;\n  generateUserMessage(prompt: string, model?: string): Promise<string>;\n}\n```\n\n## Example Test Suite\n\n```json\n{\n  \"name\": \"Customer Support Evaluations\",\n  \"agentId\": \"your-agent-id\",\n  \"apiKey\": \"your-api-key\",\n  \"tests\": [\n    {\n      \"name\": \"Basic support request — should resolve without escalation\",\n      \"userInfo\": {\n        \"name\": \"Jane Doe\",\n        \"email\": \"jane@example.com\",\n        \"subscription_status\": \"active\",\n        \"plan_name\": \"Pro Monthly\",\n        \"last_amount_charged\": \"29.99\"\n      },\n      \"guidelines\": \"You are a user with a billing question. Be polite and cooperative.\",\n      \"turns\": [\n        { \"user\": \"Hi, I have a question about my last charge.\" },\n        { \"user\": \"auto\" }\n      ],\n      \"evaluations\": [\n        {\n          \"criterion\": \"Agent should not escalate a simple billing question\",\n          \"strict\": \"not_escalated\",\n          \"category\": \"accuracy\",\n          \"weight\": 2\n        },\n        {\n          \"criterion\": \"Billing inquiry workflow should activate\",\n          \"strict\": \"tool_used\",\n          \"toolName\": \"BILLING_INQUIRY\",\n          \"category\": \"accuracy\"\n        },\n        {\n          \"criterion\": \"Eligibility check rule should fire\",\n          \"strict\": \"rule_triggered\",\n          \"ruleCode\": \"check-billing-eligibility\",\n          \"category\": \"compliance\",\n          \"aui_logic\": [\"R1_billing_check\"]\n        },\n        {\n          \"criterion\": \"Agent should extract the inquiry type\",\n          \"strict\": \"param_exists\",\n          \"paramName\": \"inquiry-type\",\n          \"category\": \"accuracy\"\n        },\n        {\n          \"criterion\": \"Agent maintained a friendly and professional tone\",\n          \"category\": \"tone\",\n          \"weight\": 1\n        },\n        {\n          \"criterion\": \"Agent should not produce an error response\",\n          \"strict\": \"not_escalated_due_to_error\",\n          \"category\": \"safety\"\n        }\n      ],\n      \"tags\": [\"billing\", \"no-escalation\"]\n    }\n  ]\n}\n```\n","readmeFilename":"README.md"}