{"_id":"@appliqation/autotest","_rev":"6-6f8fc5169102680c772b7fd979556a0f","name":"@appliqation/autotest","dist-tags":{"latest":"0.1.6"},"versions":{"0.1.1":{"name":"@appliqation/autotest","version":"0.1.1","_id":"@appliqation/autotest@0.1.1","maintainers":[{"name":"archana6","email":"accounts@appliqation.io"}],"bin":{"appliqation-autotest":"dist/cli/index.js"},"dist":{"shasum":"7f61889b1a730b6b90543da0c0156b7323d6db0b","tarball":"https://registry.npmjs.org/@appliqation/autotest/-/autotest-0.1.1.tgz","fileCount":25,"integrity":"sha512-lBCAJPT8333qhqestGvUDk/8b0mvoBTviVE+X3mGQnnJLJIGL3KKfOPHI3Jnb3ukazoxTjX69yCOBtR+hXfLCw==","signatures":[{"sig":"MEUCIQDCnMMKqawePMLkjG7QwV7POrNHCKar7TxTBTDuczbMXAIgND3hND33DlvI474NFf+yJ2+Xa3KhX50V3D8JmDSUP8M=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":98366},"type":"module","engines":{"node":">=20"},"gitHead":"9d6673884081b031c97d707c11cc97773a213ddb","scripts":{"dev":"tsx src/cli/index.ts","lint":"eslint src --ext .ts","test":"vitest run","build":"tsc -p tsconfig.build.json","typecheck":"tsc -p tsconfig.json --noEmit","test:watch":"vitest"},"_npmUser":{"name":"archana6","email":"accounts@appliqation.io"},"_npmVersion":"11.7.0","description":"Standalone autonomous testing agent that executes Appliqation MCP workflows against a live app under test.","directories":{},"_nodeVersion":"23.10.0","dependencies":{"zod":"^3.24.2","dotenv":"^16.4.7","commander":"^13.1.0","playwright":"^1.51.0","@appliqation/agent-core":"^0.1.0","@appliqation/automation-sdk":"^2.7.0"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.19.3","vitest":"^3.2.7","typescript":"^5.8.2","@types/node":"^22.13.10"},"_npmOperationalInternal":{"tmp":"tmp/autotest_0.1.1_1787298000375_0.9076240793267794","host":"s3://npm-registry-packages-npm-production"}},"0.1.2":{"name":"@appliqation/autotest","version":"0.1.2","license":"MIT","_id":"@appliqation/autotest@0.1.2","maintainers":[{"name":"archana6","email":"accounts@appliqation.io"}],"homepage":"https://github.com/appliqation/autotest#readme","bugs":{"url":"https://github.com/appliqation/autotest/issues"},"bin":{"appliqation-autotest":"dist/cli/index.js"},"dist":{"shasum":"ea97809b3119bef38d6ce84ce3947dcc3e1c5b98","tarball":"https://registry.npmjs.org/@appliqation/autotest/-/autotest-0.1.2.tgz","fileCount":25,"integrity":"sha512-wUAuqoddJS7eHiOx5rsUyl9B9cg+DLh2b1sedvH/szvW+Rn4DMeoDwE+ySG8TJjBqAQwvDZsKWs0YY85dekhKg==","signatures":[{"sig":"MEUCIQClnUOaZNkHS5SUoD56I7tVe8Qp+yF1uwoH9yyVBusoNAIgGIwPB++ImqxGGrmaLqlL9ACFjMIv/GojEOilI1C/Uk0=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":100578},"type":"module","engines":{"node":">=20"},"gitHead":"7af437c61494a8d250fc1e5c38f740deb12c2b7d","scripts":{"dev":"tsx src/cli/index.ts","lint":"eslint src --ext .ts","test":"vitest run","build":"tsc -p tsconfig.build.json","typecheck":"tsc -p tsconfig.json --noEmit","test:watch":"vitest"},"_npmUser":{"name":"archana6","email":"accounts@appliqation.io"},"repository":{"url":"git+https://github.com/appliqation/autotest.git","type":"git"},"_npmVersion":"11.7.0","description":"Standalone autonomous testing agent that executes Appliqation MCP workflows against a live app under test.","directories":{},"_nodeVersion":"23.10.0","dependencies":{"dotenv":"^16.4.7","commander":"^13.1.0","playwright":"^1.51.0","@appliqation/agent-core":"^0.1.5","@appliqation/automation-sdk":"^2.7.0"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.19.3","eslint":"^9.39.5","vitest":"^3.2.7","@eslint/js":"^9.39.5","typescript":"^5.8.2","@types/node":"^22.13.10","typescript-eslint":"^8.67.0"},"_npmOperationalInternal":{"tmp":"tmp/autotest_0.1.2_1787382366505_0.34011817339250494","host":"s3://npm-registry-packages-npm-production"}},"0.1.3":{"name":"@appliqation/autotest","version":"0.1.3","license":"MIT","_id":"@appliqation/autotest@0.1.3","maintainers":[{"name":"archana6","email":"accounts@appliqation.io"}],"homepage":"https://github.com/appliqation/autotest#readme","bugs":{"url":"https://github.com/appliqation/autotest/issues"},"bin":{"appliqation-autotest":"dist/cli/index.js"},"dist":{"shasum":"16fe6de4b20912e2eb688f5a1602c16ed90db982","tarball":"https://registry.npmjs.org/@appliqation/autotest/-/autotest-0.1.3.tgz","fileCount":25,"integrity":"sha512-+uGzoknhp48eTxayJHgW0IdH47DSqIWdJWV6LcuTUrXFrIoMvALTFp2GS9Y6nQiOmYn+KF5JVbLKuFn3AGFYIQ==","signatures":[{"sig":"MEUCIQDqSVnGDafW/NltOoF1wQRNv+HC1cr/hRpQkE0otfYHrQIgRmmXk9XMnsRsAfAuRLRQPdBJrjdny4KROnsysswrSeU=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":102429},"type":"module","engines":{"node":">=20"},"gitHead":"bf4173e06fcf0add0f29ccc9d92dcea6f7ebe50a","scripts":{"dev":"tsx src/cli/index.ts","lint":"eslint src --ext .ts","test":"vitest run","build":"tsc -p tsconfig.build.json","typecheck":"tsc -p tsconfig.json --noEmit","test:watch":"vitest"},"_npmUser":{"name":"archana6","email":"accounts@appliqation.io"},"repository":{"url":"git+https://github.com/appliqation/autotest.git","type":"git"},"_npmVersion":"11.7.0","description":"Standalone autonomous testing agent that executes Appliqation MCP workflows against a live app under test.","directories":{},"_nodeVersion":"23.10.0","dependencies":{"dotenv":"^16.4.7","commander":"^13.1.0","playwright":"^1.51.0","@appliqation/agent-core":"^0.1.5","@appliqation/automation-sdk":"^2.7.0"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.19.3","eslint":"^9.39.5","vitest":"^3.2.7","@eslint/js":"^9.39.5","typescript":"^5.8.2","@types/node":"^22.13.10","typescript-eslint":"^8.67.0"},"_npmOperationalInternal":{"tmp":"tmp/autotest_0.1.3_1788164785408_0.9023774578820907","host":"s3://npm-registry-packages-npm-production"}},"0.1.4":{"name":"@appliqation/autotest","version":"0.1.4","license":"MIT","_id":"@appliqation/autotest@0.1.4","maintainers":[{"name":"archana6","email":"accounts@appliqation.io"}],"homepage":"https://github.com/appliqation/autotest#readme","bugs":{"url":"https://github.com/appliqation/autotest/issues"},"bin":{"appliqation-autotest":"dist/cli/index.js"},"dist":{"shasum":"5af2367d99838f6a808a8baf43b5dacdd5ec3bce","tarball":"https://registry.npmjs.org/@appliqation/autotest/-/autotest-0.1.4.tgz","fileCount":25,"integrity":"sha512-fOGEv4iNj/hpH0+gX7TX2GgtgI17qrb9BklL1IzYhMlrMBVsNcWnVAUcF+7OuojlqgKJempLP6y6s9R1WZ6W6Q==","signatures":[{"sig":"MEYCIQC6MNdVjTm50iSqCw0ZSufP38nAz5GNTTa+xyEOiOYKbAIhALlaL9rs1MxA6h3KZ2zzo7U2ws1Hu3tHNmpBt2XW7Zlw","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":101553},"type":"module","engines":{"node":">=20"},"gitHead":"96012cade0246800b4aa3591e61df79f8e90c1fc","scripts":{"dev":"tsx src/cli/index.ts","lint":"eslint src --ext .ts","test":"vitest run","build":"tsc -p tsconfig.build.json","typecheck":"tsc -p tsconfig.json --noEmit","test:watch":"vitest"},"_npmUser":{"name":"archana6","email":"accounts@appliqation.io"},"repository":{"url":"git+https://github.com/appliqation/autotest.git","type":"git"},"_npmVersion":"11.7.0","description":"Standalone autonomous testing agent that executes Appliqation MCP workflows against a live app under test.","directories":{},"_nodeVersion":"23.10.0","dependencies":{"dotenv":"^16.4.7","commander":"^13.1.0","playwright":"^1.51.0","@appliqation/agent-core":"^0.1.6","@appliqation/automation-sdk":"^2.7.0"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.19.3","eslint":"^9.39.5","vitest":"^3.2.7","@eslint/js":"^9.39.5","typescript":"^5.8.2","@types/node":"^22.13.10","typescript-eslint":"^8.67.0"},"_npmOperationalInternal":{"tmp":"tmp/autotest_0.1.4_1788423847961_0.017447578977048295","host":"s3://npm-registry-packages-npm-production"}},"0.1.5":{"name":"@appliqation/autotest","version":"0.1.5","license":"MIT","_id":"@appliqation/autotest@0.1.5","maintainers":[{"name":"archana6","email":"accounts@appliqation.io"}],"homepage":"https://github.com/appliqation/autotest#readme","bugs":{"url":"https://github.com/appliqation/autotest/issues"},"bin":{"appliqation-autotest":"dist/cli/index.js"},"dist":{"shasum":"6a7ea1140a3a3e3b2ec679a87e64cc2a62715fff","tarball":"https://registry.npmjs.org/@appliqation/autotest/-/autotest-0.1.5.tgz","fileCount":25,"integrity":"sha512-aXY829J9ZNwh3TVBh8BGpc/xNSCs7yU15qSrPmg6PhBv9W8g/UTTftfyvucwLsfy3WeKibL2r42cJTZtnNXOFA==","signatures":[{"sig":"MEUCIQCW5N1rZwnDx3gEt18MRa0BUpRe1Llek6pPVVdtxGM5ZQIgf47SQmzUYRdkzjrV/5iqn/dPEAGeQzJHQYs3svgzw9c=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":104742},"type":"module","engines":{"node":">=20"},"gitHead":"1ad3a58f7e380b928c1a070f3fdc4ee92c5befc1","scripts":{"dev":"tsx src/cli/index.ts","lint":"eslint src --ext .ts","test":"vitest run","build":"tsc -p tsconfig.build.json","typecheck":"tsc -p tsconfig.json --noEmit","test:watch":"vitest"},"_npmUser":{"name":"archana6","email":"accounts@appliqation.io"},"repository":{"url":"git+https://github.com/appliqation/autotest.git","type":"git"},"_npmVersion":"11.7.0","description":"Standalone autonomous testing agent that executes Appliqation MCP workflows against a live app under test.","directories":{},"_nodeVersion":"23.10.0","dependencies":{"dotenv":"^16.4.7","commander":"^13.1.0","playwright":"^1.51.0","@appliqation/agent-core":"^0.1.7","@appliqation/automation-sdk":"^2.7.0"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.19.3","eslint":"^9.39.5","vitest":"^3.2.7","@eslint/js":"^9.39.5","typescript":"^5.8.2","@types/node":"^22.13.10","typescript-eslint":"^8.67.0"},"_npmOperationalInternal":{"tmp":"tmp/autotest_0.1.5_1788433770186_0.929968756090843","host":"s3://npm-registry-packages-npm-production"}},"0.1.6":{"name":"@appliqation/autotest","version":"0.1.6","description":"Standalone autonomous testing agent that executes Appliqation MCP workflows against a live app under test.","license":"MIT","repository":{"type":"git","url":"git+https://github.com/appliqation/autotest.git"},"homepage":"https://github.com/appliqation/autotest#readme","bugs":{"url":"https://github.com/appliqation/autotest/issues"},"type":"module","bin":{"appliqation-autotest":"dist/cli/index.js"},"engines":{"node":">=20"},"scripts":{"build":"tsc -p tsconfig.build.json","dev":"tsx src/cli/index.ts","typecheck":"tsc -p tsconfig.json --noEmit","lint":"eslint src --ext .ts","test":"vitest run","test:watch":"vitest"},"dependencies":{"@appliqation/agent-core":"^0.1.7","@appliqation/automation-sdk":"^2.7.0","commander":"^13.1.0","dotenv":"^16.4.7","playwright":"^1.51.0"},"devDependencies":{"@eslint/js":"^9.39.5","@types/node":"^22.13.10","eslint":"^9.39.5","tsx":"^4.19.3","typescript":"^5.8.2","typescript-eslint":"^8.67.0","vitest":"^3.2.7"},"gitHead":"d3b0e15c02c7372461f0a3eadb8d9516aedc5a04","_id":"@appliqation/autotest@0.1.6","_nodeVersion":"23.10.0","_npmVersion":"11.7.0","dist":{"integrity":"sha512-1EUKvWHdAiqgrrQJI+zBPWaezSh+RkcPKS+znNBTRIagzk4i6WoSkijtv0kP9/N523MKuOtnDYuSJQdAl2jpCw==","shasum":"a8d6280ec95717cbdbc93344bb58b95e7bae6682","tarball":"https://registry.npmjs.org/@appliqation/autotest/-/autotest-0.1.6.tgz","fileCount":25,"unpackedSize":107185,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQC5b1qGoRYopaDd/K64QgnFdUXlGbF+Vi+8fzAzRX3hvQIgCwkNu0cGo4EFkfCDFQ8X/UpwQAUc5b8f0eQg7WxM6ek="}]},"_npmUser":{"name":"archana6","email":"accounts@appliqation.io"},"directories":{},"maintainers":[{"name":"archana6","email":"accounts@appliqation.io"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/autotest_0.1.6_1788493083727_0.17063123043917328"},"_hasShrinkwrap":false}},"time":{"created":"2026-08-21T07:40:00.252Z","modified":"2026-09-04T03:38:04.080Z","0.1.1":"2026-08-21T07:40:00.528Z","0.1.2":"2026-08-22T07:06:06.648Z","0.1.3":"2026-08-31T08:26:25.557Z","0.1.4":"2026-09-03T08:24:08.091Z","0.1.5":"2026-09-03T11:09:30.331Z","0.1.6":"2026-09-04T03:38:03.905Z"},"bugs":{"url":"https://github.com/appliqation/autotest/issues"},"license":"MIT","homepage":"https://github.com/appliqation/autotest#readme","repository":{"type":"git","url":"git+https://github.com/appliqation/autotest.git"},"description":"Standalone autonomous testing agent that executes Appliqation MCP workflows against a live app under test.","maintainers":[{"name":"archana6","email":"accounts@appliqation.io"}],"readme":"# Appliqation Autotest\n\n**Autonomously executes a test case in a real browser, then has a second, independent AI judge the result from evidence alone — never from the first agent's own claim.**\n\nPoint it at one test case, a whole scenario, or a whole test set (regression/sanity/smoke — the most common CI shape), and it drives a real Playwright browser, captures real evidence (screenshots, console/network logs, accessibility snapshots), and writes an honest, appq-polled verdict back to Appliqation. No fabricated pass/fail — a validator that can't confirm something reports `blocked`, not a guess.\n\n## Why two agents, not one\n\nA single model that both executes a test and grades its own execution is grading its own homework. This repo genuinely separates the two roles — **executor** and **validator** each run as their own fresh, isolated tool-calling loop with no shared context between them. The validator never sees the executor's reasoning, only what it explicitly submitted as evidence via `submit_execution_evidence`. That's the entire mechanism behind trustworthy self-verification here: isolation, not a prompt asking the model to \"be objective.\"\n\n## How it works\n\n```mermaid\nsequenceDiagram\n    participant E as Executor\n    participant B as Real Browser\n    participant Ev as Evidence Store\n    participant V as Validator\n    participant A as Appliqation\n\n    E->>B: drive the test steps\n    B-->>E: screenshots, console/network, DOM\n    E->>Ev: submit_execution_evidence\n    Note over E,V: fresh context — no shared conversation\n    V->>Ev: read only the submitted evidence\n    V->>V: judge each step: met / not_met / blocked\n    V->>A: write the real verdict (update_run_results)\n    A-->>V: authoritative run status\n```\n\n- **Verdicts are polled, not parsed.** The final status comes from Appliqation's own run matrix (`get_test_results`), not scraped out of the validator's report prose.\n- **A destructive-action gate** blocks any click matching a destructive-verb/`mailto:`/`tel:`/`sms:` pattern before it ever dispatches — checked in code, not left to the model to notice.\n- **Per-TC role inference.** Mixed-role scenarios (admin sees X, standard user gets 403 on the same page) get the right authenticated session per test case automatically, from each TC's own tag or name — no manual per-run role juggling.\n\n## Quick start\n\n```bash\nnpm install -g @appliqation/autotest\nnpx playwright install chromium\n```\n\nCreate a `.env` file (in whatever directory you'll run it from) with:\n\n```\nAPPQ_API_KEY=your-appliqation-api-key\nANTHROPIC_API_KEY=your-anthropic-key   # or OPENAI_API_KEY — pick one\n```\n\n```bash\n# one test case\nappliqation-autotest judge --test-case-uuid <uuid> --environment Stage --dry-run\n\n# an entire scenario\nappliqation-autotest judge --scenario-id <id> --environment Stage --dry-run\n\n# a whole test set (regression / sanity / smoke — the common CI shape; can span multiple scenarios)\nappliqation-autotest judge --test-set-id <id> --environment Stage --dry-run\n```\n\nExactly one of `--test-case-uuid` / `--scenario-id` / `--test-set-id` is required — mutually exclusive scopes, not combinable.\n\n`--dry-run` is the recommended default for your first run against a real project — it computes real verdicts but suppresses the actual Appliqation writeback. Drop it once you trust the result. `--coverage` (`always` / `on-script-absence` / `sampled:N` / `external`) controls when this agentic pass runs alongside your existing deterministic Playwright pipeline in scenario/test-set mode; `--json`/`--ci` give a structured summary and a CI-friendly exit code.\n\n## CLI reference\n\n`appliqation-autotest judge [options]`\n\n**Scope — exactly one required:**\n\n| Option | Description |\n|---|---|\n| `--test-case-uuid <uuid>` | One test case to judge. `scenario_id` is always derived from it — `--scenario-id` is not accepted alongside it. |\n| `--scenario-id <id>` | An entire scenario, agentic pair judged per TC per the coverage policy. |\n| `--test-set-id <id>` | An entire test set (can span multiple scenarios, the common regression/sanity/smoke shape) — gets exactly one shared run covering every TC in it. `--run-id` is not supported in this mode. |\n\n**Required:**\n\n| Option | Description |\n|---|---|\n| `--environment <name>` | Environment name — its URL is what the browser navigates to. |\n\n**Optional:**\n\n| Option | Description |\n|---|---|\n| `--run-id <id>` | Reuse an existing run instead of creating one. Not available in test-set mode. |\n| `--role <name>` | Authenticate the executor as this role before navigating, using the Playwright storageState `appq-auth-setup` writes. Omit for ungated projects. |\n| `--coverage <policy>` | `always` \\| `on-script-absence` (default) \\| `on-failure-or-absence` \\| `sampled:N` \\| `external` — only meaningful in whole-scenario/test-set mode; decides when the agentic pair runs alongside the deterministic canonical-script pipeline. |\n| `--test-type <ui\\|api>` | Force `ui` (browser) or `api` (`http_request`) execution for every TC this invocation touches. Omit and each TC's own tag decides instead. |\n| `--poll-timeout-ms <ms>` | Whole-scenario/test-set mode: how long to wait for the deterministic path to settle before reporting. Defaults to `POLL_TIMEOUT_MS`. |\n| `--mandatory-image-check` | Fetch and attach every step's screenshot to the validator unconditionally, instead of leaving it to the model's own `view_screenshot` judgment. Real cost/reliability tradeoff, not free. |\n| `--dry-run` | Compute verdicts normally but suppress the actual `update_run_results`/`create_defect` calls — logs what would have been sent instead. |\n| `--json` | Print the final result as a single JSON object instead of a human-readable table. |\n| `--ci` | Shorthand for `--json`. |\n\n## Configuration\n\nCopy `.env.example` to `.env`. Requires `APPQ_API_KEY` and one of `ANTHROPIC_API_KEY`/`OPENAI_API_KEY`. Separate executor/validator model overrides are supported — a cheaper model for judging captured evidence is a reasonable choice even when the executor needs a stronger one for open-ended browsing.\n\n## Running this safely\n\nThe executor drives a real browser turn by turn on the model's own decisions — where to navigate, what to click, what to inspect — against whatever page content the site under test happens to serve. The destructive-action gate on `browser_click`/`browser_evaluate` (see `@appliqation/agent-core`'s `destructiveActionGate.ts`) blocks the obvious failure mode, but it's a code-level backstop, not a substitute for containment: this process holds `APPQ_API_KEY`, your LLM provider key, and (for authenticated runs) real project credentials, all reachable from wherever the model decides to navigate.\n\n**Run this inside a container with an egress allowlist**, not directly on a machine with broad network access. This process only ever legitimately needs to reach:\n\n- your LLM provider (`api.anthropic.com` or `api.openai.com`)\n- your configured `APPQ_ORIGIN` (`appq.appliqation.io` by default)\n- the project's own site under test — whatever URL `--environment` resolves to via `get_project_settings`\n\nAnything else this process tries to reach is unexpected and worth investigating, not routing around.\n\n## Development\n\n```bash\ngit clone https://github.com/appliqation/autotest.git\ncd autotest\nnpm install\ncp .env.example .env   # fill in APPQ_API_KEY and one LLM provider key\nnpm run dev -- judge --test-case-uuid <uuid> --environment <name>\nnpm run typecheck\nnpm test\n```\n\nSee `CLAUDE.md` for a map of this repo if you're working in it with an AI coding assistant.\n\n## License\n\nMIT — see [LICENSE](./LICENSE).\n","readmeFilename":"README.md"}