{"_id":"@cordya-ai/froid-test-architecture-enterprise","name":"@cordya-ai/froid-test-architecture-enterprise","dist-tags":{"latest":"1.23.5"},"versions":{"1.23.5":{"$schema":"https://json.schemastore.org/package.json","name":"@cordya-ai/froid-test-architecture-enterprise","version":"1.23.5","description":"Master Test Architect for quality strategy, test automation, and release gates","keywords":["froid","test-architect","test-engineering","testing","quality","quality-gates","risk-based-testing","verification-and-validation","traceability","nfr","test-automation","automation","contract-testing","pact","playwright","cypress","mobile","maestro"],"repository":{"type":"git","url":"git+https://github.com/cordya-ai/froid-plane-test-architecture-enterprise.git"},"license":"MIT","author":{"name":"Murat K Ozcan","url":"TEA Creator"},"main":"","bin":{"tea-test-review":"cli/test-review.js"},"scripts":{"docs:build":"node tools/build-docs.js","docs:dev":"npm --prefix website run dev","docs:fix-links":"node tools/fix-doc-links.js --write","docs:preview":"npm --prefix website run preview","docs:validate-links":"node tools/validate-doc-links.js","eval:all":"node test/eval-all.js","eval:fragment-selection":"node test/eval-fragment-selection.js","eval:test-review":"node test/eval-test-review.js","format:check":"prettier --check .","format:fix":"prettier --write .","format:fix:staged":"prettier --write","lint":"eslint . --max-warnings 0","lint:fix":"eslint . --fix","lint:md":"markdownlint-cli2 '**/*.md'","prepare":"command -v husky >/dev/null 2>&1 && husky || exit 0","release:major":"gh workflow run publish.yaml -f channel=latest -f bump=major","release:minor":"gh workflow run publish.yaml -f channel=latest -f bump=minor","release:next":"gh workflow run publish.yaml -f channel=next","release:patch":"gh workflow run publish.yaml -f channel=latest -f bump=patch","test":"npm run test:schemas && npm run test:install && npm run test:knowledge && npm run test:criteria-fragments && npm run test:enforce-hook && npm run test:eval-data && npm run test:release-metadata && npm run test:changelog && npm run test:tea-workflow-descriptions && npm run validate:schemas && npm run lint && npm run lint:md && npm run format:check","test:changelog":"node test/test-stamp-changelog.js","test:cli":"node test/test-test-review-cli.js","test:coverage":"c8 npm test","test:criteria-fragments":"node tools/validate-criteria-fragments.js","test:enforce-hook":"node test/test-enforce-hook.js","test:eval-data":"node test/eval-fragment-selection.js --validate-only","test:install":"node test/test-installation-components.js","test:knowledge":"node test/test-knowledge-base.js","test:release-metadata":"node test/test-release-metadata.js","test:schemas":"node test/test-agent-schema.js","test:tea-workflow-descriptions":"node tools/validate-tea-workflow-descriptions.js","validate:schemas":"node tools/validate-agent-schema.js","validate:tea-workflow-descriptions":"node tools/validate-tea-workflow-descriptions.js"},"lint-staged":{"*.{js,cjs,mjs}":["eslint --fix","prettier --write --ignore-unknown"],"*.yaml":["eslint --fix","prettier --write --ignore-unknown"],"*.json":["prettier --write --ignore-unknown"],"*.md":["markdownlint-cli2 --fix","prettier --write --ignore-unknown"]},"c8":{"all":true,"reporter":["text","html"],"reports-dir":"coverage"},"dependencies":{"@clack/prompts":"^0.11.0","boxen":"^5.1.2","cli-table3":"^0.6.5","commander":"^14.0.0","csv-parse":"^6.1.0","glob":"^11.0.3","ignore":"^7.0.5","js-yaml":"^4.1.0","ora":"^5.4.1","semver":"^7.6.3","wrap-ansi":"^7.0.0","xml2js":"^0.6.2","yaml":"^2.7.0"},"devDependencies":{"@astrojs/sitemap":"^3.6.0","@astrojs/starlight":"^0.37.0","@eslint/js":"^9.33.0","archiver":"^7.0.1","astro":"^5.16.0","c8":"^10.1.3","eslint":"^9.33.0","eslint-config-prettier":"^10.1.8","eslint-plugin-n":"^17.21.3","eslint-plugin-unicorn":"^60.0.0","eslint-plugin-yml":"^1.18.0","husky":"^9.1.7","jest":"^30.0.4","lint-staged":"^16.1.1","markdownlint-cli2":"^0.19.1","prettier":"^3.7.4","prettier-plugin-packagejson":"^2.5.19","sharp":"^0.33.5","yaml-eslint-parser":"^1.2.3","yaml-lint":"^1.7.0"},"engines":{"node":">=20.0.0"},"publishConfig":{"access":"public"},"_id":"@cordya-ai/froid-test-architecture-enterprise@1.23.5","bugs":{"url":"https://github.com/cordya-ai/froid-plane-test-architecture-enterprise/issues"},"homepage":"https://github.com/cordya-ai/froid-plane-test-architecture-enterprise#readme","_nodeVersion":"26.7.0","_npmVersion":"11.19.0","dist":{"integrity":"sha512-AtpkigWYaTpA+m60A69Z/xRqyILd/DUavlDe638vjleGITbSjy35CPPSdQ/wywV4VxB/HvF+OQZSXOkKrya+vA==","shasum":"ca621252c949d446cc10559edeb4bf619c1d9ba4","tarball":"https://registry.npmjs.org/@cordya-ai/froid-test-architecture-enterprise/-/froid-test-architecture-enterprise-1.23.5.tgz","fileCount":945,"unpackedSize":12264642,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEYCIQCWgDQjrX1i3PyCkDHV7ScXbGFyDJ+IZOE2ask9+Qcj5QIhAPllzLF6KONf8kP/u3x639poB+ZHwlywvhKn0pk0JsMH"}]},"_npmUser":{"name":"evandroreis-cordya","email":"npm@cordya.ai"},"directories":{},"maintainers":[{"name":"evandroreis-cordya","email":"npm@cordya.ai"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/froid-test-architecture-enterprise_1.23.5_1788314041385_0.013724674246312407"},"_hasShrinkwrap":false}},"time":{"created":"2026-09-02T01:54:01.245Z","1.23.5":"2026-09-02T01:54:01.593Z","modified":"2026-09-02T01:54:01.868Z"},"maintainers":[{"name":"evandroreis-cordya","email":"npm@cordya.ai"}],"description":"Master Test Architect for quality strategy, test automation, and release gates","homepage":"https://github.com/cordya-ai/froid-plane-test-architecture-enterprise#readme","keywords":["froid","test-architect","test-engineering","testing","quality","quality-gates","risk-based-testing","verification-and-validation","traceability","nfr","test-automation","automation","contract-testing","pact","playwright","cypress","mobile","maestro"],"repository":{"type":"git","url":"git+https://github.com/cordya-ai/froid-plane-test-architecture-enterprise.git"},"author":{"name":"Murat K Ozcan","url":"TEA Creator"},"bugs":{"url":"https://github.com/cordya-ai/froid-plane-test-architecture-enterprise/issues"},"license":"MIT","readme":"# TEA: Test Engineering Architect\n\n[Node Version](https://nodejs.org)\n[License: MIT](./LICENSE)\n\n**TEA** stands for **Test Engineering Architect**. The npm package and repository slug `froid-plane-test-architecture-enterprise` is a package name and never an expansion of the acronym.\n\nTEA is a standalone FROID module that delivers risk-based test strategy, test automation guidance, and release gate decisions. It ships:\n\n- one expert agent, Murat, Master Test Architect and Quality Advisor\n- nine workflows spanning Teach Me Testing (TEA Academy), test design, framework setup, CI guidance, ATDD, automation, test review, NFR Evidence Audit, and traceability\n- a 35-row criteria registry that fixes the severity of every reviewable violation, so a score is a lookup rather than a judgment call\n- `tea-test-review`, a headless CLI that runs the review workflow as a CI gate with real exit codes\n- a write-time enforcement hook that blocks the mechanically decidable violations before they reach disk\n\nTEA is two layers. **TEA Core** decides what must be verified, at what depth, with what evidence, and whether that evidence is sufficient to release; it assumes nothing about your language, framework, or platform. **Execution targets** turn those decisions into runnable tests on a specific stack, and that layer is swappable. See [Verification Architecture](./docs/explanation/verification-architecture.md) for the split, and [Execution Targets](./docs/reference/execution-targets.md) for exactly which stacks are covered at which depth.\n\nDocs: [https://cordya-ai.github.io/froid-plane-test-architecture-enterprise/](https://cordya-ai.github.io/froid-plane-test-architecture-enterprise/)\n\n## Why TEA\n\n- Risk-based prioritization (P0-P3) from probability × impact, with measurable quality gates\n- Requirements traced to evidence, and PASS / CONCERNS / FAIL / WAIVED release decisions that survive an audit\n- NFR thresholds set at design time and audited against real evidence, defaulting to CONCERNS when evidence is missing\n- Consistent, knowledge-base driven outputs instead of whatever the model felt like producing\n- Stack-aware execution: Playwright and Cypress for browsers, Maestro for mobile native, Pact for contracts, pytest / JUnit / Go test / xUnit / RSpec for backend services, k6 and scanners as NFR evidence\n- Three enforcement points rather than one: fragments steer generation, a hook blocks the write, and `test-review` scores what actually landed\n\n## How Froid Works\n\nFroid works because it turns big, fuzzy work into **repeatable workflows**. Each workflow is broken into small steps with clear instructions, so the AI follows the same path every time. It also uses a **shared knowledge base** (standards and patterns) so outputs are consistent, not random. In short: **structured steps + shared standards = reliable results**.\n\n## How TEA Fits In\n\nTEA plugs into Froid the same way a specialist plugs into a team. It uses the same step‑by‑step workflow engine and shared standards, but focuses exclusively on testing and quality gates. That means you get a **risk‑based test plan**, **automation guidance**, and **go/no‑go decisions** that align with the rest of the Froid process.\n\n## How It Actually Works\n\n### The problem it is built around\n\nAsk a model to \"write tests for this feature\" and you reliably get four things: redundant coverage, incorrect assertions, flaky tests, and diffs nobody can review. The cause is a category error. Prompt-driven generation is nondeterministic, and it is being pointed at the one artifact whose entire job is determinism.\n\nTEA's answer is not a better prompt. It is to make the work repeatable at three levels.\n\n**Repeatable instructions.** A single 5,000-word instruction file fails predictably: the model skims it, improvises past the vague parts (\"analyze codebase then generate tests\" specifies nothing), keeps going because nothing told it where to stop, and returns something different next run. So every workflow is cut into step files that each do one thing, state what \"finished\" means, restate the context they need, list what they must not do, and load one at a time. Consistent output for the same input is what makes everything else possible: you cannot parallelize work whose boundaries are undefined.\n\n**Repeatable standards.** 59 knowledge fragments carry the patterns, and an index decides which ones enter context for the task at hand. The model is not asked to remember how fixtures compose or what network-first means. It is handed the fragment.\n\n**Repeatable judgment.** Risk scores, priorities, quality scores, and gate decisions are computed from stated rules rather than produced as opinions. This is the part most tools skip, and it is the difference between a review you can act on and a review you have to re-litigate.\n\n### The order you run things\n\nThe nine workflows are a directed graph, not a menu. `src/module-help.csv` encodes it.\n\n```text\nPhase 3, solutioning, once per project\n  TD  test-design (system-level)  →  TF  framework  →  CI  ci\n\nPhase 4, implementation, per story\n  create-story  →  AT  atdd  →  dev implements  →  TA  automate\n\nEpic or release gate\n  TA  automate  →  RV  test-review\n  TA  automate  →  NR  nfr\n  RV  test-review  →  TR  trace (Phase 2 gate decision)\n```\n\nPhase 3 order matters and is deliberate: run `test-design` first so NFR evidence needs can shape the infrastructure, then `framework` once the architecture and the test design have settled the stack, then `ci` once the framework exists so the pipeline wires to real commands.\n\n`test-design` is dual-mode. At system level it produces an architecture-facing document and a QA-facing one. Per epic it produces `test-design-epic-N.md`. `teach-me-testing` sits outside the lifecycle and runs once per learner.\n\n`module-help.csv`'s single `phase` column records the phase a workflow's catalog row is sequenced under (its `preceded-by`/`followed-by` chain), not every phase the workflow can run in. `test-design`'s row is `3-solutioning` because that is the chain the row encodes (`test-design` → `framework`); the epic-level Phase 4 invocation above has no dependency edges of its own and so gets no second row, only this prose.\n\nFor the full lifecycle diagram including the Froid phases around TEA, see [TEA Overview](./docs/explanation/tea-overview.md).\n\n### One epic, end to end\n\nHere is the same feature moving through the chain, with the rules TEA actually applies at each step.\n\n**1. Risk, in** `test-design`**.** Every identified risk gets a probability of 1 to 3 (unlikely, possible, likely) and an impact of 1 to 3 (minor, degraded, critical). Score is the product, so the range is 1 to 9, and the score determines the action:\n\n| Score | Action   | Gate impact          |\n| ----- | -------- | -------------------- |\n| 1-3   | DOCUMENT | none                 |\n| 4-5   | MONITOR  | none, watch closely  |\n| 6-8   | MITIGATE | CONCERNS at the gate |\n| 9     | BLOCK    | automatic FAIL       |\n\nA checkout risk scored `probability 2 × impact 3 = 6` lands in MITIGATE. It gets a row in the test design with a named owner and a date, and it will surface as CONCERNS at the gate until the mitigation is real.\n\nPriority is a separate judgment that the risk score informs rather than determines. P0 is revenue-critical, security-critical, data-integrity, regulatory, or previously broken. P1 is core journeys and complex logic. P2 is secondary features. P3 is nice-to-have. The design pins the effort too: P0 tests are budgeted at 2 hours each, P1 at 1, P2 at 0.5, P3 at 0.25.\n\n**2. Test level, still in** `test-design`**.** Favor unit when logic can be isolated with no side effects; integration for persistence, service contracts, and component boundaries; E2E for user-facing critical paths and multi-system interactions. Before adding any test, the duplicate-coverage guard asks whether a lower level already covers it. Overlap is allowed only for genuinely different aspects, defense in depth on critical paths, or regression prevention on something that broke before.\n\n**3. Red tests, in** `atdd` (optional). Run before implementation. It generates acceptance tests that all carry `test.skip()`, plus data factories, fixtures, and an implementation checklist that lists, per test, the tasks required to make it pass and the command to run it. The developer un-skips one test, confirms it fails, then makes it pass. The red phase is the point: a test that has never failed has proven nothing.\n\n**4. Coverage, in** `automate`**.** Run after implementation. Workers generate API, E2E, backend, and mobile tests in parallel depending on the detected stack, and the aggregation step reports the totals broken down by priority. It also rolls up every deviation from an active integration mandate as `file:line: reason`, and writes `None` when there are none, because a reader cannot tell an empty section from a forgotten one.\n\n**5. Quality, in** `test-review`**.** Every finding must cite a registry row (`C1`, `H2`, `M4`), and the row carries the severity. Score starts at 100:\n\n```text\nStarting Score:          100\nCritical Violations:     -{count} × 10\nHigh Violations:         -{count} × 5\nMedium Violations:       -{count} × 2\nLow Violations:          -{count} × 1\nBonus (6 categories, each 0 or 5, max +30)\nFinal Score:             clamped to 0-100     Grade: A ≥90, B ≥80, C ≥70, D ≥60, else F\n```\n\nThe verdict is then derived from the findings, not written by the model:\n\n```javascript\nif (CRITICAL > 0) return 'Block'; // a test that cannot fail is not a suggestion\nif (HIGH > 0) return 'Request Changes';\nif (score < 70) return 'Request Changes'; // volume of MEDIUM/LOW can also fail the bar\nif (MEDIUM + LOW > 0) return 'Approve with Comments';\nreturn 'Approve';\n```\n\nThis is the part worth sitting with. A suite with one CRITICAL and three MEDIUM findings, earning two bonus categories, scores `100 - 10 - 6 + 10 = 94`, a grade A, and is still a **Block**. Score measures the suite. The verdict answers a different question: is there anything here that makes the suite lie? One committed `.skip` on the test that mattered, or one `expect(true).toBe(true)`, means green proves nothing, which is worse than an absent test because it buys false confidence.\n\nThat separation exists because it was measured. Two reviewers of the same four files scored 82 and 85, which is noise, and returned opposite verdicts, which is not. `--fail-on request-changes` acts on the verdict, so the gate was being decided by the unpinned half of the report. `DESIGN-CRITERIA-REGISTRY.md` records the whole investigation.\n\n**6. Gate, in** `trace` **Phase 2.** Requirements are mapped to tests with Given/When/Then, coverage is computed per priority, and the decision is deterministic:\n\n| Condition                                                       | Decision     |\n| --------------------------------------------------------------- | ------------ |\n| P0 coverage below 100%                                          | **FAIL**     |\n| Overall coverage below 80%                                      | **FAIL**     |\n| P1 coverage below 80%                                           | **FAIL**     |\n| P0 at 100%, overall ≥ 80%, P1 ≥ 90%                             | **PASS**     |\n| P0 at 100%, overall ≥ 80%, P1 between 80% and 89%               | **CONCERNS** |\n| Stakeholder-approved waiver with the complete approval contract | **WAIVED**   |\n\nOur epic finishes at P0 100%, P1 87%, overall 84%, so it gates at CONCERNS with the residual risk named rather than passing quietly. Two overlays can lower that result further and can never raise it: a requirement resting only on recorded live verification caps at CONCERNS, and a synthetic coverage oracle below high confidence does the same. WAIVED is never derived; a human sets it, and the artifact demands an approver, approval date, reason, expiry, monitoring plan, remediation owner, and fix target. The authoritative contract is in the [Traceability template](./src/workflows/testarch/froid-testarch-trace/trace-template.md#waiver-details).\n\nIf the run is not gate-eligible at all, because evidence collection was waived, restricted, inaccessible, or deferred, TEA emits no decision rather than computing one on partial evidence.\n\n## Architecture & Flow\n\nFroid is a small **agent + workflow engine**. There is no external orchestrator; everything runs inside the LLM context window through structured instructions. TEA adds two pieces that run outside it: a Node hook that intercepts writes in your project, and a CLI that drives the review workflow headlessly in CI.\n\n### Building Blocks\n\n| File / Scope                                               | What it does                                                                                                 | When it loads                                                               |\n| ---------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------ | --------------------------------------------------------------------------- |\n| `src/agents/froid-tea/SKILL.md`                            | Murat's activation sequence and critical actions; renders the `{agent.menu}` placeholder                     | First, activates the TEA agent                                              |\n| `src/agents/froid-tea/customize.toml`                      | Agent customization surface: `[[agent.menu]]` items (code to skill), persona fields, persistent facts, hooks | During agent activation                                                     |\n| `src/workflows/testarch/<workflow>/SKILL.md`               | Workflow entrypoint: resolves workflow customization, picks mode, routes to the first step                   | When a TEA workflow is invoked                                              |\n| `src/workflows/testarch/<workflow>/customize.toml`         | Workflow customization surface: activation hooks, persistent facts, optional `on_complete` behavior          | During workflow activation                                                  |\n| `src/workflows/testarch/<workflow>/workflow.yaml`          | Machine-readable metadata: config bindings, run variables, tool hints, output paths                          | Installer, tooling, and workflow metadata lookups                           |\n| `instructions.md`                                          | Workflow-specific summary and operator notes                                                                 | On demand                                                                   |\n| `steps-c/*.md`                                             | **Create** steps: primary execution, 5 to 12 files per workflow, 75 across the module                        | One at a time (just-in-time)                                                |\n| `steps-c/step-NNx-subagent-*.md`                           | **Worker** steps: one isolated dimension each, dispatched in parallel                                        | When an orchestrator step delegates                                         |\n| `steps-e/*.md`                                             | **Edit** steps: always 2 files, assess target then apply edit                                                | One at a time                                                               |\n| `steps-v/*.md`                                             | **Validate** steps: always 1 file, evaluate against the checklist                                            | On demand                                                                   |\n| `checklist.md`                                             | Validation criteria: what \"done\" looks like for this workflow                                                | Read by steps-v                                                             |\n| `*-template.md`                                            | Output skeleton with `{PLACEHOLDER}` vars, filled in by steps to produce the artifact                        | Read by steps-c when generating output                                      |\n| `froid-testarch-test-review/steps-c/criteria-registry.md`  | The 35 scoreable rows, each with a fixed severity and gate class. Severity is read here                      | Read by every review worker before it scores anything                       |\n| `froid-testarch-framework/resources/hooks/tea-enforce.cjs` | Project-level guardrail for mechanically decidable test-quality violations; framework Create scaffolds it    | Before writes, after writes or shell commands, and when an agent turn stops |\n| `resources/tea-index.csv`                                  | Knowledge fragment index: id, name, description, tags, tier, path. 59 rows                                   | Read before recommendations and by knowledge-loading steps                  |\n| `resources/knowledge/*.md`                                 | 59 reusable fragments: standards, patterns, API references, integration mandates                             | Selectively read into context by tier and config flags                      |\n\nNine copies of the knowledge base exist on purpose: the agent carries one, and so does each of the eight workflows that consult it. Every copy is byte-identical. A workflow skill has to stay self-contained so it can be installed, copied, or invoked without reaching across skill boundaries, so when knowledge changes, propagate the update into the affected workflow resource directories rather than replacing them with a central runtime path. `froid-teach-me-testing` is the exception; it carries a curated pointer file at `data/tea-resources-index.yaml` instead of the fragments themselves.\n\n```mermaid\nflowchart TB\n  U[User] --> A[Agent activation<br/>persona + config + menu]\n  A --> W[Workflow entry: SKILL.md<br/>mode: Create / Resume / Validate / Edit]\n  W --> S[Step files<br/>steps-c / steps-e / steps-v]\n  S --> K[Knowledge fragments<br/>tea-index.csv to knowledge/*.md]\n  S --> T[Templates & checklists]\n  S --> P[Orchestrator step]\n  P --> X[Isolated workers<br/>one dimension each]\n  X --> G[Aggregation step<br/>scored against criteria-registry.md]\n  S --> O[Outputs: plans, tests, reports]\n  G --> O\n  O --> V[Validation: steps-v + checklist.md]\n  O --> C[Checkpoint frontmatter<br/>resume where it stopped]\n```\n\n### How It Works at Runtime\n\n**1. Activation.** `/froid-tea` or `$froid-tea` loads the agent skill. It resolves its customization block across base, team, and user layers, adopts the persona, loads persistent facts and `_froid/tea/config.yaml`, greets you, and renders `{agent.menu}` as a numbered table. Naming an intent in your first message (\"let's design tests for this epic\") skips the menu and dispatches directly.\n\n**2. Workflow entry.** Direct workflow commands use the installed skill name, such as `/froid-testarch-automate` or `$froid-testarch-automate`, depending on the host's invocation syntax. `TA` is the equivalent agent-menu code, available only once TEA is active. Either way, the workflow's `SKILL.md` resolves its own `[workflow]` customization block and asks which mode to run: Create, Resume, Validate, or Edit. Create and Resume both route into `steps-c/`; Validate into `steps-v/`; Edit into `steps-e/`. `test-review` alone supports `headless: true`, which skips the greeting and the menu and runs Create directly. That is how the CLI drives it in CI.\n\n**3. Steps.** Each step file declares its own wiring in YAML frontmatter: `outputFile`, `nextStepFile`, and where relevant `knowledgeIndex` and `resumeStepFile`. A step loads on its own, pulls only the fragments its mode and config flags call for, fills any `*-template.md` placeholders, writes its output, and names the next step. Nothing loads the whole workflow at once, and the step files say so in as many words: \"Do not load the next step until this step is complete.\"\n\n**4. Progress and resume.** Every create step appends itself to a checkpoint file's YAML frontmatter (`stepsCompleted`, `lastStep`, `lastSaved`), so an interrupted run resumes at the next incomplete step rather than from the top. `test-design` checkpoints additionally carry run identity: `runScope` and `runKey` are resolved before anything is saved, the file is named `test-design-progress-{run_key}.md`, and Resume refuses to continue a checkpoint whose `runKey` belongs to a different run. Interrupting a system-level run and starting an epic-level one no longer clobbers the first. `framework` and `ci` scaffold once per project, so a single fixed checkpoint is the right shape there and they keep one.\n\n**5. Validation.** `steps-v/` scores the finished output against `checklist.md`.\n\nSee [Step-File Architecture](./docs/explanation/step-file-architecture.md) for the loading model, worker isolation, and the per-workflow step patterns.\n\n### Parallel Workers and Execution Modes\n\nFive workflows split their heaviest step across isolated workers. An orchestrator step does no work of its own; it resolves the mode, dispatches, and hands off to an aggregation step.\n\n| Workflow      | Workers                                                        |\n| ------------- | -------------------------------------------------------------- |\n| `test-review` | determinism, isolation, maintainability, performance           |\n| `nfr`         | security, performance, reliability, maintainability            |\n| `automate`    | API, E2E, backend, mobile. Stack-gated, so 1 to 3 of the 4 run |\n| `atdd`        | failing API tests, failing E2E tests                           |\n| `test-design` | system-level mode may generate its two documents in parallel   |\n\n`tea_execution_mode` decides how they run: `auto`, `agent-team`, `subagent`, or `sequential`. With `tea_capability_probe` at its default of `true`, `auto` probes the runtime and prefers agent-team, then subagent, then sequential, which keeps behavior portable across supported agent runtimes. With probing off, TEA honors the configured mode strictly and fails with an explicit error rather than falling back silently. Mode changes orchestration only. The output schema, the validation rules, and the aggregation contract are identical in every mode.\n\nWorkers exchange nothing directly. Each writes a JSON file under `/tmp` keyed by a shared run timestamp, and the aggregation step asserts every expected file exists before it scores anything. Isolation is what makes a parallel review honest. The shared criteria registry is what stops two isolated workers from disagreeing about what a finding is worth: every worker loads it, and none of them choose a severity.\n\n### How Knowledge Gets Selected\n\n`tea-index.csv` classifies all 59 fragments into three tiers: **core** (24, always loaded), **extended** (19, loaded when deeper analysis is called for), and **specialized** (16, loaded only when the case matches, such as contract testing on a real consumer-provider boundary). Steps name the fragments they need, and they name their exclusions just as explicitly. A Maestro run is told not to load the browser fragments, because a device flow has no DOM and no request interceptor, and loading them invites browser patterns into a device flow.\n\nOver-loading is treated as a real defect, not a harmless cost. `npm run eval:fragment-selection` measures both directions: recall of the fragments a step requires, and the rate at which a run pulls one the step excludes by name.\n\n### The Three Control Points\n\nA test rule can be enforced at three moments, and most tools occupy one of them. TEA occupies all three. The release gate is the fourth row below because it consumes what the other three produce, rather than being a fourth place to enforce a rule.\n\n| Point          | When                   | Mechanism                                              | What it closes                                                               |\n| -------------- | ---------------------- | ------------------------------------------------------ | ---------------------------------------------------------------------------- |\n| **Generation** | before the test exists | knowledge fragments and integration mandates           | the model improvising a pattern TEA already has a standard for               |\n| **Write**      | as the file lands      | `tea-enforce.cjs` on PreToolUse, PostToolUse, and Stop | a `.only`, a hard wait, or a tautological assertion getting committed at all |\n| **Review**     | after the fact         | `test-review` scored against the criteria registry     | severity drifting with whichever model happened to run the review            |\n| **Gate**       | at release             | `trace` Phase 2, PASS / CONCERNS / FAIL / WAIVED       | shipping on evidence nobody checked was sufficient                           |\n\nThe hook has separate installation and runtime lifecycles. The `froid-testarch-framework` Create path installs it during its documentation and scripts step. Resume reaches the same step when installation is still incomplete. Framework Validate and Edit do not install it, and no other TEA workflow calls it.\n\nOnce installed, the hook is project-scoped rather than workflow-scoped. It runs on matching tool events across TEA workflows, other agents, and ordinary coding sessions without requiring Murat or a TEA workflow to be active.\n\nThe three passes cover different user-visible moments. `--pre` checks content before a direct file write reaches disk and rejects a blocking violation with a fix. `--post` re-reads the affected file after direct file writes or shell commands, then reports violations that only become visible in the complete file. `--stop` scans recently modified test files when the agent turn finishes, including outputs a code generator did not name in its command. All three passes are limited to the test and Pact configuration globs written for the detected stack in `.tea/enforce-config.json`.\n\nThis closes the gap between advisory generation guidance and a later `test-review`. It blocks seven mechanically decidable Absolute rules and warns on one: focused tests, tautological assertions, hard waits, oversized test files, Maestro flows that cannot fail, two Pact parallelism rules, and undocumented disabled tests as the warning. Rules that require semantic judgment stay in `test-review`. The hook fails open on its own errors so a broken guardrail cannot lock the agent out of writing. Agent platforms without a write-time hook API skip installation and rely on `test-review` for enforcement.\n\n**How workflows become commands.** `npx froid-plane install` copies each TEA skill into the host runtime's skill directory under its own name. Invoking that name loads the skill, and the step-file process takes over. The skill name is identical on every platform the Froid installer supports.\n\n## Install\n\n```bash\nnpx froid-plane install\n# Select: Test Architect (TEA)\n```\n\n**Note:** TEA is automatically added to party mode after installation. Use `/party` to collaborate with TEA alongside other Froid agents.\n\n### Invocation Syntax\n\n| Host convention       | Example                                    |\n| --------------------- | ------------------------------------------ |\n| Slash command         | `/froid-testarch-automate`                 |\n| Dollar-prefixed skill | `$froid-tea` or `$froid-testarch-automate` |\n\n## Quickstart\n\n1. Install TEA (above)\n2. Load the TEA menu with `/froid-tea` or `$froid-tea` if you want a conversational entrypoint.\n3. Run one of the core workflows:\n\n- `TD` / `/froid-testarch-test-design` / `$froid-testarch-test-design` — test design, risk assessment, and NFR planning\n- `AT` / `/froid-testarch-atdd` / `$froid-testarch-atdd` — failing acceptance tests first (TDD red phase)\n- `TA` / `/froid-testarch-automate` / `$froid-testarch-automate` — expand automation coverage\n\n1. Or use in party mode: `/party` to include TEA with other agents\n\n## Engagement Models\n\n- **No TEA**: Use your existing testing approach\n- **TEA Solo**: Standalone use on non-Froid projects\n- **TEA Lite**: Start with `automate` only for fast onboarding\n- **Integrated (Froid Plane / Enterprise)**: Use TEA in Phases 3–4 and release gates\n\n## Workflows\n\n| Trigger | Slash Command                 | Dollar Skill                  | Purpose                                                                     |\n| ------- | ----------------------------- | ----------------------------- | --------------------------------------------------------------------------- |\n| TMT     | `/froid-teach-me-testing`     | `$froid-teach-me-testing`     | Teach Me Testing (TEA Academy)                                              |\n| TD      | `/froid-testarch-test-design` | `$froid-testarch-test-design` | System-level or epic-level test design and NFR planning                     |\n| TF      | `/froid-testarch-framework`   | `$froid-testarch-framework`   | Scaffold test framework (frontend, backend, fullstack, or mobile)           |\n| CI      | `/froid-testarch-ci`          | `$froid-testarch-ci`          | Set up CI/CD quality pipeline (multi-platform)                              |\n| AT      | `/froid-testarch-atdd`        | `$froid-testarch-atdd`        | Generate failing acceptance tests + checklist                               |\n| TA      | `/froid-testarch-automate`    | `$froid-testarch-automate`    | Expand test automation coverage                                             |\n| RV      | `/froid-testarch-test-review` | `$froid-testarch-test-review` | Review test quality and score                                               |\n| NR      | `/froid-testarch-nfr`         | `$froid-testarch-nfr`         | Audit implemented NFR evidence                                              |\n| TR      | `/froid-testarch-trace`       | `$froid-testarch-trace`       | Trace requirements to tests + gate decision                                 |\n| GATE    | agent menu only               | agent menu only               | Route the release gate: test review, NFR evidence audit, then trace Phase 2 |\n\n`GATE` is a routing prompt on the agent menu, so it has no standalone command. Load the agent with `/froid-tea` or `$froid-tea` and pick it there.\n\n## The Release Gate\n\n`trace` Phase 2 produces the decision: PASS, CONCERNS, FAIL, or WAIVED. Two mechanics sit under that vocabulary and are easy to miss.\n\n**Live evidence is capped.** A requirement covered only by recorded live verification forces PASS down to CONCERNS, with a rationale naming the recorded source SHA. The overlay only ever lowers a PASS or annotates an existing CONCERNS. It can never lift a FAIL. Only a `pass` recorded against the commit under trace counts; `stale`, `unverifiable`, `contradicted`, `blocked`, and the rest are reported as blockers. The JSON contract is published at [Live Verification Results](./docs/reference/live-verification-results.md), so any runner can produce it. Trace reads that file and never runs anything itself.\n\n**Some runs are not gate-eligible at all.** A collection status of `waived`, `restricted`, `inaccessible`, or `deferred_shared` means no decision is emitted rather than a decision computed on partial evidence. A missing manifest resolves to `INACCESSIBLE`, not to 0% coverage.\n\n### `tea-test-review` in CI\n\nInstalling this package also installs a `tea-test-review` binary that runs the review workflow headlessly against a pull request diff.\n\n```bash\nnpx tea-test-review --base origin/main --min-score 80\n```\n\nIt scopes to changed tests (`--base`, `--files`), runs through an agent adapter with a pinned review model, isolates the filesystem, emits a JSON verdict, and separates its exit codes: `0` pass, `1` verdict failure, `2` environment or configuration failure, and `3` agent failure or an unparseable or untrusted report. For example, a missing credential exits `2`, while a runner crash after launch exits `3`.\n\nThe recommendation is derived from the findings rather than taken from the agent's prose. Any CRITICAL derives Block. Any HIGH, or a score under 70, derives Request Changes. The agent's own stated recommendation is preserved as `reportedRecommendation` when the two disagree. `--waive` exists for the exceptions and requires an expiry.\n\nA copy-paste workflow lives at `cli/examples/pr-test-review.yml`, and the full flag, exit-code, and security reference is at [tea-test-review CLI](./docs/reference/tea-test-review-cli.md).\n\n## Configuration\n\nTEA variables are defined in `src/module.yaml` and prompted during install. Ten are wired into workflows today; the last four are placeholders that nothing reads yet.\n\n- `test_artifacts` — base output folder for test artifacts\n- `tea_use_playwright_utils` — enable Playwright Utils integration (boolean, default true). When true **and the package is installed**, `@seontechnologies/playwright-utils` becomes the default implementation for everything it covers: generated Playwright tests use `interceptNetworkCall`, `apiRequest`, `recurse`, and `log` without being asked, and `test-review` flags a vanilla equivalent that carries no stated reason. See [Integrate Playwright Utils](https://cordya-ai.github.io/froid-plane-test-architecture-enterprise/how-to/customization/integrate-playwright-utils/)\n- `tea_use_pactjs_utils` — enable Pact.js Utils integration for contract testing (boolean, default true). It decides how Pact suites are written, not whether a project gets one: TEA still requires a real consumer-provider boundary before scaffolding a contract test. When on **and the package is installed**, generated Pact code uses `createProviderState`, `buildVerifierOptions`, and `createRequestFilter` rather than raw Pact boilerplate. A flag with no install generates the raw path and reports one recommendation rather than flagging every file\n- `tea_pact_mcp` — SmartBear MCP for PactFlow/Broker interaction: mcp, none (string, default mcp). Safe without a broker: every broker-dependent step degrades to provider source or an OpenAPI spec and reports that the broker was unreachable\n- `tea_browser_automation` — browser automation mode: auto, cli, mcp, none (string, default auto)\n- `tea_execution_mode` — how TEA orchestrates multi-step generation and evaluation: auto, subagent, agent-team, sequential (string, default auto)\n- `tea_capability_probe` — probe the runtime before selecting an execution mode (boolean, default true). With it off, TEA honors the configured mode strictly and fails loudly instead of falling back\n- `test_stack_type` — detected or configured stack type (auto, frontend, backend, fullstack, mobile). Mobile is checked before frontend, because a React Native project carries React in `package.json` and would otherwise misdetect as web\n- `ci_platform` — CI platform (auto, github-actions, gitlab-ci, jenkins, azure-devops, harness, circle-ci, other)\n- `test_framework` — detected or configured test framework (auto, Playwright, Cypress, Jest, Vitest, pytest, JUnit, Go test, dotnet test, RSpec, Maestro, other)\n- `risk_threshold` — risk cutoff for mandatory testing. Prompted at install, not yet read by any workflow\n- `test_design_output`, `test_review_output`, `trace_output` — subfolders under `test_artifacts`. Prompted at install, not yet read by any workflow\n\nFull option reference: [Configuration](./docs/reference/configuration.md).\n\n## Knowledge Base\n\nTEA relies on a curated testing knowledge base of 59 fragments, indexed by tier:\n\n- Index: `src/agents/froid-tea/resources/tea-index.csv`\n- Fragments: `src/agents/froid-tea/resources/knowledge/`\n- Tiers: 24 core, 19 extended, 16 specialized\n\nWorkflows load only the fragments required for the current task, and the required set is named in the step file rather than inferred from index tags. See [Knowledge Base](./docs/reference/knowledge-base.md).\n\n## Repository Layout\n\n```text\nsrc/                     # the shipped module\n├── module.yaml          # install-time variables and post-install notes\n├── module-help.csv      # workflow catalog: menu codes, phases, ordering\n├── agents/froid-tea/     # SKILL.md, customize.toml, resources/{tea-index.csv, knowledge/}\n└── workflows/testarch/  # nine self-contained workflow skills\n    ├── froid-teach-me-testing/\n    ├── froid-testarch-atdd/\n    ├── froid-testarch-automate/\n    ├── froid-testarch-ci/\n    ├── froid-testarch-framework/        # resources/hooks/tea-enforce.cjs lives here\n    ├── froid-testarch-nfr/\n    ├── froid-testarch-test-design/\n    ├── froid-testarch-test-review/      # steps-c/criteria-registry.md lives here\n    └── froid-testarch-trace/\n\ncli/                     # tea-test-review: the headless CI gate\ndocs/                    # source of truth for the docs site\nwebsite/                 # Astro + Starlight, consumes docs/ through a symlink\ntools/                   # validators, doc build, changelog stamping\ntest/                    # quality gate suites and the two eval harnesses\n```\n\n## How TEA Keeps Itself Honest\n\nTEA has deterministic checks and live evals. These cover specific risks. They are not end-to-end evals of every skill.\n\n### Current Eval Coverage\n\nThe eight suites under `test/evals/` measure one decision inside each knowledge-bearing workflow: whether the agent selects the required knowledge fragments and avoids fragments the workflow excludes. They do not execute the complete workflow or grade its final artifact.\n\n`test-review` has an additional behavioral eval. It runs the complete review against files containing nine planted defects plus one clean file, then scores recall, precision, score variance, and verdict stability.\n\n| Skill                        | Fragment-selection cases | Full behavioral eval                                       |\n| ---------------------------- | ------------------------ | ---------------------------------------------------------- |\n| `froid-tea`                  | N/A                      | None                                                       |\n| `froid-teach-me-testing`     | N/A                      | None; this skill has no workflow knowledge index           |\n| `froid-testarch-atdd`        | 3                        | None                                                       |\n| `froid-testarch-automate`    | 5                        | None                                                       |\n| `froid-testarch-ci`          | 2                        | None                                                       |\n| `froid-testarch-framework`   | 3                        | None                                                       |\n| `froid-testarch-nfr`         | 2                        | None                                                       |\n| `froid-testarch-test-design` | 5                        | None                                                       |\n| `froid-testarch-test-review` | 2                        | Yes; three files, nine planted defects, and one clean file |\n| `froid-testarch-trace`       | 2                        | None                                                       |\n\nA passing fragment-selection eval means the workflow loaded the right knowledge. It makes no claim about the quality of the workflow's final output. Full behavioral evals for the other skills remain a coverage gap. The source-controlled [Eval Quality and Behavioral Coverage Roadmap](./docs/explanation/eval-quality-roadmap.md) records the per-skill contracts, runner work, CI plan, and intended boundary with the upcoming standalone `eval-quality` project.\n\n### Deterministic Checks\n\n`npm test` chains thirteen deterministic checks, including three that keep the rules, guidance, hook, and eval data aligned:\n\n- `test:criteria-fragments` fails when a registry row is neither mapped to a knowledge fragment nor declared a known gap. A rule the reviewer scores but no fragment teaches is a rule TEA punishes without ever having explained it. All 35 rows are currently mapped across 48 anchors. Because the declared-gap list is empty, the validator feeds itself a synthetic unmapped row on every run to prove that path still works.\n- `test:enforce-hook` fails when a new Absolute registry row appears in neither the hook's enforced list nor its deferred list. This prevents a rule from being added without an explicit write-time enforcement decision.\n- `test:eval-data` checks that all 24 fragment-selection cases are structurally usable: their workflow context files exist, every expected fragment exists and is indexed for that workflow, and the required and forbidden sets do not overlap. The expected sets come from the workflow step files. This check does not ask an agent to select anything.\n\nThese checks produce the same answer from the same repository state. They need no agent credential, network call, or model budget. `test:eval-data` runs through `npm test`, the local pre-commit hook, pull-request quality checks, and the publish workflow.\n\n### Start Here\n\nYou do not start an interactive agent session. A live eval launches the selected agent CLI as a headless subprocess, sends it each prompt, waits for the result, and scores the result.\n\nThe normal path is one command. It runs fragment selection across all eight covered workflow skills, then runs the behavioral `test-review` eval:\n\n```bash\nnpm run eval:all -- --agent codex\n```\n\nUse `claude` or `agy` instead, or run all three built-in adapters:\n\n```bash\nnpm run eval:all -- --agent claude\nnpm run eval:all -- --agent agy\nnpm run eval:all -- --agent agy --agent claude --agent codex\n```\n\n`eval:all` uses two repetitions per fragment-selection case and three repetitions for `test-review`. One runner makes 51 agent calls: 48 fragment selections plus 3 reviews. All three built-in runners make 153 calls.\n\nCheck the data, executable, login, fixtures, and expected results without making a model call:\n\n```bash\nnpm run eval:all -- --agent codex --preflight-only\nnpm run eval:all -- --agent claude --preflight-only\nnpm run eval:all -- --agent agy --preflight-only\n```\n\nOutput ending with `nothing measured` is expected in preflight mode. It means the static eval data is valid and the selected executable passed the available readiness checks. Some runners cannot expose session authentication to this probe, so a preflight pass does not guarantee that the later live call will authenticate. The flag intentionally exits before launching the agent.\n\n### A La Carte Live Evals\n\nUse the focused commands when debugging one metric or skill. A one-call review smoke test is:\n\n```bash\n# One review. Recall and precision are measured; variance and stability are not.\nnpm run eval:test-review -- --agent codex --runs 1\n\n# Complete eval with one runner.\nnpm run eval:test-review -- --agent codex\nnpm run eval:test-review -- --agent claude\n\n# Complete eval with all three built-in runners. This makes nine review calls.\nnpm run eval:test-review -- --agent agy --agent claude --agent codex\n```\n\n### Run Fragment Selection by Skill\n\nEach command below runs one repetition. Use `--runs 2` for the complete stability measurement.\n\n| Skill                 | Copy-paste command                                                                                |\n| --------------------- | ------------------------------------------------------------------------------------------------- |\n| `atdd`                | `npm run eval:fragment-selection -- --agent codex --workflow froid-testarch-atdd --runs 1`        |\n| `automate`            | `npm run eval:fragment-selection -- --agent codex --workflow froid-testarch-automate --runs 1`    |\n| `ci`                  | `npm run eval:fragment-selection -- --agent codex --workflow froid-testarch-ci --runs 1`          |\n| `framework`           | `npm run eval:fragment-selection -- --agent codex --workflow froid-testarch-framework --runs 1`   |\n| `nfr`                 | `npm run eval:fragment-selection -- --agent codex --workflow froid-testarch-nfr --runs 1`         |\n| `test-design`         | `npm run eval:fragment-selection -- --agent codex --workflow froid-testarch-test-design --runs 1` |\n| `test-review` routing | `npm run eval:fragment-selection -- --agent codex --workflow froid-testarch-test-review --runs 1` |\n| `trace`               | `npm run eval:fragment-selection -- --agent codex --workflow froid-testarch-trace --runs 1`       |\n\nRun every suite with one or all built-in runners:\n\n```bash\n# 24 cases run twice: 48 calls.\nnpm run eval:fragment-selection -- --agent codex\nnpm run eval:fragment-selection -- --agent claude\n\n# All three runners: 144 calls.\nnpm run eval:fragment-selection -- --agent agy --agent claude --agent codex\n```\n\n### Antigravity, Claude, Codex, and Custom Agent CLIs\n\nThe built-in adapters are `claude`, `codex`, and `agy` (Antigravity CLI). Run live evals with any built-in adapter:\n\n```bash\nnpm run eval:all -- --agent agy\nnpm run eval:all -- --agent claude\nnpm run eval:all -- --agent codex\n```\n\nAny other headless CLI can use `--agent custom`. The runner must:\n\n1. Read the complete prompt from standard input.\n2. Run non-interactively in the repository working directory.\n3. Print its final response to standard output. The review eval must also allow the agent to write the report path named in the prompt.\n4. Exit with a nonzero status when the agent call fails.\n\n[Gemini CLI headless mode](https://geminicli.com/docs/cli/headless/) accepts standard input alongside a `-p` prompt. Once Gemini is installed and authenticated, run every live eval with:\n\n```bash\nnpm run eval:all -- \\\n  --agent custom \\\n  --agent-cmd gemini \\\n  --agent-arg -p \\\n  --agent-arg \"Follow the complete instructions from standard input.\" \\\n  --agent-arg --output-format \\\n  --agent-arg text \\\n  --agent-arg --approval-mode \\\n  --agent-arg yolo \\\n  --agent-arg --skip-trust \\\n  --env-pass GEMINI_API_KEY \\\n  --env-pass GOOGLE_API_KEY\n```\n\nThis uses `yolo` because the review eval must write its report. To run fragment selection alone with a read-only policy:\n\n```bash\nnpm run eval:fragment-selection -- \\\n  --agent custom \\\n  --agent-cmd gemini \\\n  --agent-arg -p \\\n  --agent-arg \"Follow the complete instructions from standard input.\" \\\n  --agent-arg --output-format \\\n  --agent-arg text \\\n  --agent-arg --approval-mode \\\n  --agent-arg plan \\\n  --agent-arg --skip-trust \\\n  --env-pass GEMINI_API_KEY \\\n  --env-pass GOOGLE_API_KEY\n```\n\n`--env-pass` is required only for credentials stored in environment variables. Stored CLI logins use the home directory that the harness already passes through. Model selection for a custom runner is also explicit, using repeated `--agent-arg` values for that CLI's model flag and value.\n\n### What Passes\n\n| Eval                              | Passing result                                                                                                                                        | Default volume               |\n| --------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------- |\n| `npm run eval:all -- --agent ...` | Both live evals below pass for the selected runner                                                                                                    | 48 selections plus 3 reviews |\n| Fragment selection                | At least 90% required-fragment recall, at most 10% forbidden-fragment selection, and stable choices across repeated cases                             | 24 cases twice: 48 calls     |\n| Test review                       | At least 70% overall recall, 100% CRITICAL recall, at least 80% clean-file precision, score standard deviation no higher than 3, and a stable verdict | Three complete reviews       |\n| `npm run test:eval-data`          | Every case references valid workflow files and indexed fragments; required and forbidden sets do not overlap                                          | No agent calls               |\n\n### CI Usage\n\nRun the deterministic check on every pull request:\n\n```bash\nnpm ci\nnpm run test:eval-data\n```\n\nRun live evals in a scheduled or manually triggered CI job after installing and authenticating the selected agent CLI:\n\n```bash\nnpm ci\nnpm run eval:all -- --agent codex\n```\n\nThe eval harnesses use CI-compatible exit codes: `0` means every threshold passed, `1` means a measured result missed a threshold, and `2` means the environment could not run the eval. Live jobs consume model quota and can vary as models change, so keep their result separate from the deterministic pull-request gate until the team chooses to make model quality a required check.\n\nTEA applies the same evidence rule to its documentation. Unproven explanations are labeled as hypotheses, and workarounds are labeled as countermeasures. `DESIGN-CRITERIA-REGISTRY.md` records the investigations behind the review rules and scoring decisions.\n\n## Extending TEA\n\nCustom workflows are still compatible with TEA, but they are no longer implicitly absorbed into TEA core. The supported path is:\n\n1. Package the workflow as custom content or a custom module.\n2. Attach it to `froid-tea` using the agent customization flow.\n3. Reinstall/update FROID so the new menu item and workflow are registered.\n\nSee [Extend TEA with Custom Workflows](docs/how-to/customization/extend-tea-with-custom-workflows.md) and the FROID customization guide at [Froid Plane/docs/how-to/customize-froid.md](https://github.com/cordya-ai/froid-plane/blob/main/docs/how-to/customize-froid.md).\n\n## Contributing\n\nSee `CONTRIBUTING.md` for guidelines.\n\n---\n\n**📦 Release Guide (for Maintainers)**\n\n## Publishing TEA to NPM\n\nTEA uses an automated publish workflow modeled after the main `Froid Plane` repo. It supports:\n\n- `next` prereleases published automatically from `main`\n- manual stable releases on the `latest` dist-tag\n- trusted npm publishing (no `NPM_TOKEN` secret)\n- metadata sync for `package.json`, `package-lock.json`, and `.claude-plugin/marketplace.json`\n\n### Prerequisites (One-Time Setup)\n\n1. **npm Trusted Publishing:**\n\n- In npm package settings for `froid-plane-test-architecture-enterprise`, configure Trusted Publishers for this GitHub repository\n- Allow publishes from the `cordya-ai/froid-plane-test-architecture-enterprise` repo and the `.github/workflows/publish.yaml` workflow\n- GitHub Actions must be able to request an OIDC token (`id-token: write`), which the workflow already does\n\n1. **GitHub App Secrets for Stable Releases:**\n\n- Add `RELEASE_APP_ID`\n- Add `RELEASE_APP_PRIVATE_KEY`\n- Install the corresponding GitHub App on this repository with contents write access\n- If `main` is protected, ensure the app is allowed to push the release commit and tag\n- These are used only for manual stable releases so the workflow can push the version bump commit and tag back to `main`\n\n1. **Verify Package Configuration:**\n\n```bash\n # Check package.json settings\n cat package.json | grep -A 3 \"publishConfig\"\n # Should show: \"access\": \"public\"\n if grep -Eq '\"private\"[[:space:]]*:[[:space:]]*true' package.json; then\n   echo '❌ package.json must not set \"private\": true'\n else\n   echo '✅ package.json is publishable (\"private\": true not present)'\n fi\n```\n\n### Release Process\n\n#### Option 1: Using npm Scripts (Recommended)\n\nFrom your local terminal after merging to `main`:\n\n```bash\n# Publish the next prerelease from current main\nnpm run release:next\n\n# Publish a stable patch release\nnpm run release:patch\n\n# Publish a stable minor release\nnpm run release:minor\n\n# Publish a stable major release\nnpm run release:major\n```\n\n#### Option 2: Manual Workflow Trigger\n\n1. Go to **Actions** tab in GitHub\n2. Click **\"Publish\"** workflow\n3. Click **\"Run workflow\"**\n4. Choose the branch to release, typically `main`\n5. Select channel:\n\n- `next` for a prerelease publish\n- `latest` for a stable release\n\n1. If using `latest`, choose the bump type (`patch`, `minor`, `major`)\n2. Click **\"Run workflow\"**\n\n### What Happens Automatically\n\nThe workflow performs these steps:\n\n1. ✅ **Validation**: Runs the full `npm test` chain: schema checks, install tests, knowledge checks, criteria-to-fragment traceability, enforce-hook coverage, eval data validation, release metadata, changelog, workflow descriptions, linting, markdown linting, and formatting. The CLI suite (`npm run test:cli`) runs as its own CI job because it takes over twelve minutes\n2. ✅ **Version Bump**:\n\n- `next`: derives the next prerelease version and publishes it with dist-tag `next`\n- `latest`: bumps the stable version (`patch`, `minor`, or `major`)\n\n1. ✅ **Metadata Sync**: Updates `.claude-plugin/marketplace.json` to match the package version before publishing\n2. ✅ **Publish**: Publishes to npm with provenance enabled\n\n- `next` → `npm publish --tag next --provenance`\n- `latest` → `npm publish --tag latest --provenance`\n\n1. ✅ **Stable Release Finalization**: For `latest`, creates a version bump commit, tags it, pushes it to `main`, and creates a GitHub Release\n\n### Channel Strategy\n\n- `next`: prerelease channel for the newest merged changes\n- `latest`: stable channel for intentional releases\n- `patch`: bug fixes, no breaking changes\n- `minor`: new features, backwards compatible\n- `major`: breaking changes\n\n**Recommended Release Path:**\n\n1. Merge releasable work to `main`\n2. Let `next` publish for early validation\n3. When ready, cut a stable `latest` release via `patch`, `minor`, or `major`\n\n### Verify Publication\n\n**Check NPM:**\n\n```bash\nnpm view froid-plane-test-architecture-enterprise\nnpm view froid-plane-test-architecture-enterprise dist-tags\n```\n\n**Install TEA:**\n\n```bash\nnpx froid-plane install\n# Select \"Test Architect (TEA)\"\n```\n\n**Test Workflows:** type these in the assistant chat, not in a shell.\n\n```text\n/froid-tea                     # load the agent persona and menu\n/froid-testarch-test-design    # run a workflow directly\n```\n\nHosts that use dollar-prefixed skills use `$` in place of `/`.\n\n### Rollback a Release (if needed)\n\nIf you need to unpublish a version:\n\n```bash\n# Unpublish specific version (within 72 hours)\nnpm unpublish froid-plane-test-architecture-enterprise@1.13.2-next.0\n\n# Deprecate version (preferred for older releases)\nnpm deprecate froid-plane-test-architecture-enterprise@1.13.2-next.0 \"Use version X.Y.Z instead\"\n```\n\n### Troubleshooting\n\n**Trusted publishing failed:**\n\n- Verify npm Trusted Publishing is configured for this repository and workflow\n- Verify the workflow has `id-token: write`\n- Confirm the publish is running from the canonical repository, not a fork\n\n**\"Package already exists\":**\n\n- Check if package name is already taken on NPM\n- Update `name` in `package.json` if needed\n\n**\"Version push failed\":**\n\n- Verify `RELEASE_APP_ID` and `RELEASE_APP_PRIVATE_KEY` are configured\n- Verify the GitHub App is installed on this repository with contents write access\n- If branch protection is enabled on `main`, verify the app is allowed to push the release commit and tag\n\n**\"Tests failed\":**\n\n- Fix failing tests before release\n- Run `npm test` locally to verify\n\n**\"Git push failed (protected branch)\":**\n\n- This is not expected once the release GitHub App is configured correctly\n- Verify branch protection allows the app to push the release commit and tag\n- If needed, create the GitHub Release manually after resolving the app permissions\n\n### Release Checklist\n\nBefore releasing:\n\n- [ ] All tests passing: `npm test`\n- [ ] Documentation up to date\n- [ ] CHANGELOG.md updated\n- [ ] No uncommitted changes\n- [ ] On `main` branch\n- [ ] npm Trusted Publishing configured\n- [ ] `RELEASE_APP_ID` and `RELEASE_APP_PRIVATE_KEY` configured\n- [ ] Package name available on NPM\n\nAfter releasing:\n\n- [ ] Verify NPM publication: `npm view froid-plane-test-architecture-enterprise`\n- [ ] Test installation: `npx froid-plane install`\n- [ ] Verify workflows work\n- [ ] Check GitHub Release created\n- [ ] Monitor for issues\n\n---\n\n## Community\n\n- [Discord](https://discord.gg/gk8jAdXWmj) — Get help, share ideas, collaborate\n- [YouTube](https://youtube.com/@FroidCode) — Tutorials, master class, and more\n- [X / Twitter](https://x.com/FroidCode)\n- [Website](https://froidcode.com)\n\n## Support Froid\n\nFroid is free for everyone and always will be. Star this repo, [buy me a coffee](https://buymeacoffee.com/froid), or email [contact@froidcode.com](mailto:contact@froidcode.com) for corporate sponsorship.\n\n## License\n\nSee `LICENSE`.\n","readmeFilename":"README.md","_rev":"1-313aabb30b9ee0f27b241e55295d707b"}