{"_id":"@cordya-ai/froid-eval-quality","name":"@cordya-ai/froid-eval-quality","dist-tags":{"latest":"0.1.2"},"versions":{"0.1.2":{"name":"@cordya-ai/froid-eval-quality","version":"0.1.2","description":"Compile disciplined Behavioral Evaluation Contracts and score their ability to catch known defects.","author":{"name":"Murat Ozcan"},"license":"Apache-2.0","private":false,"main":"dist/index.js","module":"dist/index.js","types":"dist/index.d.ts","bin":{"eval-quality":"dist/cli/main.js"},"type":"module","keywords":["eval","evals","evaluation","agent","agent-eval","llm","behavioral-contract","oracle","contract-strength","testing","ai"],"exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"},"./adapters":{"types":"./dist/adapters/index.d.ts","default":"./dist/adapters/index.js"},"./conformance":{"types":"./dist/testing/index.d.ts","default":"./dist/testing/index.js"},"./schemas/*":"./schemas/*","./corpus/*":"./corpus/*","./package.json":"./package.json"},"homepage":"https://github.com/cordya-ai/froid-eval-quality#readme","repository":{"type":"git","url":"git+https://github.com/cordya-ai/froid-eval-quality.git"},"bugs":{"url":"https://github.com/cordya-ai/froid-eval-quality/issues"},"publishConfig":{"access":"public"},"engines":{"node":">=22.20.0"},"scripts":{"hooks:install":"husky","build":"tsc -p tsconfig-build.json","clean":"rm -rf dist","typecheck":"tsc --noEmit -p tsconfig.json","lint":"biome check .","lint:fix":"biome check --write .","format":"biome format --write .","check:docs":"node scripts/check-docs.mjs","check:doc-invocations":"node scripts/check-doc-invocations.mjs","lint:spine":"python3 scripts/spine-lint/lint_spine.py --registry-ad 5 --workspace-root . --fail-on high","test:spine-lint":"uv run --with pytest pytest scripts/spine-lint/tests -q","build:shareable":"node scripts/build-shareable.mjs","test":"vitest run","test:coverage":"vitest run --coverage","test:conformance":"npm run build && vitest run tests/adapters tests/testing tests/conformance","test:watch":"vitest","check:vectors":"python3 tests/fixtures/derive_vectors.py --check","generate:schemas":"node scripts/generate-schemas.ts","check:schemas":"node scripts/check-schemas.ts","check:ad5-registry":"node scripts/check-ad5-registry.ts","check:ad28-registry":"node scripts/check-ad28-registry.ts","generate:ad31-table":"node scripts/generate-ad31-table.ts","check:ad31-table":"node scripts/check-ad31-table.ts","check:layers":"node scripts/check-dependency-direction.ts","check:lineage":"node scripts/check-lineage-ownership.ts","check:boundary":"node scripts/check-package-boundary.ts","generate:dev-corpus":"node scripts/generate-dev-corpus.ts","check:corpus":"node scripts/check-dev-corpus.ts","check:website-deps":"node scripts/audit-lockfile-age.mjs --lockfile website/package-lock.json && node scripts/check-licenses.mjs --lockfile website/package-lock.json","check:shareable":"node scripts/check-shareable.mjs","bench:digest":"node scripts/bench-digest.ts","docs:dev":"npm --prefix website run dev","docs:build":"node tools/build-docs.mjs","docs:preview":"npm --prefix website run preview","docs:validate-links":"node tools/validate-doc-links.js","release:prepare":"node scripts/release-prepare.mjs","release:publish":"gh workflow run publish.yml --ref main","validate":"npm run build && npm run typecheck && npm run lint && npm run check:docs && npm run check:doc-invocations && npm run check:shareable && npm run lint:spine && npm run check:vectors && npm run check:schemas && npm run check:ad5-registry && npm run check:ad28-registry && npm run check:ad31-table && npm run check:layers && npm run check:lineage && npm run check:boundary && npm run check:corpus && npm run check:website-deps && npm run test:coverage","prepack":"npm run clean && npm run build","prepublishOnly":"node scripts/assert-publish-authorized.mjs"},"dependencies":{"zod":"4.4.3"},"devDependencies":{"@biomejs/biome":"2.5.10","@types/node":"22.20.1","@vitest/coverage-v8":"4.1.11","ajv":"8.20.0","husky":"9.1.7","lint-staged":"16.4.0","marked":"18.0.10","typescript":"7.0.2","vitest":"4.1.11"},"overrides":{"vite":"7.3.6"},"lint-staged":{"*.{ts,js,json}":["biome check --write --no-errors-on-unmatched"]},"_id":"@cordya-ai/froid-eval-quality@0.1.2","_nodeVersion":"26.7.0","_npmVersion":"11.19.0","dist":{"integrity":"sha512-gEAP/iUuozJjqy3XnrPgKRjHbuAqF7PFDgLc9WGpW0ROXfOZRzO0kgJ3ghGFE8aEE1tJ35D49aynk1kJjTFYdQ==","shasum":"6fad418227ec9c4e1eedc8b2c31e9ab520b99ece","tarball":"https://registry.npmjs.org/@cordya-ai/froid-eval-quality/-/froid-eval-quality-0.1.2.tgz","fileCount":230,"unpackedSize":1321722,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQDCGBEcOtiGaKB5WjR+dcM2vW9ZfXnNJLcar0JsNpXp1AIgRuQc2IBOyxwkS8eyJfwT2u4DeyKTVdL5bs0bmPrb/38="}]},"_npmUser":{"name":"evandroreis-cordya","email":"npm@cordya.ai"},"directories":{},"maintainers":[{"name":"evandroreis-cordya","email":"npm@cordya.ai"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/froid-eval-quality_0.1.2_1788314110828_0.9967835080836109"},"_hasShrinkwrap":false}},"time":{"created":"2026-09-02T01:55:10.609Z","0.1.2":"2026-09-02T01:55:10.999Z","modified":"2026-09-02T01:55:11.334Z"},"maintainers":[{"name":"evandroreis-cordya","email":"npm@cordya.ai"}],"description":"Compile disciplined Behavioral Evaluation Contracts and score their ability to catch known defects.","homepage":"https://github.com/cordya-ai/froid-eval-quality#readme","keywords":["eval","evals","evaluation","agent","agent-eval","llm","behavioral-contract","oracle","contract-strength","testing","ai"],"repository":{"type":"git","url":"git+https://github.com/cordya-ai/froid-eval-quality.git"},"author":{"name":"Murat Ozcan"},"bugs":{"url":"https://github.com/cordya-ai/froid-eval-quality/issues"},"license":"Apache-2.0","readme":"# `eval-quality`\n\n**[Documentation](https://cordya-ai.github.io/froid-eval-quality/)** ·\n[Getting started](https://cordya-ai.github.io/froid-eval-quality/tutorials/getting-started/) ·\n[CLI reference](https://cordya-ai.github.io/froid-eval-quality/reference/cli-commands/) ·\n[npm](https://www.npmjs.com/package/eval-quality)\n\n```bash\nnpx eval-quality --help\n```\n\n### `eval-quality` does three things\n\n1. **Compile**: validate and normalize an eval contract into a machine-readable artifact.\n2. **Seal**: render the brief for the independent evaluator while hiding the planted bug and scoring answer.\n3. **Preflight**: verify baseline environment readiness and probe reachability before running an evaluator.\n\nScoring is the next milestone: comparing the evaluator’s completed findings with the hidden bug signature to determine whether the bug was actually caught.\n\n### What is the eval spec?\n\nIt is the test.\n\nMore precisely, it is the evaluator’s instructions for how to expose a failure and what evidence counts as finding it.\n\nIt defines:\n\n- the behavior being evaluated;\n- the probes the evaluator should perform;\n- the evidence it should inspect;\n- the negative behavior it must rule out;\n- the oracle that determines pass or fail.\n\nFor example:\n\n> Send malformed input.\n> Confirm the request fails.\n> Inspect the full response body.\n> Confirm the expected error.\n> Verify that no record was created.\n\nThe planted bug might be:\n\n> The API returns the correct error but still creates the record.\n\nA weak eval checks only the response and misses the bug.\n\nA strong eval checks the response **and** persistence, so it catches the bug.\n\n### Caveman summary\n\nWrite the eval. Hide the bug. See if the eval catches it.\n\n## Key Concepts\n\nUnderstanding `eval-quality` requires three core artifacts:\n\n| Concept | What it is | Example |\n| --- | --- | --- |\n| **Contract** (`eval-contract.json`) | The test specification defining expected behaviors, oracles (checks), permitted tools, and evidence rules. | \"Verify API rejects invalid JWT and creates zero database records.\" |\n| **Probe** (`probe.json`) | A diagnostic request sent to the environment to test baseline state, reachability, or fault injection. | A request sending an expired token to `/api/v1/resource`. |\n| **Observation** (`observation.json`) | The empirical response evidence recorded when a probe is executed against the environment. | `{ responseStatus: 401, responseBody: { error: \"token_expired\" } }` |\n\n### How They Fit Together\n\n```text\n┌────────────────────────┐      ┌────────────────────────┐      ┌────────────────────────┐\n│     Eval Contract      │      │         Probe          │      │      Observation       │\n│  (The Specification)   │ ───► │ (Diagnostic Request)   │ ───► │   (Empirical Result)   │\n│  \"What should happen\"  │      │  \"Send malformed JWT\"  │      │   \"Got 401, 0 records\" │\n└────────────────────────┘      └────────────────────────┘      └────────────────────────┘\n```\n\n## Elaboration\n\nCompile disciplined agent eval contracts, then check whether those contracts can catch known bugs.\n\nAn agent can produce an answer that reads as correct and is materially wrong. An eval can make the same mistake.\n\nWeak oracle:\n\n```text\nCheck malformed input is handled correctly.\n```\n\nAn evaluator given that instruction sends one malformed request, sees an error come back, and reports success. The record that should never have been created was created anyway. Nobody looked.\n\nStrong oracle:\n\n```text\nSend malformed input. Verify the request fails, inspect the full response body,\nconfirm the specific error, and confirm no record was created.\n```\n\nA passing eval says little when the contract never asked for the probe that would expose the failure. Testing whether the eval can catch a failure you already know about is the first check worth running.\n\n```text\nproduct spec\n  → Behavioral Evaluation Contract\n  → known defect or gameability probe\n  → independent evaluator\n  → per-oracle evidence and a gate decision\n```\n\n## What each part provides\n\n`eval-quality` provides:\n\n- the Behavioral Evaluation Contract schema\n- the oracle vocabulary and authoring rules\n- the contract compiler\n- the environment pre-flight\n- Eval Contract strength scoring (next milestone)\n- versioned evidence output and PASS / WAIVED / CONCERNS / FAIL governance (next milestone)\n\nThe caller provides:\n\n- execution of its chosen agent, harness, or person\n- repeated trials\n- cost accounting\n- the live system and environment-probe implementation\n- a sealed run record returned for ingestion\n\n`eval-quality` executes nothing: it never spawns a process, calls a model, drives a system under test,\nor invokes a judge. Its pure stages are compile, seal, ingest, pre-flight, score, and emit; compile,\nseal, and pre-flight ship, and ingest, score, and emit are the next milestone. Pre-flight probes the\nfixture through the environment-probe port, so a contract that declares a fixture reset\nneeds the caller's probe policy to authorize that operation's method as well as the read methods\nevery other pre-flight leg uses. Engine integration is a later adapter behind a port, not a v0\ndependency. See\n[ADR-004](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-29/ADR-004-execution-boundary.md).\n\n## Who it is for\n\nTeams shipping AI agents, coding skills, review bots, MCP-based assistants, or automated test-generation systems, and teams operating human-on-the-loop or dark-factory delivery.\n\nUse `eval-quality` when all three are true:\n\n- An agent, skill, or model judgment is involved.\n- A plausible-looking output can still be materially wrong.\n- Observable evidence or probes can expose the wrong behavior.\n\nDeterministic work does not need it and already has cheaper, stronger evidence from unit, integration, contract, E2E and performance testing.\n\n## Behavioral Evaluation Contracts\n\nA **Behavioral Evaluation Contract** is a versioned specification of the behaviors to probe, the evidence to collect, the negative cases to exercise, and the rules that decide whether the system passes or fails. **Eval Contract** is the shorthand used from here on. The individual checks inside it are **oracles**. The contract carries no prescribed action sequence; the evaluator chooses its own path.\n\nThe authoring discipline is a small set of rules that survived the experiments: separate the success indicator from the body, read the whole body, probe malformed and negative inputs, verify per record, and cross-check sibling parameters and sibling tools.\n\nA compiler enforces these rules mechanically against the contract artifact, in three classes. Structural errors fail compilation. Coverage gaps score down without blocking. A waived pattern is allowed when it records the named rule, a rationale, a machine-checkable condition, and the approval.\n\nRubrics compile under the same discipline: an anchored scale, a bounded length, named failure-mode penalties, rubric identifiers unique across the contract and criterion identifiers unique inside their own rubric, every criterion stating a question, and every criterion's evidence pointer resolving against the declared interfaces. Authored rubric text that asks a judge to grade the subject's own stated reasoning fails a closed-vocabulary check over the wording.\n\n## How Eval Contract strength scoring works\n\nDo not trust a contract because it looks thorough. Put a known defect behind it, run the evaluator, and check whether the contract's oracles caused the defect to be caught.\n\nTwo probe classes go behind a contract, and a strong contract rejects both:\n\n- **Defect probes**, where the behavior is simply wrong.\n- **Gameability probes**, where the behavior looks compliant while dodging the oracle's intent. A test that raises coverage while asserting nothing is the familiar version of this.\n\nProbes come from qualified historical defects or verified controlled mutations. The corpus separates a visible development set from an immutable sealed set for each scoring version.\n\nEvery required oracle check resolves to exactly one state, and the state travels with the result, so\n\"the check reported\" is never sufficient on its own: `caught`, `confirmed`, `missed`,\n`passed-clean-control`, `false-positive`, `abstained`, `bypassed`, `unreached`, `oracle-error`,\n`judge-error`, `infrastructure-error`, or `not-applicable`.\n\nA required oracle that missed, abstained, errored, or is absent prevents PASS, and a high overall score never overrides it. An infrastructure error or a failed environment pre-flight is not a behavioral result at all; it invalidates the run and is re-executed rather than scored.\n\n## Using it\n\n`eval-quality` is its own repository and package, not a plugin inside another framework.\n\nThe **library** is the primary surface. It exports the contract schema, the oracle vocabulary, the compiler, the pre-flight, and the evidence types. The published typed schema is what lets coding agents author contracts correctly by default, which is how the discipline scales beyond the people who went looking for the tool.\n\nThe **CLI** wraps the same library for callers that cannot import TypeScript: CI jobs, GitHub Actions, PR-review and unit-test bots, other frameworks' skills, and any agent permitted to run a shell command.\n\n### What the CLI Commands Do\n\n- **`compile`**: Typechecks an authored `eval-contract.json`. Verifies that all behaviors, oracles, rubrics, and sensitivity witnesses comply with structural and authoring rules.\n- **`seal`**: Generates a `sealed-evaluator-brief.json` by stripping secret defect signatures, planted answers, and author commentary. The brief carries only the directions and safety bounds the evaluator needs.\n- **`preflight`**: Reduces caller-supplied probe observations against the contract to verify environment baseline readiness and probe reachability. Halts early with exit code `3` if the environment is unready.\n\n### Running the CLI\n\nEvery command runs through `npx` without installing anything:\n\n```bash\nnpx eval-quality compile --in contract.json --out ./eval-out\n\nnpx eval-quality seal --in contract.json --out ./eval-out\n\nnpx eval-quality preflight --contract contract.json \\\n  --probes probes.json --observations observations.json \\\n  --run-id 2026-08-28-a --out ./eval-out\n```\n\nEvery command is non-interactive: no prompt, no terminal check, and no behaviour that differs when\nstdin is a pipe. Each one is a single call into the library plus artifact serialization.\n\n**Input and output.** An input flag left out reads stdin, and `-` names stdin explicitly; at most one\ninput may be `-`. Without `--out` the artifact goes to stdout, so a command composes with a pipe.\nAn `--out` ending in `.json` is a file path; anything else is a directory, and the artifact is\nwritten to `<target>/<kind>.json` where `kind` is `eval-contract`, `sealed-evaluator-brief`, or\n`preflight-verdict`. Diagnostics and errors go to stderr, always, so stdout carries the artifact\nalone.\n\n**Exit codes.**\n\n| Exit Code | Meaning |\n| --- | --- |\n| `0` | success, and every verdict other than FAIL or a promoted CONCERNS |\n| `1` | CONCERNS promoted by `--strict` |\n| `2` | FAIL |\n| `3` | invalid: a pre-flight verdict that did not pass |\n| `4` | structural failure |\n| `5` | runtime fault |\n| `64` | usage error |\n\nCodes 1 and 2 report a scored verdict. Scoring ships in a later release, so no command here reaches\neither yet, and `--strict` changes no code this binary produces. The flag and the two codes are part\nof the published contract, so they are documented now and wired now.\n\n`--strict` is the gate-promotion flag and is accepted on every command. `--strict-inputs` and\n`--no-strict-inputs` are a different switch: they set the compiler's input strictness, which is on\nby default.\n\n**The published JSON Schema.** A consumer that does not read TypeScript validates against the\ntwelve generated documents, published at the `eval-quality/schemas/*` subpath:\n\n```ts\nimport spec from 'eval-quality/schemas/eval-contract.schema.json' with { type: 'json' }\n```\n\nThe import attribute is required: ESM on Node 22 and 24 both throw `ERR_IMPORT_ATTRIBUTE_MISSING`\nwithout it. The development corpus ships the same way, at `eval-quality/corpus/dev/`, so an adopter\ncan read real compiled contracts and one compiled-and-sealed pair without cloning this repository.\n\n## Relationship with Froid and TEA\n\nThe dependency runs one way: TEA uses `eval-quality`, and `eval-quality` knows nothing about TEA.\n\n```mermaid\ngraph LR\n  TEA[\"TEA<br/>(reference authoring client)\"] -- \"drafts a contract, then calls\" --> EQ[\"eval-quality<br/>(this package)\"]\n```\n\nTEA is the reference authoring client. It reads Froid planning artifacts, notices eval-relevant work, drafts a contract, and calls this package. It is not co-installed, and `eval-quality` holds no knowledge of TEA, Froid, or any planning-artifact format.\n\nAny human, bot, CI job, skill, or other framework can author a contract and use `eval-quality` directly. The discipline still applies, because the compiler judges the artifact rather than trusting whoever produced it.\n\nEvaluator runs remain isolated to prevent builder-context leakage and preserve traceability. Stronger contract oracles produced the measured detection improvement.\n\n### Real-World Walkthrough: Testing a `froid-tea` Knowledge Harness\n1. Author an `eval-contract.json` declaring required knowledge step files (e.g. `playwright-utils-mandate.md`).\n2. Run `eval-quality compile --in contract.json` to validate contract structure and discipline rules.\n3. Run `eval-quality seal --in contract.json --out ./run` to generate `sealed-evaluator-brief.json`.\n4. Pass `sealed-evaluator-brief.json` to `froid-tea` to execute the task without seeing answer keys.\n\n## Evidence and limitations\n\nHolding the model, the budget, the system, and the defects fixed, and changing only how the Eval Contract was authored, sealed-evaluator detection moved from **0.33 to 1.00** across three naturally occurring defects, three repetitions per arm, 19 scored runs.\n\nBoth experiment rounds missed at least one preregistered gate. Round 1 recorded `DARK-FACTORY REJECTED`; round 2 block 1 recorded `CONTRACT-DISCIPLINE NOT SUPPORTED`, failing one gate of five on a single unreplicated clean control. The separation comes from two of the three defects, since both arms detected the third in every repetition, and both separating cases carry a recorded measurement-layer confound. The sample covers three defects, one system, and one model. This supports a product-direction decision at narrow scale. Certification would require broader replication.\n\nRead the [product brief](_froid-output/planning-artifacts/briefs/brief-eval-quality-2026-07-17/brief.md) for the product rationale and the [PRD](_froid-output/planning-artifacts/prds/prd-eval-quality-2026-07-17/prd.md) for build requirements. The experiment record includes the [round 1 verdict](experiments/hypothesis-validation/DECISION.md), [round 2 results](experiments/hypothesis-validation/PHASE2-RESULTS.md), [metric summary](experiments/hypothesis-validation/results/summary.md), and [protocol](experiments/hypothesis-validation/HYPOTHESIS_VALIDATION_PLAN.md).\n\n## Architecture status\n\nThe [architecture spine](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-29/ARCHITECTURE-SPINE.md) is split by pipeline half: the compile-and-seal half is epic-ready, while the score half is not. Gate C closed at zero blocking authoring points and 14 of 14 declaration-only predicates. Gate D's generated-current-fields arm matched the hand-written positive control at 3 of 3 seeded-defect catches, so `seal` joins the stage-one order without adding an evidence-precondition field.\n\nContract strength scoring has been open since [ADR-007](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-29/ADR-007-compile-score-split.md): three rounds of external review established that the catch rate was 1.00 by construction, because nothing matched a finding to the defect its probe seeded. That input now exists and the mapping that reads it is owed to a reference implementation.\n\nContract compilation was declared ready in ADR-007 and a fourth review withdrew that claim in [ADR-008](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-29/ADR-008-compile-half-owed-to-calibration.md). The named calibration is now complete. The absent local-only mut2 arm was reconstructed from its recorded base, reproduced its prior black-box behavior, and ran under a pre-registered three-arm, three-repetition design. All three arms composed filters and detected the seeded defect in every valid repetition. This closes the calibration gate narrowly; it does not generalize the historical 0.33-to-1.00 effect beyond one behavior and one controlled mutation.\n\nBoth are documented as defects rather than dressed as decisions, because four rounds have shown that a confidently worded revision is the thing that goes wrong here.\n\nThe decision record, in order: [ADR-001](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-22/ADR-001-evaluator-isolation-boundary.md) on evaluator isolation, [ADR-002](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-22/ADR-002-contract-authoring-discipline.md) on why authoring discipline is the product, [ADR-003](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-29/ADR-003-measurement-mechanics.md) on measurement mechanics, [ADR-004](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-29/ADR-004-execution-boundary.md) on why this package executes nothing, [ADR-005](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-29/ADR-005-review-round-corrections.md) and [ADR-006](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-29/ADR-006-interaction-plan.md) on what review and hand-authoring corrected, [ADR-007](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-29/ADR-007-compile-score-split.md) on the split, [ADR-008](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-29/ADR-008-compile-half-owed-to-calibration.md) on why the other half stopped claiming to be finished too, and [ADR-009](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-29/ADR-009-adversarial-gate-corrections.md) on the seventeen places where two conforming implementations still disagreed. Review triage lives in [`reviews/`](_froid-output/planning-artifacts/architecture/architecture-eval-quality-2026-07-29/reviews/).\n\n## Not building now\n\nDeferred until the contract layer is in real use: claim-to-evidence lineage, semantic checkpoint scoring, process and outcome separation, and first material error attribution.\n\nOut of scope entirely: a new eval engine, a hosted service, a dashboard or GUI, multimodal evaluators, automatic prompt repair, and a generic judge-calibration platform.\n\n## Development\n\n```bash\nnpm install\nnpm run validate            # typecheck, lint, docs, shareable, spine, vectors, schemas, registries, AD-31 table, layers, lineage, boundary, corpus, tests with coverage\nnpm run build               # emit to dist/\nnpm run lint:fix            # auto-fix with Biome\nnpm run test:coverage       # run the suite and fail below AD-30's 90 percent statement and branch floor on core/\nnpm run generate:schemas    # rebuild schemas/*.schema.json from the Zod source\nnpm run check:schemas       # fail if the committed schemas differ from the source by one byte\nnpm run check:ad5-registry  # fail if the failure-code list drifts from the AD-5 table\nnpm run check:lineage       # fail if a module outside the stage table writes an artifact's lineage fields\nnpm run check:boundary      # fail if anything the tarball carries references the planning system that produced it\nnpm run generate:ad31-table # rebuild docs/ad31-coverage-predicates.generated.md from the predicates\nnpm run check:ad31-table    # fail if the committed AD-31 table differs from the builder by one byte\nnpm run generate:dev-corpus # rebuild corpus/dev/ from the contract fixtures through the shipped compile and seal\nnpm run check:corpus        # fail if the committed corpus differs from the builder by one byte\nnpm run build:shareable     # render the planning artifacts to self-contained HTML\nnpm run test:conformance    # run the published port conformance suite against every shipped adapter\n```\n\n`schemas/` holds the twelve published JSON Schema documents, generated from the Zod definitions and\ncommitted. They are the contract for consumers who do not read TypeScript, so they are proven\nequivalent to the source rather than assumed to be: a byte-exact drift check, a rejection suite\nasserting the validator keyword and instance path for every negative fixture, a differential check\ncomparing Zod's verdict against a third-party validator's over a generated corpus, and a\nkeyword-mutation sweep that deletes each published constraint and requires some fixture to notice.\nEdit the Zod schema and regenerate; never hand-edit a file under `schemas/`.\n\nEvery artifact the library hands back is deep-frozen, so it cannot be changed in place. This package\nis ES modules, which are always strict, so an attempt throws a `TypeError` there; a sloppy-mode\ncaller sees the write fail silently. A revision is minted as a new artifact carrying its parent's\ndigest and a revision count one greater. `check:lineage` fails the build when a lineage field is\nwritten outside `src/core/schemas/`, `src/core/lineage/`, and the modules the AD-24 stage table\nnames as that artifact's producer, which today are `src/core/seal/seal.ts` and\n`src/core/preflight/reduce.ts`.\n\nThe `eval-quality/conformance` subpath publishes the port boundary: the four port types, the message\nshapes they carry, and an executable conformance suite. An adapter is conforming when\n`runCorpusPortConformance`, `runClockPortConformance`, `runFileSystemPortConformance`, or\n`runEnvironmentProbePortConformance` returns a report whose `passed` is true, which is the definition\nrather than a paraphrase of one; each returns a report instead of asserting, so the suite carries no\ntest framework and runs under whichever one you already use.\n\n```ts\nimport { runCorpusPortConformance, type CorpusPort } from 'eval-quality/conformance'\n```\n\nThe suite drives a subject through four scenarios and checks six assertions per port method: a\nmechanism failure is a typed fault, exactly one underlying call happens on success and on failure, an\naborted signal rejects promptly, an in-band error value is thrown rather than returned, and a\nsuccessful call returns a response the published schema accepts. The environment-probe port adds\nthirteen more from AD-35's default-deny target policy. `npm run test:conformance` runs the suite\nagainst the three adapters this package ships and against an in-repository probe subject that exists\nonly as the suite's own subject.\n\n`docs/ad31-coverage-predicates.generated.md` holds AD-31's published predicate table, emitted from\nthe seven relevance predicates and their seven satisfaction twins run over a hand-authored contract\ncorpus. It is generated by `npm run generate:ad31-table` and guarded by `npm run check:ad31-table`,\na byte-exact drift check that fails when a predicate changes and the committed document does not, so\nthe table is evidence the predicates produce rather than documentation kept beside them. Regenerate\nrather than hand-edit it.\n\n`build:shareable` renders this README, the product brief, the PRD, the architecture spine, all nine ADRs, and every document those pages link to (contributing, code of conduct, security, licence, and the four experiment records) to `_froid-output/shareable/` as standalone styled HTML for sharing outside the repo. Rendering the linked documents is what lets a recipient without repository access follow the evidence, contribution, security, and licence links instead of hitting a 404; anything that has no page of its own, such as a directory, is marked in the export as needing repository access. Regenerate rather than hand-edit those files: `check:shareable` fails the build when the committed export is stale or carries a repository URL that is not the canonical one. Mermaid diagrams render as code blocks there, which is a known limitation.\n\n## Contributing\n\nSee [CONTRIBUTING.md](CONTRIBUTING.md) and our [Code of Conduct](CODE_OF_CONDUCT.md).\n\n## Security\n\nSee [SECURITY.md](SECURITY.md). Please do not open a public issue for vulnerabilities.\n\n## License\n\nApache-2.0 © Murat Ozcan. See [LICENSE](LICENSE).\n","readmeFilename":"README.md","_rev":"1-db2276cfd4af104ee39f05b749e65495"}