{"_id":"@yylo/benchmark","_rev":"9-19b35579d41ce2820a3d1b33f1f5bfab","name":"@yylo/benchmark","dist-tags":{"next":"0.1.1-rc.2","latest":"0.2.1"},"versions":{"0.1.0-rc.1":{"name":"@yylo/benchmark","version":"0.1.0-rc.1","author":{"name":"JUNO AI INC.","email":"support@yylo.dev"},"license":"MIT","_id":"@yylo/benchmark@0.1.0-rc.1","maintainers":[{"name":"the_golch","email":"golchin@askbudi.ai"}],"homepage":"https://yylo.dev","bugs":{"url":"https://github.com/yylo-dev/yylo-benchmark/issues"},"bin":{"yylo-benchmark":"dist/bin.js"},"dist":{"shasum":"23225ac47c3639bf5bb48a1c0be44a42d27e3ecc","tarball":"https://registry.npmjs.org/@yylo/benchmark/-/benchmark-0.1.0-rc.1.tgz","fileCount":19,"integrity":"sha512-cgUMOauej2PcSBd+nJuSdkhhVAeTGSfEOWI1q53Di+DHyutIMNMyffqYC0vrAFgalcHSQut7wUbVtrwfNj3QFg==","signatures":[{"sig":"MEUCIFgCUpYJzVno1l9SPx7GjFwNFTBQzMtD09hOm1jVd5arAiEAu1pKWzhZ5sLK7Tnf9SnudRTpuOERmy3uOnox2RNMw8c=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":2284335},"main":"./dist/index.js","type":"module","_from":"file:/private/tmp/yylo-release-rc12-ledger-rc2-plan.json.artifacts/npm-benchmark/yylo-benchmark-0.1.0-rc.1.tgz","types":"./dist/index.d.ts","engines":{"node":">=20.10.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"test":"vitest run","build":"tsup","clean":"rm -rf dist","prepack":"npm run clean && npm run build","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"the_golch","email":"golchin@askbudi.ai"},"_resolved":"/private/tmp/yylo-release-rc12-ledger-rc2-plan.json.artifacts/npm-benchmark/yylo-benchmark-0.1.0-rc.1.tgz","_integrity":"sha512-cgUMOauej2PcSBd+nJuSdkhhVAeTGSfEOWI1q53Di+DHyutIMNMyffqYC0vrAFgalcHSQut7wUbVtrwfNj3QFg==","repository":{"url":"git+https://github.com/yylo-dev/yylo-benchmark.git","type":"git"},"_npmVersion":"11.17.0","description":"YYLO Benchmark: immutable task-case and workflow evaluation contracts and CLI","directories":{},"_nodeVersion":"22.22.3","dependencies":{"zod":"^3.22.4","yaml":"^2.9.0","commander":"^11.1.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.5.1","vitest":"^1.0.4","typescript":"^5.3.3","@types/node":"^20.10.0","@vitest/coverage-v8":"^1.6.1"},"_npmOperationalInternal":{"tmp":"tmp/benchmark_0.1.0-rc.1_1787628084992_0.6342464542253483","host":"s3://npm-registry-packages-npm-production"}},"0.1.0-rc.9":{"name":"@yylo/benchmark","version":"0.1.0-rc.9","author":{"name":"JUNO AI INC.","email":"support@yylo.dev"},"license":"MIT","_id":"@yylo/benchmark@0.1.0-rc.9","maintainers":[{"name":"the_golch","email":"golchin@askbudi.ai"}],"homepage":"https://yylo.dev","bugs":{"url":"https://github.com/yylo-dev/yylo-benchmark/issues"},"bin":{"yylo-benchmark":"dist/bin.js"},"dist":{"shasum":"4fc1f803c5b771901b4a1ede81b43a16085b7620","tarball":"https://registry.npmjs.org/@yylo/benchmark/-/benchmark-0.1.0-rc.9.tgz","fileCount":10,"integrity":"sha512-I2BBEkpi+H6I+AUahAAgzemQ2L98xV1YCeZ9lfwhqAvsaJyNL9Wm+0RTH2p6TyWDjQ/ar3DCH3V/6fNSbR71NQ==","signatures":[{"sig":"MEUCIQC2y7f86twoKDkGxMV6oUJ3QGBf+zwAdn4ay+N431/S4wIgewE7eZWgIQQIf+LlSoLyL6GF5M3NSZbdfbQ+sUmpSmk=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1540224},"main":"./dist/index.js","type":"module","_from":"file:/private/tmp/yylo-cli-rc11-benchmark-rc9-plan.json.artifacts/npm-benchmark/yylo-benchmark-0.1.0-rc.9.tgz","types":"./dist/index.d.ts","engines":{"node":">=20.10.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"test":"vitest run","build":"tsup","clean":"rm -rf dist","prepack":"npm run clean && npm run build","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"the_golch","email":"golchin@askbudi.ai"},"_resolved":"/private/tmp/yylo-cli-rc11-benchmark-rc9-plan.json.artifacts/npm-benchmark/yylo-benchmark-0.1.0-rc.9.tgz","_integrity":"sha512-I2BBEkpi+H6I+AUahAAgzemQ2L98xV1YCeZ9lfwhqAvsaJyNL9Wm+0RTH2p6TyWDjQ/ar3DCH3V/6fNSbR71NQ==","repository":{"url":"git+https://github.com/yylo-dev/yylo-benchmark.git","type":"git"},"_npmVersion":"9.5.0","description":"YYLO Benchmark: flexible isolated task and workflow evaluation","directories":{},"_nodeVersion":"18.15.0","dependencies":{"zod":"^3.22.4","yaml":"^2.9.0","dotenv":"^16.6.1","commander":"^11.1.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"readmeFilename":"README.md","devDependencies":{"tsup":"^8.5.1","vitest":"^1.0.4","typescript":"^5.3.3","@types/node":"^20.10.0","@vitest/coverage-v8":"^1.6.1"},"_npmOperationalInternal":{"tmp":"tmp/benchmark_0.1.0-rc.9_1788372684700_0.6682539515867605","host":"s3://npm-registry-packages-npm-production"}},"0.1.0":{"name":"@yylo/benchmark","version":"0.1.0","author":{"name":"JUNO AI INC.","email":"support@yylo.dev"},"license":"MIT","_id":"@yylo/benchmark@0.1.0","maintainers":[{"name":"the_golch","email":"golchin@askbudi.ai"}],"homepage":"https://yylo.dev","bugs":{"url":"https://github.com/yylo-dev/yylo-benchmark/issues"},"bin":{"yylo-benchmark":"dist/bin.js"},"dist":{"shasum":"c7450eff819b8c2e66a586b88a7e7b3558b5b826","tarball":"https://registry.npmjs.org/@yylo/benchmark/-/benchmark-0.1.0.tgz","fileCount":10,"integrity":"sha512-PlaJQis5v9/wyCyXsQsIQyglAnj3Ceb8sGkCLOzxtJ/oGJHsq353+MRNGGtz54DbAJXT+5I3rDBktE0vn6dr4w==","signatures":[{"sig":"MEUCIQDfGpyrDfsmLWN+Jog56wWI6+28aGKcm2sdyZirnVwsewIgIfHUZHSm6gfhoGTZJnxqjlOtKytuU5cVsi0jZ2I8K9k=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1541389},"main":"./dist/index.js","type":"module","_from":"file:/private/tmp/yylo-v0.2.2-release-worktree/.release-artifacts/yylo-benchmark-0.1.0.tgz","types":"./dist/index.d.ts","engines":{"node":">=20.10.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"test":"vitest run","build":"tsup","clean":"rm -rf dist","prepack":"npm run clean && npm run build","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"the_golch","email":"golchin@askbudi.ai"},"_resolved":"/private/tmp/yylo-v0.2.2-release-worktree/.release-artifacts/yylo-benchmark-0.1.0.tgz","_integrity":"sha512-PlaJQis5v9/wyCyXsQsIQyglAnj3Ceb8sGkCLOzxtJ/oGJHsq353+MRNGGtz54DbAJXT+5I3rDBktE0vn6dr4w==","repository":{"url":"git+https://github.com/yylo-dev/yylo-benchmark.git","type":"git"},"_npmVersion":"9.5.0","description":"YYLO Benchmark: flexible isolated task and workflow evaluation","directories":{},"_nodeVersion":"18.15.0","dependencies":{"zod":"^3.22.4","yaml":"^2.9.0","dotenv":"^16.6.1","commander":"^11.1.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.5.1","vitest":"^1.0.4","typescript":"^5.3.3","@types/node":"^20.10.0","@vitest/coverage-v8":"^1.6.1"},"_npmOperationalInternal":{"tmp":"tmp/benchmark_0.1.0_1788804927245_0.5417985629519226","host":"s3://npm-registry-packages-npm-production"}},"0.1.1-rc.2":{"name":"@yylo/benchmark","version":"0.1.1-rc.2","author":{"name":"JUNO AI INC.","email":"support@yylo.dev"},"license":"MIT","_id":"@yylo/benchmark@0.1.1-rc.2","maintainers":[{"name":"the_golch","email":"golchin@askbudi.ai"}],"homepage":"https://yylo.dev","bugs":{"url":"https://github.com/yylo-dev/yylo-benchmark/issues"},"bin":{"yylo-benchmark":"dist/bin.js"},"dist":{"shasum":"5003d81a0457f3229c060966f3ee7c0da28a037b","tarball":"https://registry.npmjs.org/@yylo/benchmark/-/benchmark-0.1.1-rc.2.tgz","fileCount":13,"integrity":"sha512-XulCxkBbf9s+P9tdhFBGTYn0+H+IRMpliYLou9odNDuZgTiGGEeQ01RlUdl67CKiv9gWGviewR9vYYKyNSHbJw==","signatures":[{"sig":"MEQCICRfqQuGw2QVkfC7h+FNc1EOGH3aSafRXxq98ynfSQfEAiAb0SYpHr+bA414fldczYDE8j8sDxrPMjQVH27LDk9dTA==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":2816183},"main":"./dist/index.js","type":"module","_from":"file:/Users/mahdiyar/JunoReleaseReceipts/juno-mono-002/20260911T020000Z-yylo-cli-0.2.3-rc.2-benchmark-0.1.1-rc.2/yylo-benchmark-0.1.1-rc.2.tgz","types":"./dist/index.d.ts","engines":{"node":">=20.10.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"test":"vitest run","build":"tsup","clean":"rm -rf dist","prepack":"npm run clean && npm run build","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"the_golch","email":"golchin@askbudi.ai"},"_resolved":"/Users/mahdiyar/JunoReleaseReceipts/juno-mono-002/20260911T020000Z-yylo-cli-0.2.3-rc.2-benchmark-0.1.1-rc.2/yylo-benchmark-0.1.1-rc.2.tgz","_integrity":"sha512-XulCxkBbf9s+P9tdhFBGTYn0+H+IRMpliYLou9odNDuZgTiGGEeQ01RlUdl67CKiv9gWGviewR9vYYKyNSHbJw==","repository":{"url":"git+https://github.com/yylo-dev/yylo-benchmark.git","type":"git"},"_npmVersion":"11.17.0","description":"YYLO Benchmark: isolated evaluation and governed production workflows","directories":{},"_nodeVersion":"22.22.3","dependencies":{"zod":"^3.22.4","yaml":"^2.9.0","dotenv":"^16.6.1","commander":"^11.1.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"readmeFilename":"README.md","devDependencies":{"tsup":"^8.5.1","vitest":"^1.0.4","typescript":"^5.3.3","@types/node":"^20.10.0","@vitest/coverage-v8":"^1.6.1"},"_npmOperationalInternal":{"tmp":"tmp/benchmark_0.1.1-rc.2_1789099191162_0.8455372235722054","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@yylo/benchmark","version":"0.1.1","author":{"name":"JUNO AI INC.","email":"support@yylo.dev"},"license":"MIT","_id":"@yylo/benchmark@0.1.1","maintainers":[{"name":"the_golch","email":"golchin@askbudi.ai"}],"homepage":"https://yylo.dev","bugs":{"url":"https://github.com/yylo-dev/yylo-benchmark/issues"},"bin":{"yylo-benchmark":"dist/bin.js"},"dist":{"shasum":"7e9557d4d672c7249badd19529ac7b05212b8bf5","tarball":"https://registry.npmjs.org/@yylo/benchmark/-/benchmark-0.1.1.tgz","fileCount":13,"integrity":"sha512-eIqUmqo8FOCyQHMQlRFj6XCdkhU7k+CJcGF6wZI1imVLjaHYxuIdw1BFqIw0DTf5gyHv/7pH+kCLQynWChwL2Q==","signatures":[{"sig":"MEYCIQDIDUK/QCUgnlYGEp8331IDaptzvBL8sQrUcbPaeKXYPAIhAMK4v5ZL2zHwam4qacbz7EcqUeh51PwPPpYufVyVou1x","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"sig":"MEYCIQCD2han4uTdWe6TOPsMx8Px0vdB3RLtH2+2HdMEpsKBCQIhANf7lZs8LZhL7RMb7CGhzYtbgggT0o0S6KXAkXIJScsU","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":2816158},"main":"./dist/index.js","type":"module","_from":"file:/Users/mahdiyar/.local/share/yylo/releases/cli-0.2.3-3ecc4d8cadc1/artifacts/yylo-benchmark-0.1.1.tgz","types":"./dist/index.d.ts","engines":{"node":">=20.10.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"test":"vitest run","build":"tsup","clean":"rm -rf dist","prepack":"npm run clean && npm run build","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"the_golch","email":"golchin@askbudi.ai"},"_resolved":"/Users/mahdiyar/.local/share/yylo/releases/cli-0.2.3-3ecc4d8cadc1/artifacts/yylo-benchmark-0.1.1.tgz","_integrity":"sha512-eIqUmqo8FOCyQHMQlRFj6XCdkhU7k+CJcGF6wZI1imVLjaHYxuIdw1BFqIw0DTf5gyHv/7pH+kCLQynWChwL2Q==","repository":{"url":"git+https://github.com/yylo-dev/yylo-benchmark.git","type":"git"},"_npmVersion":"11.17.0","description":"YYLO Benchmark: isolated evaluation and governed production workflows","directories":{},"_nodeVersion":"22.22.3","dependencies":{"zod":"^3.22.4","yaml":"^2.9.0","dotenv":"^16.6.1","commander":"^11.1.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.5.1","vitest":"^1.0.4","typescript":"^5.3.3","@types/node":"^20.10.0","@vitest/coverage-v8":"^1.6.1"},"_npmOperationalInternal":{"tmp":"tmp/benchmark_0.1.1_1789544041008_0.9760098171692768","host":"s3://npm-registry-packages-npm-production"}},"0.1.2":{"name":"@yylo/benchmark","version":"0.1.2","author":{"name":"JUNO AI INC.","email":"support@yylo.dev"},"license":"MIT","_id":"@yylo/benchmark@0.1.2","maintainers":[{"name":"the_golch","email":"golchin@askbudi.ai"}],"homepage":"https://yylo.dev","bugs":{"url":"https://github.com/yylo-dev/yylo-benchmark/issues"},"bin":{"yylo-benchmark":"dist/bin.js"},"dist":{"shasum":"c629654df7ed206125f4772fd8badc4e9aa5df37","tarball":"https://registry.npmjs.org/@yylo/benchmark/-/benchmark-0.1.2.tgz","fileCount":13,"integrity":"sha512-V75nH4Q0/oepr4X/uWRkJW8+VCFza1qm9Z5XHNXR0gjTXUrHK588ln3V/URv+5Bqx9KMDl0vugm5Fyhae9QvNA==","signatures":[{"sig":"MEUCIQCjiEIVaFcJx5xZPMXShgDiZPLQodOZEM8j/emLSW0dDwIgQIMnlR7qbWkzP0hkYy8o0Ykre3YmaRroGi7dnJcQudM=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"sig":"MEUCIQCAwPLIyitWpC+67+W6XuJ5Q0ufAmUbPECrrRH/2mVkQgIgEf65q48s7FwN2LcG+txWP9DV+eccWf6lhmhJUKjXvGE=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":2817122},"main":"./dist/index.js","type":"module","_from":"file:/Users/mahdiyar/.local/state/yylo/release-artifacts/skills-fix-ea42b7342-20260921T145129/artifacts/yylo-benchmark-0.1.2.tgz","types":"./dist/index.d.ts","engines":{"node":">=20.10.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"test":"vitest run","build":"tsup","clean":"rm -rf dist","prepack":"npm run clean && npm run build","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"the_golch","email":"golchin@askbudi.ai"},"_resolved":"/Users/mahdiyar/.local/state/yylo/release-artifacts/skills-fix-ea42b7342-20260921T145129/artifacts/yylo-benchmark-0.1.2.tgz","_integrity":"sha512-V75nH4Q0/oepr4X/uWRkJW8+VCFza1qm9Z5XHNXR0gjTXUrHK588ln3V/URv+5Bqx9KMDl0vugm5Fyhae9QvNA==","repository":{"url":"git+https://github.com/yylo-dev/yylo-benchmark.git","type":"git"},"_npmVersion":"11.17.0","description":"YYLO Benchmark: isolated evaluation and governed production workflows","directories":{},"_nodeVersion":"22.22.3","dependencies":{"zod":"^3.22.4","yaml":"^2.9.0","dotenv":"^16.6.1","commander":"^11.1.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.5.1","vitest":"^1.0.4","typescript":"^5.3.3","@types/node":"^20.10.0","@vitest/coverage-v8":"^1.6.1"},"_npmOperationalInternal":{"tmp":"tmp/benchmark_0.1.2_1790018779025_0.6388634561620439","host":"s3://npm-registry-packages-npm-production"}},"0.1.3":{"name":"@yylo/benchmark","version":"0.1.3","author":{"name":"JUNO AI INC.","email":"support@yylo.dev"},"license":"MIT","_id":"@yylo/benchmark@0.1.3","maintainers":[{"name":"the_golch","email":"golchin@askbudi.ai"}],"homepage":"https://yylo.dev","bugs":{"url":"https://github.com/yylo-dev/yylo-benchmark/issues"},"bin":{"yylo-benchmark":"dist/bin.js"},"dist":{"shasum":"de983d073d371c7b68a270690923a36bf3bd2314","tarball":"https://registry.npmjs.org/@yylo/benchmark/-/benchmark-0.1.3.tgz","fileCount":13,"integrity":"sha512-E7d8F84LcBgJ6NYtP2BFfTxcc4gRqP2zH5pS1nSe512Re/vfJUTEAwfZ/kfA6h/Yk5XS0sKNIBQORlWRv6ezvg==","signatures":[{"sig":"MEYCIQCcsGsEBLpYnFSgyrDr0gVdjnrLIjEtOsGJneUMjdIPzAIhAPCTDdI8oV5c50Cdnp53L7ScIz9KnUs8ODJLeJ37IOgN","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"sig":"MEUCIQDXdvEmag6EdsfO9sou+qkEOKslx68OMUcPU6lJFzUnEAIgA8MjGvuqzqBlb7yY7pFkcLEe7RMP4SExCQD6qM7iBcs=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":2840861},"main":"./dist/index.js","type":"module","_from":"file:/Users/mahdiyar/.local/state/yylo/release-artifacts/v0.2.9-cf304a86c-session-01a0cd13-e6bd-722e-aa5b-fa255efb5350/local-install/yylo-benchmark-0.1.3.tgz","types":"./dist/index.d.ts","engines":{"node":">=20.10.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"test":"vitest run","build":"tsup","clean":"rm -rf dist","prepack":"npm run clean && npm run build","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"the_golch","email":"golchin@askbudi.ai"},"_resolved":"/Users/mahdiyar/.local/state/yylo/release-artifacts/v0.2.9-cf304a86c-session-01a0cd13-e6bd-722e-aa5b-fa255efb5350/local-install/yylo-benchmark-0.1.3.tgz","_integrity":"sha512-E7d8F84LcBgJ6NYtP2BFfTxcc4gRqP2zH5pS1nSe512Re/vfJUTEAwfZ/kfA6h/Yk5XS0sKNIBQORlWRv6ezvg==","repository":{"url":"git+https://github.com/yylo-dev/yylo-benchmark.git","type":"git"},"_npmVersion":"11.17.0","description":"YYLO Benchmark: isolated evaluation and governed production workflows","directories":{},"_nodeVersion":"22.22.3","dependencies":{"zod":"^3.22.4","yaml":"^2.9.0","dotenv":"^16.6.1","commander":"^11.1.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.5.1","vitest":"^1.0.4","typescript":"^5.3.3","@types/node":"^20.10.0","@vitest/coverage-v8":"^1.6.1"},"_npmOperationalInternal":{"tmp":"tmp/benchmark_0.1.3_1790148090808_0.8845028623540125","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@yylo/benchmark","version":"0.2.0","author":{"name":"JUNO AI INC.","email":"support@yylo.dev"},"license":"MIT","_id":"@yylo/benchmark@0.2.0","maintainers":[{"name":"the_golch","email":"golchin@askbudi.ai"}],"homepage":"https://yylo.dev","bugs":{"url":"https://github.com/yylo-dev/yylo-benchmark/issues"},"bin":{"yylo-benchmark":"dist/bin.js"},"dist":{"shasum":"aa56f8b0064e1f18421914cdd8df7e30997ce5d4","tarball":"https://registry.npmjs.org/@yylo/benchmark/-/benchmark-0.2.0.tgz","fileCount":10,"integrity":"sha512-FLF52bgif7AX7YuBoGl5Z5ckV3csyTbhRfrWvZ5rI4fgGnb7mJaM5dU3v5GOr5hqfOzXQnh70728X/Ekk04VcQ==","signatures":[{"sig":"MEUCIQCEmLMCTnOYEg5xUjecUcs69Aiu9dhlhpOIJMaL+jv6rgIgfkZ4M2wXwAv2wkNIP7of/l0AMBQ54hzB/tYF6K2xn/8=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"sig":"MEYCIQCTIP/TTI7fwKoCLhAPjpeCk7YXrjyBbjGVLL+CYTa0LwIhAJ7jOaKWx0Ceo4DgRLbl+llph3MvwBtFljv+Wogc9cYd","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":286167},"main":"./dist/index.js","type":"module","_from":"file:/Users/mahdiyar/.local/state/yylo/release-artifacts/cli-0.2.10-eb00946d8-01a0d4ce/yylo-benchmark-0.2.0.tgz","types":"./dist/index.d.ts","engines":{"node":">=20.10.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"scripts":{"test":"vitest run","build":"tsup","clean":"rm -rf dist","prepack":"npm run clean && npm run build","typecheck":"tsc --noEmit","test:watch":"vitest"},"_npmUser":{"name":"the_golch","email":"golchin@askbudi.ai"},"_resolved":"/Users/mahdiyar/.local/state/yylo/release-artifacts/cli-0.2.10-eb00946d8-01a0d4ce/yylo-benchmark-0.2.0.tgz","_integrity":"sha512-FLF52bgif7AX7YuBoGl5Z5ckV3csyTbhRfrWvZ5rI4fgGnb7mJaM5dU3v5GOr5hqfOzXQnh70728X/Ekk04VcQ==","repository":{"url":"git+https://github.com/yylo-dev/yylo-benchmark.git","type":"git"},"_npmVersion":"11.17.0","description":"YYLO Benchmark: thin task/workflow experiments and independent evaluations","directories":{},"_nodeVersion":"22.22.3","dependencies":{"zod":"^3.22.4","yaml":"^2.9.0","dotenv":"^16.6.1","commander":"^11.1.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.5.1","vitest":"^1.0.4","typescript":"^5.3.3","@types/node":"^20.10.0","@vitest/coverage-v8":"^1.6.1"},"_npmOperationalInternal":{"tmp":"tmp/benchmark_0.2.0_1790277949836_0.3098881396928801","host":"s3://npm-registry-packages-npm-production"}},"0.2.1":{"_id":"@yylo/benchmark@0.2.1","bin":{"yylo-benchmark":"dist/bin.js"},"bugs":{"url":"https://github.com/yylo-dev/yylo-benchmark/issues"},"dist":{"shasum":"9bebccd43d3a6eee865493a5a204bd7cd55a2043","tarball":"https://registry.npmjs.org/@yylo/benchmark/-/benchmark-0.2.1.tgz","fileCount":10,"integrity":"sha512-9zF6P8EZ6wwSemLsYSugC5ttnL4A2awO0ETDdoHECAOPgRIx1gmsp3/Ce0en6LzlMY7VN0ibOgM8kvpTZOMrfg==","signatures":[{"sig":"MEUCIQCjROcHJiX6dOhAw7Dc/0fmMW/zMiyMq3IgBWFL1nOY/QIgMVf2/mdZ6tgsE4TRRIu3+Y0S71D77w7TkHkHrgat2BE=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQCuiR5RqjzF483dT2MMAIOBjmsQeck05QXa9rQsqv1X9AIgX9RDvlod4jQEDfp1Sr3zX1LLhTg8V3WscQi+GKBv7W8="}],"unpackedSize":369928},"main":"./dist/index.js","name":"@yylo/benchmark","type":"module","_from":"file:/Users/mahdiyar/.local/state/yylo/release-artifacts/cli-0.2.11-fbc6582ee/artifacts/yylo-benchmark-0.2.1.tgz","types":"./dist/index.d.ts","author":{"name":"JUNO AI INC.","email":"support@yylo.dev"},"engines":{"node":">=20.10.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"license":"MIT","scripts":{"test":"vitest run","build":"tsup","clean":"rm -rf dist","prepack":"npm run clean && npm run build","typecheck":"tsc --noEmit","test:watch":"vitest"},"version":"0.2.1","_npmUser":{"name":"the_golch","email":"golchin@askbudi.ai"},"homepage":"https://yylo.dev","_resolved":"/Users/mahdiyar/.local/state/yylo/release-artifacts/cli-0.2.11-fbc6582ee/artifacts/yylo-benchmark-0.2.1.tgz","_integrity":"sha512-9zF6P8EZ6wwSemLsYSugC5ttnL4A2awO0ETDdoHECAOPgRIx1gmsp3/Ce0en6LzlMY7VN0ibOgM8kvpTZOMrfg==","repository":{"url":"git+https://github.com/yylo-dev/yylo-benchmark.git","type":"git"},"_npmVersion":"11.17.0","description":"YYLO Benchmark: thin task/workflow experiments and independent evaluations","directories":{},"maintainers":[{"name":"the_golch","email":"golchin@askbudi.ai"}],"_nodeVersion":"22.22.3","dependencies":{"zod":"^3.22.4","yaml":"^2.9.0","dotenv":"^16.6.1","commander":"^11.1.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"tsup":"^8.5.1","vitest":"^1.0.4","typescript":"^5.3.3","@types/node":"^20.10.0","@vitest/coverage-v8":"^1.6.1"},"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/benchmark_0.2.1_1791075936781_0.19222762883812017"}}},"time":{"created":"2026-08-25T03:21:24.798Z","modified":"2026-10-04T01:05:37.039Z","0.1.0-rc.1":"2026-08-25T03:21:25.195Z","0.1.0-rc.9":"2026-09-02T18:11:24.851Z","0.1.0":"2026-09-07T18:15:27.400Z","0.1.1-rc.2":"2026-09-11T03:59:51.340Z","0.1.1":"2026-09-16T07:34:01.158Z","0.1.2":"2026-09-21T19:26:19.142Z","0.1.3":"2026-09-23T07:21:30.947Z","0.2.0":"2026-09-24T19:25:49.982Z","0.2.1":"2026-10-04T01:05:36.878Z"},"bugs":{"url":"https://github.com/yylo-dev/yylo-benchmark/issues"},"author":{"name":"JUNO AI INC.","email":"support@yylo.dev"},"license":"MIT","homepage":"https://yylo.dev","repository":{"url":"git+https://github.com/yylo-dev/yylo-benchmark.git","type":"git"},"description":"YYLO Benchmark: thin task/workflow experiments and independent evaluations","maintainers":[{"name":"the_golch","email":"golchin@askbudi.ai"}],"readme":"# YYLO Benchmark\n\nA thin **trusted-host experiment runner** for historical Ledger tasks, supplied coding prompts, and workflows. Compare models, harnesses and configurations; evaluate retained outputs with different checks or judges later.\n\n```text\nreviewed case -> independent attempts -> retained outputs\n                                          |   |   |\n                                      checks  A   B  <- new judges at any time\n                                          |   |   |\n                                      comparison rows\n```\n\nBenchmark does not choose a winner or combine judge opinions with test results. It does not implement a workflow engine, production authority, provider catalog, repair loop or automatic retry.\n\n## Checklist release — 0.2.1\n\nThe package source declares **0.2.1**, coordinated with CLI **0.2.11** and canonical\nskills **2.1.1**. This adds frozen project/task checklists and deterministic loss to\nthe thin v3 runner introduced in 0.2.0. Skills remain independently distributed. These version changes do not establish a published release or\nupgrade any installed runtime. CLI 0.2.11 requires Benchmark 0.2.1 exactly; install only\nthrough a separately authorized release process.\n\nThe source now uses v3 case/attempt/evaluation records. The old v1/v2 APIs, configuration, plan/recover/doctor/regrade/rejudge commands, plugin registry and governed-production workflow boundary are retired. Ordinary Workflow Runner execution remains supported through delegation. Old plans and evidence are **not migrated, overwritten, deleted or reinterpreted**. Use their original pinned implementation if historical inspection is necessary. This source change is not a package publication or global runtime upgrade.\n\nThe historical [ten-task retrospective](https://yylo.dev/blog/ten-task-cli-benchmark) describes an exploratory study, not a ranking or certification of this implementation. Historical evidence remains historical.\n\n## Requirements and boundaries\n\nNode 20.10+, Git, tar and POSIX process groups. The selected harness and its setup dependencies must be installed. Tests require no paid provider calls. Windows execution is unsupported.\n\nEvery independent attempt starts with a fresh repository containing the reviewed pre-solution files, **not a linked worktree or cloned future history**. Default exclusions are `.juno_task`, `.gitmodules`, `hidden-graders`, and `reference-solutions`. Add case-specific answer paths with `--exclude`. When a historical task genuinely edits controller-owned product source, use a narrowly reviewed `--include .juno_task/scripts` to override the default exclusion for that source subtree only; never include task/completion/answer storage. Explicit `--exclude` always wins. Symlinks and gitlinks are rejected unless excluded or materialized in a separately reviewed input repository. Setup and generated dependencies belong in ignore rules; retained output includes tracked files and unignored new files.\n\nReference solutions, completion responses, hidden checks, and other attempts must not be supplied as candidate context. Preparation requires explicit review; Benchmark cannot discover every answer-bearing document. Known exposure is a disqualification, not a model failure.\n\nThis is **workspace/context hygiene, not filesystem/account/network isolation**. The process inherits host authentication and can deliberately access other host paths or public answers. Shared authentication is not a private account boundary. Routing/Git/Pi override environment variables are discarded to avoid accidental inherited sessions/controller routing. Pi receives a private session directory. Workflows own their own sessions. Never use this runner as an authorization grant for production workflows.\n\n## 1. Prepare a case once\n\nAsk Ledger for an assisted proposal (read-only; no candidate dispatch):\n\n```bash\nyylo-benchmark case draft --ledger-task TASK_ID > /tmp/case-draft.json\n```\n\nThe draft includes the task body and reference-commit candidate, **not the completion response**. It deliberately leaves the base unset: reconstruct the original development range, review original requirements, and create a prompt file. Do not blindly use an integration repair's parent as the original baseline. The latest task body may itself need historical review. The tool does not promise automatic historical reconstruction or a perfect oracle.\n\nFor either a historical task or your own prompt:\n\n```bash\nyylo-benchmark case create \\\n  --source /path/to/repository --base PRE_SOLUTION_COMMIT \\\n  --reference SOLUTION_COMMIT \\\n  --prompt /tmp/requirements.md --exclude private-answers \\\n  --output /tmp/cases/my-case --reviewed\n```\n\n`--reference` and `--ledger-task` are optional provenance; the reference must differ from and descend from the base. It is never copied into candidate input. Case storage must be outside the source repository. `--workflow path/in/source.yaml` captures a tracked workflow at the chosen base. No models are launched during case creation. Use baseline/reference controls to review behavior-based checks before scoring; this is evaluation work, not a new orchestration framework.\n\n### Optional frozen project/task criteria\n\nTo measure quality with a reusable checklist, add `--project-criteria /external/project.yaml`\nand/or `--criteria /external/task.yaml` to `case create`. The files are explicit\nYAML/JSON—no registry, discovery, weights, inheritance or overrides. Each has:\n\n```yaml\nname: Greeting behavior\nversion: \"1\"\nassumptions:\n  - Node.js is available; publication is outside this exercise.\ncriteria:\n  - id: TASK.greeting\n    pass_when: Running node greet.cjs Ada prints exactly Hello, Ada! followed by one newline to stdout and exits zero.\n```\n\nIDs must be unique across both files and match `[A-Za-z][A-Za-z0-9_.-]{0,63}`.\nEmpty/whitespace-only text and unknown fields are refused. `assumptions` is\noptional; `criteria` is nonempty, at most 100 entries per document. File input is\nlimited to 64 KiB per document; YAML aliases and duplicate keys are refused.\n\nThe case freezes normalized documents and their canonical digest, and exposes\nthose public criteria/assumptions in candidate instructions. Attempts and later\nevaluations retain the same contract. Changing a source file cannot revise an\nexisting case. Cases without criteria keep legacy v3 behavior.\n\nFor incomplete historical tasks, record reconstruction assumptions and obtain\noperator approval before candidate inspection. Review baseline/reference,\nnegative and valid alternative-implementation controls; never require the\nreference patch's shape. The standalone\n[benchmark-yylo skill](https://github.com/yylo-dev/yylo-skills/tree/main/skills/benchmark-yylo)\nowns the [checklist guide](https://github.com/yylo-dev/yylo-skills/blob/main/skills/benchmark-yylo/references/checklists.md)\nand reusable project/task examples. This README documents the runtime API, not a\nsecond skill distribution.\n\n## 2. Run treatments\n\nA treatment JSON file chooses one model, harness and configuration:\n\n```json\n{\n  \"name\": \"pi-model-a\",\n  \"model\": \"provider/model-a\",\n  \"harness\": \"yylo_pi\",\n  \"executable\": \"yy\",\n  \"args\": [\"--thinking\", \"medium\"],\n  \"timeout_ms\": 1800000,\n  \"configuration\": {}\n}\n```\n\n```bash\nyylo-benchmark run --case /tmp/cases/my-case \\\n  --treatment /tmp/model-a.json --treatment /tmp/model-b.json \\\n  --attempts 1 --output /tmp/experiments/comparison-1\n```\n\nThe output directory must be new. `run` emits one JSON line per attempt; `report` emits a JSON array unless `--table` is selected. Runs are sequential in explicit treatment order; repetitions are independent. There is no hidden retry or resume. Each attempt writes an intent before dispatch and a result afterward. An unfinished intent is reported as `interrupted_or_running`, never automatically relaunched. Rerun explicitly into a new directory if authorized. SIGINT/SIGTERM cancel the active process group; a cancelled matrix does not dispatch remaining variants. The harness owns any deliberately detached descendants; this is not a general process sandbox.\n\nThe optional `setup: {\"executable\":\"...\",\"args\":[\"...\"]}` runs once in the new workspace within the same timeout. It owns dependency installation/local initialization. Benchmark does not silently install, borrow dependencies, repair setup or change controller registration. YYLO Pi and Workflow Runner must be initialized in their supported local topology by that setup if needed; a missing prerequisite is retained as an error/failure, not repaired. For example, the current Workflow Runner needs its own `.juno_task` and `.venv_juno`; Simple mode currently rejects orchestration. Do not route a benchmark workspace to your real controller to bypass that refusal.\n\nPi receives `--execution-envelope`, the selected model, file-backed prompt and a fresh session directory. Reserved model/prompt/resume/session arguments cannot be overridden. Requested and observed identities stay separate because selectors may be aliases. Captured reported cost is not an invoice; missing cost stays null. Most pre-envelope timeouts have unknown usage. The prompt file proves harness input, **not necessarily the final provider message**: YYLO preprocessing or harness instructions can transform it. Review actual delivery for experiments requiring literal fidelity.\n\n### Generic command adapter\n\nUse `harness: \"command\"` with an executable and argv. No shell interpolation occurs; explicitly select a shell if you need one. The process runs in the candidate workspace, receives the prompt on stdin, and receives this environment value:\n\n```text\nYYLO_BENCHMARK_REQUEST_JSON = { model, configuration, prompt, workflow }\n```\n\nThe adapter/command owns interpreting its model/configuration. With optional `workflow` configuration, the request also supplies `{path, variables, through}` for the retained YAML prefix; otherwise `workflow` is null. A custom workflow harness must consume that supplied projection rather than silently executing a different full workflow. Exit zero means execution completed, not that the task passed. Nonzero exit is failure; launch errors, cancellation and output overflow are errors. Each captured stream is bounded at 8 MiB; overflow terminates execution and marks the retained stream incomplete. Generic commands do not claim observed model identity or cost.\n\n### Workflows and selected steps\n\n```json\n{\n  \"name\": \"workflow-model-a\",\n  \"model\": \"provider/model-a\",\n  \"harness\": \"workflow_runner\",\n  \"executable\": \"python3\",\n  \"args\": [\"/absolute/installed/workflow_runner.sh\"],\n  \"timeout_ms\": 1800000,\n  \"configuration\": {},\n  \"workflow\": {\n    \"model_variable\": \"candidate_model\",\n    \"variables\": {\"reasoning\": \"medium\"},\n    \"through\": \"review\"\n  }\n}\n```\n\nOmit `through` for a complete workflow. The workflow must use the declared model variable for its agent calls; Benchmark does not rewrite commands or claim every arbitrary command used the requested model. Configuration is supplied through the same request environment; workflow-native variables are explicit. The workflow owns session reuse, ordering, dependencies and errors.\n\nFor a selected step, Benchmark retains a YAML projection containing the original prefix through that unique step ID, then passes it to the **existing Workflow Runner** with `--workflow`, `--run-root`, `--out-dir` and `--var`. It does not interpret commands or translate session state. Unsupported dependencies/variables remain runner errors. Resume-style `continue_from_step` is not a prefix experiment.\n\n```text\nsame case -> model/harness A: first -> review -> STOP\n          -> model/harness B: first -> review -> STOP\n```\n\nThis compares **the prefix through review**, not review independently. Model selection is fixed for each run; cross-harness compatibility is not guaranteed. The native runner's manifest is checked because it can report failed steps even when its process exits zero. Pi is a single-session adapter, not a workflow engine. A different workflow harness can be supplied as a generic command with the same `workflow` configuration: it receives the prefix path/variables through the request environment and owns execution/error reporting. Both native and custom workflow attempts are labelled `workflow_prefix` when a stop step is selected.\n\n## 3. Evaluate retained output independently\n\nA check profile uses a regular command receiving a JSON packet on stdin. It must return exactly a JSON assessment and exit zero; for legacy cases test failure is `verdict: \"fail\"`, whereas nonzero exit or malformed output is an **evaluator error**:\n\n```json\n{\n  \"name\": \"behavior-checks\",\n  \"kind\": \"check\",\n  \"command\": {\"executable\": \"python3\", \"args\": [\"/private/checks.py\"]},\n  \"timeout_ms\": 600000\n}\n```\n\n```json\n{\"verdict\":\"fail\",\"findings\":[\"Missing required behavior\"]}\n```\n\nA judge profile uses a Pi or command treatment:\n\n```json\n{\n  \"name\": \"judge-a\",\n  \"kind\": \"judge\",\n  \"rubric\": \"Assess correctness and missing validation separately.\",\n  \"max_packet_bytes\": 1000000,\n  \"judge\": {\n    \"name\": \"judge-a\",\n    \"model\": \"provider/judge-model\",\n    \"harness\": \"yylo_pi\",\n    \"executable\": \"yy\",\n    \"args\": [],\n    \"timeout_ms\": 600000\n  }\n}\n```\n\n```bash\nyylo-benchmark evaluate --attempt /tmp/experiments/comparison-1/1-1 \\\n  --evaluator /tmp/judge-a.json\n# Months later: a new judge or rubric; no original catalog prebinding needed.\nyylo-benchmark evaluate --attempt /tmp/experiments/comparison-1/1-1 \\\n  --evaluator /tmp/judge-b.json\n```\n\nEach evaluation has its own ID, specification and retained result, and runs in a fresh copy of retained candidate files. Candidates are never rerun. Packets include requirements, execution status, patch, response and rubric; final files are available in the evaluator workspace. Oversized packets are rejected, not silently truncated. Candidate identity metadata is omitted, but candidate-authored text may disclose it: do not claim perfect blinding. Treat candidate text as untrusted evidence.\n\nHuman profiles use `{\"name\":\"reviewer\",\"kind\":\"human\"}` plus `--assessment /tmp/assessment.json`. Checks/judges/humans may assess a retained failed attempt, but their verdict cannot change its execution status. Judge disagreements remain independent rows.\n\n### Checklist assessments and revisions\n\nWhen criteria are present, check packets include `checklist` and\n`checklist_origin`; judges receive the same packet plus a fixed evidence-based\ninstruction. Checks, judges and humans must return **only** the following shape,\ncovering every frozen ID exactly once:\n\n```json\n{\"criteria\":[{\"id\":\"TASK.greeting\",\"result\":\"pass\",\"evidence\":[\"node greet.cjs Ada: exit 0, expected stdout; greet.cjs:1\"]}]}\n```\n\nEvery result is `pass`, `fail` or `unknown`; evidence is a nonempty array of\nnonempty strings. Missing/extra/duplicate IDs, malformed evidence, or supplied\nverdicts/scores are evaluator errors. All criteria must be assessed in one record;\nseparate partial checks/judges are not automatically combined. A supplemental\nrubric cannot introduce unlisted criteria. Execution completion is not correctness.\n\nThe runner, not the judge, computes equal-weight **loss = failed / total** (0 is\nall pass, 1 is all fail). Any unknown makes the entire assessment unscored with\n`loss: null` and `insufficient_evidence`; evaluator errors yield null with\n`evaluation_error`. Verdict is derived: unknown first, then fail, otherwise pass.\nLow loss is partial quality, not a production acceptance decision. Evidence-based\njudgments are observations, not proof of universal correctness.\n\nWithout flags, `evaluate` inherits the frozen case checklist. To change it later:\n\n```bash\nyylo-benchmark evaluate --attempt /tmp/experiments/comparison-1/1-1 \\\n  --evaluator /tmp/judge-a.json --criteria /external/task-v2.yaml \\\n  --project-criteria /external/project.yaml\n```\n\nSupplying either criteria flag **replaces the whole inherited checklist** for the\nnew independent evaluation. Resupply both files to retain both. The result records\n`checklist_origin: evaluation`; reports mark `criteria_changed` when hashes differ.\nOriginal candidate intent and earlier evaluations are untouched. Explicit criteria\ncan also assess a legacy attempt retrospectively; never describe that contract as\nfrozen before execution. Meaningful revisions should use a new document version.\n\n## 4. Report and disqualify\n\n```bash\nyylo-benchmark report --root /tmp/experiments/comparison-1 --table\nyylo-benchmark report --root /tmp/experiments/comparison-1\nyylo-benchmark disqualify --attempt /tmp/experiments/comparison-1/1-1 \\\n  --reason 'Confirmed reference solution exposure'\n```\n\nJSON rows retain execution status, treatment, scope, each assessment/validity, timing, cost coverage, integrity errors and disqualification separately. Partial experiments are reportable, not presented as complete. Original results are not overwritten by disqualification. Checksums detect retained input/output/result drift; they are integrity checks, not protection from a malicious host rewriting everything. No automatic combined score, winner or capability inference is made.\n\n```text\ncase/case.json + source/\nexperiment/1-1/attempt.json + result.json + patch.diff\nexperiment/1-1/workspace/             # original attempt, preserved\nexperiment/1-1/output/                # retained result files\nexperiment/1-1/evaluations/<id>/       # independent assessment and copied workspace\n```\n\nChecklist JSON rows additionally expose `total`, `passed`, `failed`, `unknown`,\n`criteria_results`, `loss`, `score_status`, `score_reason`, `checklist_hash`,\n`evaluator_hash` and `comparison_key`. The evaluator identity hashes its specification\nand frozen built-in checklist judge instruction (null for checks/humans). The\ncomparison identity binds case, checklist, evaluator and declared candidate execution\nsettings, excluding model/name. It does not attest mutable tools or provider weights.\nDifferent keys must not be silently pooled. Disqualification makes report loss null\nwithout changing original assessments. Legacy no-checklist evaluations have null\nloss with `no_checklist`; missing evaluations remain unscored, never zero.\n\nCost and latency remain separate from loss; missing cost is null. Report scored,\nunknown/error/disqualified counts and repetition count alongside results. One run\nis exploratory; a few repeats are not a statistically reliable leaderboard. Do\nnot cherry-pick scored runs or a favorable judge. If aggregating externally, use\nequal task weights, show missing coverage, and preserve per-task rows. No combined\nquality/cost score, automatic winner, or ROI preference is imposed.\n\nKeep bundles, attempts and evaluations outside candidate source repositories. Keep credentials out of prompt/config files and retained logs. Retention and cleanup are operator decisions, not automatic runner actions.\n\n## Standalone skill ownership\n\nThe canonical agent skill is **benchmark-yylo** in the dedicated\n[yylo-dev/yylo-skills repository](https://github.com/yylo-dev/yylo-skills/tree/main/skills/benchmark-yylo).\nIt owns the checklist judging guide and reusable examples and is **not shipped inside this npm package**.\nThere is one independently released skill, not a second project-local checklist\nskill. Follow that repository's reviewed installation process; compatible YYLO\ninstallations support explicit `yy skills install` / `yy skills update`.\n`yy scripts update` does not acquire skills. Do not overwrite customized or\nunrecorded agent copies. No skill is installed or activated by a Benchmark build.\n\nDiscover runtime support with `yylo-benchmark case create --help` and\n`yylo-benchmark evaluate --help`: both must show the criteria flags. A source\nversion alone does not establish installed support. Runtime and skills releases\nare independent; source delivery and packed verification are **not publication or\nactivation**. Do not silently upgrade or switch binaries. The standalone skill\ncovers reconstruction, operator approval, controls, evidence, revisions and limits.\nIts example assessment is synthetic, not reusable evidence about your own attempt.\nPublic names are YYLO Benchmark and `yylo-benchmark`; `juno-benchmark` is only the\ninternal monorepo directory.\n\n## Development\n\n```bash\nnpm ci\nnpm test\nnpm run typecheck\nnpm run build\nnpm pack --ignore-scripts --pack-destination /tmp\nnode scripts/verify-v2-packed-acceptance.mjs /tmp/yylo-benchmark-VERSION.tgz\n# Explicit source integration check with a separately reviewed skills checkout:\nnode scripts/verify-standalone-skill-examples.mjs /path/to/yylo-skills/skills/benchmark-yylo\n```\n\nThe retained script filename is historical; it verifies the thin v3 CLI, checklist\nscoring and the absence of bundled skills from a local tarball with synthetic\ncommands only. Its test inputs are self-contained, not canonical skill assets.\nIt installs into a fresh temporary consumer using offline npm. Its cache must\ncontain metadata and runtime dependency versions selected by the package ranges;\n`npm ci` alone may not populate all of them. Prepare that cache separately with\nnetwork permission if needed; `ENOTCACHED` is a missing prerequisite, not a model\nfailure. Scratch evidence is retained. No publication, global installation or live\nmodel calls occur. Package version changes/publication remain a separate maintainer\naction.\n","readmeFilename":"README.md"}