{"_id":"@metaharness/darwin","_rev":"29-58ef49b8d952341b5cce9b4b17b89870","name":"@metaharness/darwin","dist-tags":{"latest":"0.10.3"},"versions":{"0.1.0":{"name":"@metaharness/darwin","version":"0.1.0","keywords":["agent-harness","darwin-mode","self-improvement","evolutionary-search","metaharness","dgm","archive","sandbox","benchmark"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.1.0","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"4404239d270bc0062de32aa8412fde809ea9bb76","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.1.0.tgz","fileCount":100,"integrity":"sha512-jZJd4J54OhlOAzYM3MEV5qWWjygWnP05qoA+iy/n0GTpdgkaZLKXR6TagYOHP4s5HLShC2gZAZMBXpsW4VAl3Q==","signatures":[{"sig":"MEQCIFtOvq7Mj2WS+9RQRGu6wFqfbwTRff6+YLZ0uSJIQ141AiBWKaaq1iCO6JeQfNus/GzhF3jpK7rw7cMnVCrJksb3pA==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":266318},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"56e19a4808fdd28744f8e94eab58d50807e41df5","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Darwin Mode (ADR-070…075) — bounded, empirical, population-based self-improvement of an agent harness. Generate child harness variants, sandbox-score them, archive the lineage, and promote only measured, safe wins. The model is frozen; the harness evolves","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.1.0_1781749002851_0.8084539478087898","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@metaharness/darwin","version":"0.2.0","keywords":["agent-harness","darwin-mode","self-improvement","evolutionary-search","metaharness","dgm","archive","sandbox","benchmark"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.2.0","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"f203440fe6e756dd90eb95a76c5d94b5e210eaf5","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.2.0.tgz","fileCount":132,"integrity":"sha512-N5s5lsKRKUFyLPsPqFXzZMTnsakddiq7ivs5qTuc0o4+Dm73E9oY/KCzx/2k0N/0h5C/lnWXak9rD8Ctfg9XBw==","signatures":[{"sig":"MEQCIDOIvxgVH1YSxdjCPHWzX0EqYcd44N+secU34koPE0dXAiAmD0bKFBFlW7X4vG7zttMX043E/48HVa7OT5orKUmNqg==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":413437},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"5de5834999a7ad6164195545a91015e4ab5380a4","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Darwin Mode (ADR-070…075) — bounded, empirical, population-based self-improvement of an agent harness. Generate child harness variants, sandbox-score them, archive the lineage, and promote only measured, safe wins. The model is frozen; the harness evolves","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.2.0_1781816376480_0.7044010895507635","host":"s3://npm-registry-packages-npm-production"}},"0.2.1":{"name":"@metaharness/darwin","version":"0.2.1","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.2.1","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"c2ac6516c4ff01e5a64c2baefcc0b21082dd2806","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.2.1.tgz","fileCount":132,"integrity":"sha512-tYodc2x0/qqxc/6n6nAp5E40lZBh6rKq8iV0kQy6+1i4P3TOFAP21w6KDEVD99XENdukzXvKN1DBRWLMZC2MhQ==","signatures":[{"sig":"MEYCIQClhIdpeOcZlKDDHZj2anbOjqr5QyXUvZWUqsRPnYyTjQIhAO2GNHJ770L+ygAr11D7WbqpIEMFK8m6+9b/NqMuCn7v","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":415741},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"8d38baab6ee00fa87e6c8a42c215059ca23cb42e","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"An LLM supercharger and cost optimizer: keep your model frozen and evolve the harness around it so a cheap model performs like an expensive one — measurably better, far cheaper. Mutate→sandbox→score→archive; promote only safe, measured wins. Validated on ","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.2.1_1781829327079_0.8093003968214532","host":"s3://npm-registry-packages-npm-production"}},"0.2.2":{"name":"@metaharness/darwin","version":"0.2.2","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.2.2","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"e6b6cb1e6deddcac547c4ca99b29da258b071974","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.2.2.tgz","fileCount":137,"integrity":"sha512-MK3H2Utuv5wjpnAh9h//IRSAX1QauLl6PtG4YFeB9CzfwUwxVWd1HDiRwBoEdim8kS5iP+997cX33hOMtX8Raw==","signatures":[{"sig":"MEUCIQCiySc34Tex8Ws8Cg1+wgnjJ/P/mA3xr1INOehuXFQOGgIgZeIFdQVLnOgGdOHopHbxJnnL+e3ymI6y7nNzYuTyqh0=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":427627},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"fd20f31b2c5b3d3946de6b695da3745927501bb1","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"An LLM supercharger and cost optimizer: freeze your model, evolve the harness around it. Measured on full SWE-bench Lite (300, official swebench Docker): open-loop 7.7% -> +repair 15.3% -> +Barbarian&Scholar hybrid 33.3%, at ~$0.01-$0.34/instance (vs $1-2","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.2.2_1781892239256_0.9574507130224685","host":"s3://npm-registry-packages-npm-production"}},"0.2.3":{"name":"@metaharness/darwin","version":"0.2.3","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.2.3","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"86a3d8f582863af6c0f4ab643055eb09e3af3478","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.2.3.tgz","fileCount":138,"integrity":"sha512-6qJMawsUWbKPH+SwtmYihSM/SkxgLF3F/L64Xt7wwly4F2yhgH/bf4KgY0rpgtjFxsYBZXVOhcDGvgTOHTJF8A==","signatures":[{"sig":"MEUCIHEtyqaFf41ZIOD4kO3ZCM+4lbWR+R5zoGJ8csPisyCtAiEA+/n5C1da7eHfKAgoLLTOP5x30R8EjgXGE6OpvrgeiA8=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":432268},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"311a7c094f2507ce409c042c7b6b71bcecfa44ad","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"An LLM supercharger and cost optimizer: freeze your model, evolve the harness around it. Measured on full SWE-bench Lite (300, official swebench Docker): open-loop 7.7% -> +repair 15.3% -> +Barbarian&Scholar hybrid 33.3%, at ~$0.01-$0.34/instance (vs $1-2","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.2.3_1781892998381_0.7061818605897883","host":"s3://npm-registry-packages-npm-production"}},"0.2.4":{"name":"@metaharness/darwin","version":"0.2.4","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.2.4","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"e4830294d8f082e3ef04ff7685a50ac3504b9dc7","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.2.4.tgz","fileCount":138,"integrity":"sha512-DymSMz3HK+AtBdI2TO55VZVwK2LEF76+yrTmXPwkjtZOCkk5UOu9aIvKhhSKgexakc65AlMc2oW6QmmC4DcPpg==","signatures":[{"sig":"MEYCIQCayxbaHLmxqcJ+5C+v7SeVEsPcfO6hZbuiaFj1uFgAYAIhAIvuVLjFYXcbU6qPVmQR8MC3nqR4SkAo7qgy6wuXenU2","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":432658},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"161a2bd77afc7de52118c99b0d0c97ac24016849","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"An LLM supercharger and cost optimizer: freeze the model class, evolve the harness. Measured on full SWE-bench Lite (300, official swebench Docker): cheap-model repair 15.3% (deepseek-V3) -> 29.3% (deepseek-v4-pro) -> 33.3% Barbarian&Scholar hybrid, ~$0.0","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.2.4_1781911151065_0.26556061520779783","host":"s3://npm-registry-packages-npm-production"}},"0.2.5":{"name":"@metaharness/darwin","version":"0.2.5","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.2.5","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"14d214032125206b2792b2fe99e84e57be638127","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.2.5.tgz","fileCount":138,"integrity":"sha512-0EhGll20iiv/qfFO20ZcFdsJw0Ev2rE8xNJB1p8IeEFDYzJPO6RgapkhG7r7CYDtJHBbE2cR7v748iI7c1bTRA==","signatures":[{"sig":"MEUCIGgEvf0zlqebIzwUvnLhB8KKSeBSwm+UWpZOFEPe9MnrAiEAmKdgK3WVRntqHWr0XC8x/Z/OP0PRmFW+Ju6bwKzdorU=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":433035},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"161cfd45859ea9484603b42bf6b2b3063e886ad5","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"An LLM supercharger and cost optimizer: freeze the model class, evolve the harness. Measured on full SWE-bench Lite (300, official swebench Docker): 7.7% open-loop -> 15.3% +repair -> 29.3% (deepseek-v4-pro base) -> 40.3% v4-pro+frontier-tail hybrid, ~$0.","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.2.5_1781920069430_0.23036702824967747","host":"s3://npm-registry-packages-npm-production"}},"0.2.6":{"name":"@metaharness/darwin","version":"0.2.6","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.2.6","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"154614bbfcc96c4af291f63d91e916ae9c33b006","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.2.6.tgz","fileCount":138,"integrity":"sha512-XF02ViDNB9KLb/45Hfqc0eMwLqR9MSVO+2TKbYDP5jFbAmSYx/slaQalw9uPOmltPFQgcEMyHUVNngJiEFLeGg==","signatures":[{"sig":"MEYCIQD1YKG3t9fMp+gDgMuW8Z74uhWBif96cEykNSJ+CdareAIhALcA2mv/Di2ON1TJtU9npuE81rNq0qHQfZfW1ESvFPyF","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":433431},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"e45fd5ffee60ba44dc929a4a10de2729319b92ac","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"An LLM supercharger and cost optimizer: freeze the model class, evolve the harness. Measured on full SWE-bench Lite (300, official swebench Docker, verified): 7.7% open-loop -> 15.3% +repair -> 29.3% (v4-pro base) -> 40.3% 2-tier -> 58.3% 3-tier cheap->fr","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.2.6_1781927794797_0.7916153420845093","host":"s3://npm-registry-packages-npm-production"}},"0.2.7":{"name":"@metaharness/darwin","version":"0.2.7","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.2.7","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"c71adf5a26693ba53b135411746edfc204052daa","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.2.7.tgz","fileCount":138,"integrity":"sha512-8RPKoKFyLoVhvpce8UfV8xXrx1gYRzK1iT4VpqxF0GmahF8WtnXaRyM2Oxs9v7ycjnpE7Toqm/YOzGjFcHHCNw==","signatures":[{"sig":"MEYCIQCaC8i+ILXXoInw9IC3lwMmCC0wIU9PorYw52Ixf1siuQIhAIQFuBY4VlkYFnBGZeyHqSipc8oKicmcnDdhSuUfsEFK","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":434763},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"d879936c30ea7f6645ea180a583d544c14438cc7","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"An LLM supercharger and cost optimizer: freeze the model class, evolve the harness. Measured on full SWE-bench Lite (300, official swebench Docker, verified): 7.7% open-loop -> 15.3% +repair -> 29.3% (v4-pro base) -> 40.3% 2-tier -> 58.3% 3-tier cheap->fr","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.2.7_1781986749407_0.767473343092629","host":"s3://npm-registry-packages-npm-production"}},"0.2.8":{"name":"@metaharness/darwin","version":"0.2.8","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.2.8","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"0afb19a17dacdf6a9f69a7004ab7cf426dfc4beb","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.2.8.tgz","fileCount":138,"integrity":"sha512-B8tF7IrrSxwKS6fEPEL6N2Juth9WWn+hppLUtUYPTJ2vcHzzZPIg2cS5T9qTyNNuANlTSWnQHnvzlfvYdGNfeQ==","signatures":[{"sig":"MEYCIQDOtJUGBp2NWUwWRS/ey4vwju2h6alcHgtNOHtGfa+EngIhAMoKVDYA52lsqNK+g+QeGwnlCMyarz9GL+r9G0P86Dnp","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":435116},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"246da218d7d4720500551a2eac1b4c9e7a77c6a8","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"An LLM supercharger and cost optimizer: freeze the model class, evolve the harness. Measured on full SWE-bench Lite (300, official swebench Docker, verified): 7.7% open-loop -> 15.3% +repair -> 29.3% (v4-pro base) -> 40.3% 2-tier -> 58.3% 3-tier cheap->fr","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.2.8_1782025502378_0.6222069605480876","host":"s3://npm-registry-packages-npm-production"}},"0.3.0":{"name":"@metaharness/darwin","version":"0.3.0","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.3.0","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"f6aae2032e0620f38ad009b8f57d440be9063d04","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.3.0.tgz","fileCount":230,"integrity":"sha512-WsfyBrLXswllvIyWbz59RDzzj5fMoKKviwKNCnQsbHR0661MWI+y9gdjayFd9l0CD0W2wdKkUXnC8mfUU0ziIQ==","signatures":[{"sig":"MEQCIE3sw0afd6gMrf5M+M9/KePMcvlSt7NFSBCDJhfN9E5wAiACX7FPqe4Lo8RNMHtWKyKUXfgfkUIQ+M39FCZhhOGxiA==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":834204},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"51372ba594e061404b249de1d96e300564f547f5","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"An LLM supercharger and cost optimizer: freeze the model class, evolve the harness. Measured on full SWE-bench Lite (300, official swebench Docker, verified): 7.7% open-loop -> 15.3% +repair -> 29.3% (v4-pro base) -> 40.3% 2-tier -> 58.3% 3-tier cheap->fr","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.3.0_1782078157816_0.9438387879006636","host":"s3://npm-registry-packages-npm-production"}},"0.3.1":{"name":"@metaharness/darwin","version":"0.3.1","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.3.1","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"d4844875cc9d03dff72abbbe37ee241ac32c380c","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.3.1.tgz","fileCount":230,"integrity":"sha512-gNQQQiSsbcsSXG9g1C7IJoX50Ozw1nIg6Nez40OVKo5PSvlGjDKElBjFng7M7O7bIrlBo0fS9lcP5hjEQXi+iQ==","signatures":[{"sig":"MEUCIQC0nfC0PMRbA12EkObsYD8wN9ga3JX3EKe7Rt9fRYsCigIgMIAfSS/29dtopHKLUDIwL2b5P4RspOrcjWPWF0sFi+4=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":836182},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"c8cdc70ec23b9159d7758b0688ad9b66b8800869","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench Lite code-repair — 7.7% open-loop -> 58.3% via cheap->frontier tiering (official swebench Docker, verified), ~$0.01-$0.74/instance vs $1-20 for frontier agents; (2) Darwin Shie","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.3.1_1782079881509_0.8230841240775957","host":"s3://npm-registry-packages-npm-production"}},"0.4.0":{"name":"@metaharness/darwin","version":"0.4.0","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.4.0","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"bd8a619ec407db382ede6bed197129940cfb7e18","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.4.0.tgz","fileCount":230,"integrity":"sha512-lgMEX+3U5kqnVDZyAEjks3oTbVX1bVh0PRNtBYUVesqUroBBfkz1l9kjrZaTAd9UV6fsz77YK2GbsqAqZop+uA==","signatures":[{"sig":"MEUCIQDZ5CswTlZjwx3IFo6bvUHLD0UnWegGbbwGMOBI7GzmggIgPqTb4yXaxlhftPCfKruNEN/XD4j4mW6qz9EYLTu6QmI=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":838945},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"4ddcaf8240c031118169b95f9569bddbebba348d","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench Lite code-repair — 7.7% open-loop -> 58.3% via cheap->frontier tiering (official swebench Docker, verified), ~$0.01-$0.74/instance vs $1-20 for frontier agents; (2) Darwin Shie","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.4.0_1782155627395_0.38650108545185113","host":"s3://npm-registry-packages-npm-production"}},"0.5.0":{"name":"@metaharness/darwin","version":"0.5.0","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.5.0","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"7d4014be7016a5d1821b42779584c1318bb037e4","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.5.0.tgz","fileCount":230,"integrity":"sha512-CtfGWvRGJacuxtpBQ1eZt5bGX77aaoj6kCWkGCYcT0BhF2rZk2a21sebBiQtFAsPTUKzzVoJNrauF1nVgc9hZw==","signatures":[{"sig":"MEUCIQCmGxGvXITneXknbMXgjb18mPH/kIwmylai/YFJA6SDRgIgBcV5THefFsGEowoaWKwUZA3zmXfDGzgSVoOohSUM/m8=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":841039},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"6333b72b43675a3caab16d65290671d35719136e","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench Lite code-repair — 7.7% open-loop -> 58.3% via cheap->frontier tiering (official swebench Docker, verified), ~$0.01-$0.74/instance vs $1-20 for frontier agents; (2) Darwin Shie","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.5.0_1782161511898_0.3382049103444351","host":"s3://npm-registry-packages-npm-production"}},"0.6.0":{"name":"@metaharness/darwin","version":"0.6.0","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.6.0","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"38915123a86a5e723f75bf6a6687828bf2ec6ce6","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.6.0.tgz","fileCount":230,"integrity":"sha512-rPqy/j4p5unXeAqluQR0GAlI4PanyCffSWi0DXPZtSYTspUduzGZVaCPt51jKBoiT9OEdKzALUqza1QX1RynjQ==","signatures":[{"sig":"MEYCIQDeXGk42vVTcFKxKW2nSH/iLaz7flYA6F5R7V+KTo5YFgIhAKUaUdknG4nbBOoX7/IC8r/tIyO9DVejDLhSj/Ih337Q","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":849873},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"b3d3838dd9eb75729482283a42dff4e19d16de72","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench Lite code-repair — 7.7% open-loop -> 58.3% via cheap->frontier tiering (official swebench Docker, verified), ~$0.01-$0.74/instance vs $1-20 for frontier agents; (2) Darwin Shie","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.6.0_1782208808789_0.6582990176382062","host":"s3://npm-registry-packages-npm-production"}},"0.7.0":{"name":"@metaharness/darwin","version":"0.7.0","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.7.0","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"10fb7e297b98b03ab522a248d473988d4583782b","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.7.0.tgz","fileCount":230,"integrity":"sha512-ApndPhj982QXPN2ARedxki+iA1xiyk9hbLXu8tJo9kEcjw37T+/2wrZLgvaTfLSHfGmg87qbtrV8dCZT/20/oQ==","signatures":[{"sig":"MEQCIGWeIMjbRsdEAQo3MiY5Y6X5NdaJZP9S/qNqPnuhS6fAAiApHqSDIy2w2UC6Aa+m9DBgyp3zVDynNMs7DXB62Q5fYQ==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":863882},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"7ba1e7f02b8330d52746424f76518d34dad72cba","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench Lite code-repair — 7.7% open-loop -> 58.3% via cheap->frontier tiering (official swebench Docker, verified), ~$0.01-$0.74/instance vs $1-20 for frontier agents; (2) Darwin Shie","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.7.0_1782257071860_0.30041014224216434","host":"s3://npm-registry-packages-npm-production"}},"0.7.1":{"name":"@metaharness/darwin","version":"0.7.1","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.7.1","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"646d853777bd391d9a917d54ce5a23d948f8cbd7","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.7.1.tgz","fileCount":234,"integrity":"sha512-J9Y9kg5qpRgI8j/c2uwOTuFt38l0dIYdJ+tEKZFXQtyOiYPBt+vgymDMZaKjye08WR1j4Ow0M0iGNLnUoPXeFQ==","signatures":[{"sig":"MEQCICQmoJCPZlH6fAbfE9HmJNSBuq9wqJg1UO56hBfmfWAxAiBkUnbVwVJDWk9VUd0tpNoCcDIHlgczyK/NGufKyPbp2A==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":929866},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"}},"gitHead":"fe715f6131b2e125cc03591dca27ba9c31724442","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.7.1_1782507780912_0.8800872703265181","host":"s3://npm-registry-packages-npm-production"}},"0.8.0":{"name":"@metaharness/darwin","version":"0.8.0","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.8.0","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"d133d247b8c247233bcb3291f1a7b0bfea2572c7","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.8.0.tgz","fileCount":256,"integrity":"sha512-Pgefr/es0Btofh7GxQrOAg/i43ZKcLUfeD9rndOAkpA8s3ZYohSmfLerJLNsGOOKc2eTvmmauljl8QEVmKC2dw==","signatures":[{"sig":"MEQCIGb558j+biDRgRuRggjV0CxWFUzqeTRU+OZUXbj1oO6BAiBcx4ztYgCu4R8sovL+QhvI35HnhF7qh/B0S0bYaREevQ==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1105610},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./gepa":{"types":"./dist/gepa/index.d.ts","import":"./dist/gepa/index.js"}},"scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.8.0_1783096266149_0.5228726760717284","host":"s3://npm-registry-packages-npm-production"}},"0.8.1":{"name":"@metaharness/darwin","version":"0.8.1","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.8.1","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"f66346e2a6ef1a9260c3de80137db26413bd7736","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.8.1.tgz","fileCount":256,"integrity":"sha512-QDf5HQRNcMq9RcHm0Q5wFzmaVEoHCPIt770BTocQXzz2rlCH2k7TWBMEJTj1x3Wo9OHTPoS/5ia/lbi+y8jmCQ==","signatures":[{"sig":"MEUCIQDEG7feCbjCzo+0i2Z/XBK2XYYekEicC1ciLrmdubJncQIgL2bmV2qe9J8E7X5s2R9aarvpA2sG3488ogXaWdPTXfU=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1109110},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./gepa":{"types":"./dist/gepa/index.d.ts","import":"./dist/gepa/index.js"}},"gitHead":"fa4b28de34aea3a183ddcddb9faf8f4d52e7cc50","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:clade":"npm run build && node bench/selection/clade-throughput.mjs","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0","@metaharness/flywheel":"^0.1.1"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.8.1_1785688891022_0.5023231074785064","host":"s3://npm-registry-packages-npm-production"}},"0.8.2":{"name":"@metaharness/darwin","version":"0.8.2","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.8.2","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"22afe32610f182797f9fafce23af7ddf3199df3f","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.8.2.tgz","fileCount":256,"integrity":"sha512-gMynjSiZ47UDTOTkwq3PoszceHdGog3I+Ahcts4xfXs9KhR5QquBG80Ib/OD9/IPo51fenbW1sazjwwwgLBeMA==","signatures":[{"sig":"MEUCIQC+GNtIjbclokh90jHMqRLHVv17UzZBTOdc54airs5HNAIgYSLjEHTrgIgCjVVZlVaN/1QYS2S/ugLJHQnN3G5LP8w=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1115792},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./gepa":{"types":"./dist/gepa/index.d.ts","import":"./dist/gepa/index.js"}},"scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:clade":"npm run build && node bench/selection/clade-throughput.mjs","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0","@metaharness/flywheel":"^0.1.1"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.8.2_1786229032272_0.25536835478369424","host":"s3://npm-registry-packages-npm-production"}},"0.8.3":{"name":"@metaharness/darwin","version":"0.8.3","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.8.3","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"c94b445bddc0c39ac65ef02fccd8b71669dad77c","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.8.3.tgz","fileCount":260,"integrity":"sha512-v5Ph7y9U6nqfvWajOjJiSuQDjPwvCKwOui793cSGUxIYUcfMoVoMzDfEDxOczFU61z0UUd7lD0I1b2zl+eYnag==","signatures":[{"sig":"MEUCIQCHEZTsBv4zMJlFP6ENREGfadNn3U/WGQYjS6+cSW4sgQIgdsBKMUHlrHmT57DIrYCraJ+jz1HX5aXYnriOBJSBoo8=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1130172},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./gepa":{"types":"./dist/gepa/index.d.ts","import":"./dist/gepa/index.js"}},"scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:clade":"npm run build && node bench/selection/clade-throughput.mjs","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0","@metaharness/flywheel":"^0.1.1"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.8.3_1786234639032_0.42541686794965594","host":"s3://npm-registry-packages-npm-production"}},"0.9.0":{"name":"@metaharness/darwin","version":"0.9.0","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.9.0","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"8aff2ef49aae95cff26c7a6240d667db21fb6bdd","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.9.0.tgz","fileCount":260,"integrity":"sha512-X/u0sI4QNFi/YUhcQwZ2hcmthgc3H3n4PdPoKMbAE/y+rUWOQCS9oc1wqCcUysBN5K0jHV7U+3QApdMFvwKvRQ==","signatures":[{"sig":"MEQCIGxXNfosqkghU+NaRs8br2PJlg8GgaCx/7xb1yWLjujNAiBPZUyQzUCXsbNAM5KA6YItGMM1QUw+KpKdkQ8J6mJj2A==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1134286},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./gepa":{"types":"./dist/gepa/index.d.ts","import":"./dist/gepa/index.js"}},"scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:clade":"npm run build && node bench/selection/clade-throughput.mjs","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0","@metaharness/flywheel":"^0.1.1"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.9.0_1786381448562_0.6418176653874788","host":"s3://npm-registry-packages-npm-production"}},"0.9.1":{"name":"@metaharness/darwin","version":"0.9.1","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.9.1","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"b1fd17f875da4419821ec54d2a52c07f02e91363","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.9.1.tgz","fileCount":260,"integrity":"sha512-7vH/Dpv9TBs/v63CVjkUTzkUn1kgGmES36N35l8D8gOEnvDKN40FznS3vSmQLdTiunE7r7HYf3XVNaIHD5Tcww==","signatures":[{"sig":"MEUCIQCbiov+pTShhRUixOXBtQvwZ8iS9CtVYwM8jUY/KowpSQIgHXFh/dXYCDpWZTNITxPgPq0M9xmDaMF/vo/5fCqOEnU=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1139726},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./gepa":{"types":"./dist/gepa/index.d.ts","import":"./dist/gepa/index.js"}},"scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:clade":"npm run build && node bench/selection/clade-throughput.mjs","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0","@metaharness/flywheel":"^0.1.1"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.9.1_1786469007932_0.6530090449028774","host":"s3://npm-registry-packages-npm-production"}},"0.9.2":{"name":"@metaharness/darwin","version":"0.9.2","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.9.2","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"b8d4a231d53b8968155427f52b3ea18153e5b1c2","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.9.2.tgz","fileCount":260,"integrity":"sha512-UaAikvfaowvXO97ivNubXRdXEDbvBl28XCDbyTv979vbzPiOoHWdPpS3XPHq/D9VG+wCtK9BpS3PXm340EkYSQ==","signatures":[{"sig":"MEUCIQDCsI4vXvsnzr+nbMHURsmRHGTGqv7sM6KKvWBUW/wW7gIgTlpBno49hs/nXWubTv6qiwMt+Os3K9NUY4Rf/U4Oo0Y=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1141117},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./gepa":{"types":"./dist/gepa/index.d.ts","import":"./dist/gepa/index.js"}},"gitHead":"5453c8c990824b54e05f289774e5a8b2cea0a32e","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:clade":"npm run build && node bench/selection/clade-throughput.mjs","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0","@metaharness/flywheel":"^0.1.1"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.9.2_1786801827180_0.3755371090856081","host":"s3://npm-registry-packages-npm-production"}},"0.9.3":{"name":"@metaharness/darwin","version":"0.9.3","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.9.3","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"4547161afec66b5e4ceb5ac9692ec62aa62dcd5c","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.9.3.tgz","fileCount":260,"integrity":"sha512-V+AhQvj9ijR8OK9TvogSngtz47q8pHPjMm1mWMoDUk1JaRKz88oJu/sPUQ5BApCIgeSWEKo/bzrFSV4Krb/3Fg==","signatures":[{"sig":"MEQCIDD6hK6j+CAqtvnzrvmFaZTkgl3PN+WFnwTvcUzGgjeKAiA9qbPIakGw2ekbGton0/EuB5lyG7hT4YjPUx+5NU1UMA==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1146351},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./gepa":{"types":"./dist/gepa/index.d.ts","import":"./dist/gepa/index.js"}},"scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:clade":"npm run build && node bench/selection/clade-throughput.mjs","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.7","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"_nodeVersion":"22.22.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0","@metaharness/flywheel":"^0.1.1"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.9.3_1787352125508_0.22043674607349173","host":"s3://npm-registry-packages-npm-production"}},"0.10.0":{"name":"@metaharness/darwin","version":"0.10.0","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.10.0","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"d777105de8de6c7fb7d372373d4755680a1a911b","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.10.0.tgz","fileCount":280,"integrity":"sha512-TE8ugtpIPw0GzHjnJIoPlxA/AhkKfqKUGvawBr0Pdl/8ff/yX1YD749Z8HZ9bn33QyBBf4MkNwhi9g+BUSH34g==","signatures":[{"sig":"MEUCIQCL34UXTkvTjqqvszdY25LLFX5d6zwRy0JaDiy/AsQOjAIgfvngqUjYEygDvapQqJwyKgq5gRCtvZRobVanP9kYC6A=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1214010},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./gepa":{"types":"./dist/gepa/index.d.ts","import":"./dist/gepa/index.js"}},"scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:clade":"npm run build && node bench/selection/clade-throughput.mjs","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.8","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"_nodeVersion":"22.23.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0","@metaharness/flywheel":"^0.1.1"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.10.0_1788282197678_0.8141008490247255","host":"s3://npm-registry-packages-npm-production"}},"0.10.1":{"name":"@metaharness/darwin","version":"0.10.1","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.10.1","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"37e6a4377d3982cd20cef88205fb240484b6557b","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.10.1.tgz","fileCount":280,"integrity":"sha512-AWmyvvp0d+BlEN+2Dev/pxhG4Kdl/Pc2m905r2Hk0p0Mt5nv/+5Xv5Jj+oovHrmlXseM6KgYMAco2t7bh6QRpA==","signatures":[{"sig":"MEYCIQDHEowvI9Bo4NqtLmjaF7Kd0nlxvSsOWYNybi+w6zF+PAIhALgqlRGEwZtZyGhbMH3UQGBUNCZrC6qgw7ucduwwQ+HM","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1215644},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./gepa":{"types":"./dist/gepa/index.d.ts","import":"./dist/gepa/index.js"}},"scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:clade":"npm run build && node bench/selection/clade-throughput.mjs","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.8","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"_nodeVersion":"22.23.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0","@metaharness/flywheel":"^0.1.1"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.10.1_1788292884956_0.18138700789429874","host":"s3://npm-registry-packages-npm-production"}},"0.10.2":{"name":"@metaharness/darwin","version":"0.10.2","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","_id":"@metaharness/darwin@0.10.2","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"homepage":"https://github.com/ruvnet/agent-harness-generator","bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"bin":{"metaharness-darwin":"dist/cli.js"},"dist":{"shasum":"c72f368b4e44d5a469aa6605b101184b04d645eb","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.10.2.tgz","fileCount":280,"integrity":"sha512-Glczy9YJDLf5x4HlfVuQVbZuPuue45K+8ohfLixZPJ18oc0Q8RR+BcsopHUQsL4hBkvs3YPXUyZ16piDJQp4gw==","signatures":[{"sig":"MEUCIAszsUUzKfrIPB0WGV1fV47ZpYYETNRtZtjJg0G9jylVAiEArV9FMzcytHtceYnnUSAWWW4pJ1uqXzwhkkg49XCSWXo=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1217992},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./gepa":{"types":"./dist/gepa/index.d.ts","import":"./dist/gepa/index.js"}},"scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:clade":"npm run build && node bench/selection/clade-throughput.mjs","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.8","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"_nodeVersion":"22.23.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0","@metaharness/flywheel":"^0.1.1"},"_npmOperationalInternal":{"tmp":"tmp/darwin_0.10.2_1788312295817_0.996899084796123","host":"s3://npm-registry-packages-npm-production"}},"0.10.3":{"_id":"@metaharness/darwin@0.10.3","bin":{"metaharness-darwin":"dist/cli.js"},"bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"dist":{"shasum":"9f4256acdca100b94ef856270a96b38255549b3d","tarball":"https://registry.npmjs.org/@metaharness/darwin/-/darwin-0.10.3.tgz","fileCount":284,"integrity":"sha512-xqFWTqG8tHk2Gwts07EJgU2E7WANkhjwPc5Dk9nG3tbW8Hqb+O7Ut5ixp8+HoBlod8EgNigMT9MSrOpCpHSXEQ==","signatures":[{"sig":"MEQCICM80V5PSUl2pgjP7w5npn+NUTaFPtlZFJM50bx+paeCAiAP1qn5TKU2ilXPyBcs1kDTNX4vCo29fWbbKNgACIlcmg==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"},{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEYCIQD7ajhZ9qp75P4KVjKnmkDITC2qNpF6/m/FQ4tbHmVA2AIhAMYKhVQ1RBhlrgSYUL/q6iG0YHKAyZs53ikH6jiHmWV+"}],"unpackedSize":1244663},"main":"./dist/index.js","name":"@metaharness/darwin","type":"module","types":"./dist/index.d.ts","author":{"name":"rUv","email":"ruv@ruv.net"},"engines":{"node":">=20.0.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./gepa":{"types":"./dist/gepa/index.d.ts","import":"./dist/gepa/index.js"}},"license":"MIT","scripts":{"lint":"tsc --noEmit","test":"vitest run","build":"tsc","bench:clade":"npm run build && node bench/selection/clade-throughput.mjs","bench:shield":"npm run build && node bench/security/darwin-shield-bench.mjs"},"version":"0.10.3","_npmUser":{"name":"ruvnet","email":"ruv@ruv.net"},"homepage":"https://github.com/ruvnet/agent-harness-generator","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"_npmVersion":"10.9.8","description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","directories":{},"maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"_nodeVersion":"22.23.2","publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"vitest":"^2.0.0","typescript":"^5.4.0","@metaharness/flywheel":"^0.1.1"},"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/darwin_0.10.3_1790437810529_0.2308368643568206"}}},"time":{"created":"2026-06-18T02:16:42.669Z","modified":"2026-09-26T15:50:11.043Z","0.1.0":"2026-06-18T02:16:43.084Z","0.2.0":"2026-06-18T20:59:36.629Z","0.2.1":"2026-06-19T00:35:27.256Z","0.2.2":"2026-06-19T18:03:59.461Z","0.2.3":"2026-06-19T18:16:38.543Z","0.2.4":"2026-06-19T23:19:11.203Z","0.2.5":"2026-06-20T01:47:49.572Z","0.2.6":"2026-06-20T03:56:34.940Z","0.2.7":"2026-06-20T20:19:09.550Z","0.2.8":"2026-06-21T07:05:02.501Z","0.3.0":"2026-06-21T21:42:37.975Z","0.3.1":"2026-06-21T22:11:21.662Z","0.4.0":"2026-06-22T19:13:47.540Z","0.5.0":"2026-06-22T20:51:52.102Z","0.6.0":"2026-06-23T10:00:08.971Z","0.7.0":"2026-06-23T23:24:32.116Z","0.7.1":"2026-06-26T21:03:01.079Z","0.8.0":"2026-07-03T16:31:06.288Z","0.8.1":"2026-08-02T16:41:31.170Z","0.8.2":"2026-08-08T22:43:52.425Z","0.8.3":"2026-08-09T00:17:19.196Z","0.9.0":"2026-08-10T17:04:08.759Z","0.9.1":"2026-08-11T17:23:28.091Z","0.9.2":"2026-08-15T13:50:27.335Z","0.9.3":"2026-08-21T22:42:05.659Z","0.10.0":"2026-09-01T17:03:17.825Z","0.10.1":"2026-09-01T20:01:25.131Z","0.10.2":"2026-09-02T01:24:56.010Z","0.10.3":"2026-09-26T15:50:10.613Z"},"bugs":{"url":"https://github.com/ruvnet/agent-harness-generator/issues"},"author":{"name":"rUv","email":"ruv@ruv.net"},"license":"MIT","homepage":"https://github.com/ruvnet/agent-harness-generator","keywords":["llm","cost-optimization","llm-optimizer","cheap-llm","compute-arbitrage","agent-harness","self-improvement","evolutionary-search","swe-bench","metaharness","darwin-mode","sandbox"],"repository":{"url":"git+https://github.com/ruvnet/agent-harness-generator.git","type":"git","directory":"packages/darwin-mode"},"description":"Freeze the model, evolve the harness. Two measured applications: (1) SWE-bench code-repair — conformant GLM->Opus empty-patch cascade resolves 51.3% Lite (n=300) and 55.6% Verified (278/500, Wilson 95% CI [51.2, 59.9], official swebench gold eval, no gold","maintainers":[{"name":"ruvnet","email":"ruv@ruv.net"}],"readme":"# @metaharness/darwin\n\n> **An LLM supercharger and cost optimizer.** Keep your model frozen — evolve the\n> harness around it so a *cheap* model performs like an expensive one, for a fraction\n> of the cost.\n\nDarwin Mode makes the LLM you already use **measurably better and cheaper** by\nevolving the *operating system around it* — planner, context builder, reviewer,\nretry/tool/memory/score policy — instead of paying for a bigger model. It mutates one\nsurface at a time, tests each change in a sandbox, and keeps only what *measurably*\nimproves, building an archive of successful descendants. No weight updates, no\nfine-tuning — just a population, a benchmark, and an archive.\n\n**Why it pays off (measured, not marketing — every number links to proof):**\n- **Conformant bug-fixing for half a cent.** A single interactive trajectory (DeepSeek-V4-Flash,\n  no gold tests in-loop) resolves **34.0%** of real **SWE-bench Lite** issues — `102/300`,\n  Wilson 95% CI `[28.9, 39.5]`, **~$0.005/instance** — on the full 300-instance set, official\n  `swebench` Docker harness. [eval report](https://github.com/ruvnet/agent-harness-generator/blob/main/packages/darwin-mode/bench/swebench/setA-full300-eval-report.json)\n  · [LEARNINGS §17](https://github.com/ruvnet/agent-harness-generator/blob/main/packages/darwin-mode/LEARNINGS.md)\n- **Best-of-3 + LLM-judge selection ≈ 52%** (pilot, n=25, conformant; full-300 in progress) at **~$0.015/instance** —\n  [LEARNINGS §15–16](https://github.com/ruvnet/agent-harness-generator/blob/main/packages/darwin-mode/LEARNINGS.md).\n- **Cost-Pareto frontier, not raw score.** At ~$0.005–0.015/instance Darwin sits on the\n  resolve-per-dollar frontier vs. real leaderboard systems (which run $0.1–$2.5+/instance for\n  comparable resolve) — see the live **[Cost-Pareto leaderboard](https://ruvnet.github.io/agent-harness-generator/cost-pareto.html)**\n  (real scores from [swe-bench/experiments](https://github.com/swe-bench/experiments/tree/main/evaluation/lite); competitor costs estimated from disclosed models).\n- **The harness is the multiplier.** The breakthrough was a scaffold change (MCTS → stateful\n  interactive ReAct with the repo's own tests as the regression gate), same cheap model — [LEARNINGS §13](https://github.com/ruvnet/agent-harness-generator/blob/main/packages/darwin-mode/LEARNINGS.md).\n\nThis follows the **Darwin Gödel Machine** lineage: iteratively mutate the source of a\ncoding agent, then *empirically validate* each variant.\n\n## ⭐ The product: Test-Driven Repair (TDR) — a CI Autofixer that resolves 68.3% for pennies\n\n**Hand Darwin a failing test, get a verified-fix PR — at ~$0.01–0.08/instance.** On **SWE-bench Lite**,\nTDR resolves **68.3%** of real issues *when given the acceptance test* (the realistic CI/CD setting —\nwhere a developer or a failing CI job already has the test). Measured on the official `swebench` Docker\nharness, Wilson 95% CI — [RESULTS §30](https://github.com/ruvnet/agent-harness-generator/blob/main/packages/darwin-mode/bench/results/RESULTS.md). This is the hero workflow: a high-margin, low-cost autonomous\nmaintainer for the case that actually matters in production — **a bug with a reproducing test.**\nThis is a *with-acceptance-test* number, **not** a leaderboard claim; the leaderboard-legal (no-test-in-loop)\nresults are the conformant 34%/52% above.\n\n### Two modes (ADR-175) — chosen by whether a test exists\n| mode | when | signal | what you get |\n|------|------|--------|------|\n| **Test-Driven Repair** ⭐ (default) | you have a failing/CI test | gate on **your** test | **CI Autofixer** — verified-fix PR for pennies, **68.3%** with-test |\n| **Conformant** (`--no-test-oracle`) | no test, just a ticket | agent writes its own `reproduce_bug.py`, MCTS-searches the fix (ADR-174) | **Legacy Modernizer** — best-effort fix when no test exists |\n\nSame engine, one flag. **TDR (with your test) is the product** — 68.3%, the number that matters for CI.\nThe conformant (no-test) mode is a genuinely harder capability with a measured, honest ceiling — see the\n[research appendix](#research-appendix-where-no-test-autonomous-repair-tops-out-adr-177) below. The 68.3%\nis a *with-acceptance-test* product claim, deliberately **not** presented as a leaderboard entry (those\nforbid the test in-loop).\n\n```\nrepo\n  → profile      RepoProfile (pkg mgr, test cmd, source/risk files)\n  → baseline     generate the seven mutation-surface files\n  → mutate       pick ONE approved surface, perturb it (behind the gate)\n  → sandbox      safety-inspect → run the test command (no shell, no net, no secrets)\n  → score        weighted base score − hard penalty layer\n  → archive      record parent→child as a TREE (not a single best branch)\n  → select       sample the next generation from the WHOLE archive\n  → repeat\n```\n\nDependency-free: **Node ≥ 20 built-ins only**, no runtime dependencies.\n\n## Quick start\n\nBuild (TypeScript → `dist/`):\n\n```bash\nnpm run build      # tsc\n```\n\nThen evolve a repo with the CLI (one verb, `evolve`):\n\n```bash\nmetaharness-darwin evolve <repo> [--generations N] [--children N] [--concurrency N] [--seed N] \\\n    [--bench <suite.json>] [--tie faster] \\\n    [--selection score|quality-diversity|behavioral-diversity|niche-steering|clade|pareto] \\\n    [--crossover] [--epistasis] [--risk-budget N] [--fdr Q] [--curriculum] [--sandbox real|mock|agent]\n```\n\n| Flag | Meaning | Default |\n|------|---------|---------|\n| `--generations N` | number of generations to run | `3` |\n| `--children N`    | children produced per parent per generation | `4` |\n| `--concurrency N` | max variants evaluated concurrently (bounded fan-out) | `4` |\n| `--seed N`        | deterministic seed for mutation selection | `0` |\n| `--bench <suite.json>` | route promotion through the statistical benchmark gate (ADR-087) | off |\n| `--tie faster`    | break score ties by efficiency (ADR-086) | `insertion` |\n| `--selection …`   | parent-selection strategy (see *Evolutionary stack*) | `score` |\n| `--crossover`     | recombine two parents' surfaces (ADR-089) | off |\n| `--epistasis`     | topology-aware crossover via learned linkage (ADR-093) | off |\n| `--risk-budget N` | SGM cumulative risk cap on promotions (ADR-090) | off |\n| `--fdr Q`         | Benjamini-Hochberg FDR control on promotion (ADR-096) | off |\n| `--curriculum`    | difficulty-ladder over a graded suite (ADR-097) | off |\n| `--sandbox …`     | evaluation substrate: `real` (repo test) · `mock` (surface params, ADR-102) · `agent` (real surface code, ADR-106) | `real` |\n\nAll flags are **opt-in and additive** over a frozen, reproducible core — every default-path run is byte-identical to the ADR-070…075 baseline.\n\nThe `<repo>` argument defaults to the current directory. Everything is written\nunder a self-describing `.metaharness/` work tree inside the repo:\n\n```\n<repo>/.metaharness/\n├── archive.json          # the population TREE: ArchiveRecord[] (variant + score + children)\n├── lineage.json          # serialized graph { nodes, edges } for rendering\n├── variants/             # one directory per variant (its mutation-surface files)\n│   ├── baseline/\n│   ├── g1_v0/  …\n├── runs/                 # one <variantId>.json per variant: { traces, score }\n└── reports/\n    └── winner.json       # the best scored ArchiveRecord\n```\n\nSample run output (leaderboard + winner lineage, printed to stdout):\n\n```\nDarwin Mode — leaderboard\n  0.842  g2_v1  [contextBuilder]  safety=1.00  pass=1.00 ◀ winner\n  0.791  g1_v0  [reviewer]        safety=1.00  pass=1.00\n  0.788  baseline  [planner]      safety=1.00  pass=1.00\n  0.000  g1_v3  [toolPolicy]      safety=0.00  pass=0.00\n\nWinner: g2_v1\nLineage: baseline → g1_v0 → g2_v1\nDelta over baseline: +0.054\n\nArtifacts: <repo>/.metaharness\n```\n\n## The seven mutation surfaces\n\nA child variant may mutate **exactly one** surface per generation, and a variant\ndirectory may contain **only** these seven files — nothing else (the allowlist is\nenforced by `safety.ts`, see `FILE_BY_SURFACE` / `APPROVED_FILES`). Each surface\nis pure, side-effect-free policy logic over injected data.\n\n| Surface (`MutationSurface`) | File | Governs |\n|-----------------------------|------|---------|\n| `planner`        | `planner.ts`         | task string → ordered plan steps (map → inspect → patch → verify) |\n| `contextBuilder` | `context_builder.ts` | ranks candidate files by term overlap with the task |\n| `reviewer`       | `reviewer.ts`        | flags changed files against an injected risk-file list + test outcome |\n| `retryPolicy`    | `retry_policy.ts`    | whether/how to retry given a symbolic failure classification |\n| `toolPolicy`     | `tool_policy.ts`     | allow-list + deterministic ordering over symbolic command kinds |\n| `memoryPolicy`   | `memory_policy.ts`   | whether an outcome record is worth remembering |\n| `scorePolicy`    | `score_policy.ts`    | the weight vector a variant *proposes* over the positive scoring terms |\n\nA variant may *propose* score weights via `scorePolicy`, but it can never\nre-grade itself: the verdict that decides promotion is computed by the frozen\nkernel scorer (see below), not by the variant's own file.\n\n## Scoring and the promotion gate\n\nThe scorer (`src/scorer.ts`, ADR-072) is a **pure function** — re-running it on\nthe same traces yields an identical verdict. It is a weighted base score over six\n`[0,1]` terms (weights from `scoreWeights()`, summing to 1.0):\n\n```\nbaseScore = 0.35·taskSuccess + 0.20·testPassRate + 0.15·traceQuality\n          + 0.10·costEfficiency + 0.10·latencyEfficiency + 0.10·safetyScore\n```\n\nminus a hard **penalty layer** read out of the run traces (a single safety\nviolation can drive the final score negative — that is the point):\n\n```\nfinalScore = baseScore − 0.30·secretExposure − 0.25·destructiveAction\n                       − 0.20·hallucinatedFile − 0.15·toolLoop − 0.10·costOverrun\n```\n\nA child replaces its parent only when **all four** promotion clauses hold\nagainst the parent:\n\n```\n1. beatsParent       finalScore > parentFinalScore + promotionDelta   (default delta 0.05)\n2. safetyOk          safetyScore ≥ 0.95\n3. noRegression      testPassRate ≥ parentTestPassRate\n4. noBlockedActions  safetyScore == 1.0  (zero blocked actions in any trace)\n```\n\nNon-promoted variants are **retained**, not deleted — \"did not clear the gate\"\nmeans \"not chosen as a parent by the default policy,\" never \"removed.\"\n\n## The archive: evolve like species, not release like software\n\nThe archive (`src/archive.ts`, ADR-073) is a **tree** of variants keyed by id and\npersisted as `archive.json`, not a single best branch. Selection\n(`selectParents`) samples the **whole** archive — including older, non-promoted\nbranches — which is how evolution escapes hill-climbing: when a generation\nstalls (no promotions), a weak-looking ancestor can still seed a strong branch.\nInsertion order is preserved, so `best()`, tie-breaks, and `selectParents` are\nall deterministic and reproducible from `archive.json` alone.\n\n## Safety model\n\nA self-modifying agent that can edit anything is a liability. Darwin Mode's bound\nis enforced in `src/safety.ts` (ADR-071) as the **load-bearing security\nboundary**, with two independent, defense-in-depth checks:\n\n- **`inspectVariant(dir)`** runs *before any variant executes*. It disqualifies a\n  variant directory containing anything other than the seven approved files, a\n  blocked filename (`.env`, `secret`, `id_rsa`, `.git`, `package.json`, …), a\n  symlink or nested directory, or blocked content (`process.env`,\n  `child_process`, `eval`, `fetch`, restricted node builtins, shell strings, …).\n- **`validateGeneratedCode(code)`** runs *before generated code is written to\n  disk* (the LLM-mutator path). Independent pattern set; a violating generation\n  is **discarded**, never repaired in place.\n\nThe gate runs **first**: a disqualified variant never has its test command run —\nthe sandbox seals the trace with the reserved exit code `99` and records the\nfindings as `blockedActions`, which zeroes `safetyScore` and makes promotion\nimpossible. When a variant *is* admitted, the sandbox (`src/sandbox.ts`) is\n**shell-free** (the test command is split to argv and run via `execFile`, never a\nshell — no command-injection surface) and runs under a **scrubbed environment**\n(only `PATH` plus three identifying variables; nothing else from `process.env`\nleaks, so secrets, tokens, and proxy settings never reach a variant).\n\nSee [`SECURITY.md`](../../SECURITY.md) for the full threat model.\n\n## Programmatic API\n\n```ts\nimport { evolve } from '@metaharness/darwin';\n\nconst result = await evolve({\n  repoRoot: '/abs/path/to/repo',\n  workRoot: '/abs/path/to/repo/.metaharness',\n  generations: 3,\n  childrenPerGeneration: 4,\n  concurrency: 4,\n  promotionDelta: 0.05,\n  seed: 0,\n  tasks: [\n    'run repository test suite',\n    'verify generated harness safety',\n    'check trace quality',\n  ],\n});\n\nresult.winner;        // the best scored ArchiveRecord (or null)\nresult.winnerLineage; // ['baseline', 'g1_v0', 'g2_v1'] — root → winner\nresult.records;       // every ArchiveRecord, in insertion order\nresult.baseline;      // the baseline record\n```\n\nThe package also re-exports the building blocks behind `evolve`: `profileRepo`,\n`generateBaselineHarness`, `createChildVariant`, `DeterministicMutator` /\n`CodeGenerator`, `runVariantTask` / `runVariantTasks`, `scoreVariant` /\n`scoreWeights`, `Archive`, `inspectVariant` / `validateGeneratedCode`, plus the\n`SURFACES`, `FILE_BY_SURFACE`, and `APPROVED_FILES` constants.\n\n## GEPA — the prompt-policy learning engine (`@metaharness/darwin/gepa`, ADR-228)\n\n\"Freeze the model, evolve the harness\" applied to **prompt policy**. The cheap executor's\noperating policy is a **genome** of named text components; a reflection LM reads per-instance\ntextual feedback (ASI) and proposes targeted single-component mutations; **Pareto selection**\nover per-instance score vectors keeps candidates that win on different task subsets; a\n**strict holdout promotion rule** (gold-no-regress AND empty-patch-improves AND\ncost/resolved-not-worse) gates what ships.\n\n**What ships:** the engine — genome algebra (`genome`), the pre-registered metric +\nfailure-class taxonomy + ASI generation (`metric`), the budgeted optimize loop with a\n**pluggable evaluator** (`loop`), the promotion rule + report helpers (`promotion`) — plus\n**cand-6**, the first holdout-confirmed promoted genome (edit-by-midpoint: holdout gold\n2/12 → 3/12, zero regressions, empty-patch rate 0.583 → 0.333; provenance in\n`genomes/PROVENANCE.md`).\n\n**What does NOT ship:** the SWE-bench/Docker evaluator. It is repo-bound and remains in-repo\nas the reference wiring at\n[`packages/darwin-mode/bench/swebench/gepa/`](https://github.com/ruvnet/metaharness/tree/main/packages/darwin-mode/bench/swebench/gepa)\n(`evaluate-genome.mjs`, `run-gepa.mjs`, `learn.mjs`). You bring the evaluator: any\n`async (genome) => { scores, feedbacks, cost, metricCalls }` over your own task slice.\n\n```ts\nimport { gepaOptimize, loadCand6Genome, buildSystemFromGenome } from '@metaharness/darwin/gepa';\n\nconst seed = loadCand6Genome(); // or SEED_GENOME, or your own dict[str,str] genome\nconst result = await gepaOptimize({\n  seed,\n  // toy in-memory evaluator — replace with your real harness rollout + scoring\n  evaluate: async (genome) => {\n    const sys = buildSystemFromGenome(genome, 'py', '*.py');\n    const scores = { 'task-1': sys.includes('line_edit') ? 1 : 0, 'task-2': 0 };\n    return { scores, feedbacks: { 'task-2': 'task-2: score 0 (gold FAIL).\\nmutation target: retrieval_policy' }, cost: 0 };\n  },\n  // the reflection LM — wire your own chat call; must return the proposal text\n  reflect: async (prompt) => ({ raw: '```component\\nStrategy: read the traceback file first, edit by mid-budget.\\n```', cost: 0 }),\n  maxCandidates: 3,\n});\nresult.best;     // id of the highest-mean candidate\nresult.frontier; // the FULL Pareto frontier (candidates best on ≥1 instance)\nresult.pool;     // evaluated genomes with per-instance score vectors\n```\n\nPromotion is a separate, deliberate step: run your holdout slice through `summarizeEval` and\n`evaluatePromotion({ seed, cand })` — it only says `promote` when the candidate strictly\nimproves out-of-sample without losing anything the seed already solved.\n\n## Evolutionary stack (ADR-084–105)\n\nThe baseline above is the frozen core. On top of it, a set of **opt-in, additive,\nreproducible** mechanisms turn the loop from a single-best search into a real\nevolutionary algorithm. Every one is off by default (so the core stays\nbyte-identical) and individually toggled:\n\n| Capability | ADR | How to enable |\n|---|---|---|\n| **Failure-driven mutation** — feed a parent's failing traces into the mutator | 084 | always (the deterministic mutator ignores it) |\n| **LLM mutator** — `OpenRouterMutator` as a `CodeGenerator`, behind the same safety gate; model chosen by a 15-model execution benchmark | 085 | `config.generator` |\n| **Efficiency tie-break** — break score ties by speed | 086 | `--tie faster` |\n| **Graded statistical promotion** — public∧hidden∧regression∧safety + seeded bootstrap CI over a hash-pinned suite | 087 | `--bench s.json` |\n| **MAP-Elites** — keep the elite per behaviour niche | 088 | `--selection quality-diversity` |\n| **Genetic crossover** — recombine two parents' surfaces | 089 | `--crossover` |\n| **SGM risk budget** — bound cumulative self-modification | 090 | `--risk-budget N` |\n| **Hyperbolic phenotyping** — Poincaré-ball behavioural niche from traces | 091 | `--selection behavioral-diversity` |\n| **Active niche steering** — drive toward under-explored regions | 092 | `--selection niche-steering` |\n| **Epistatic linkage** — topology-aware crossover of co-adapted surfaces | 093 | `--crossover --epistasis` |\n| **Clade metaproductivity** — select parents by descendant potential (Huxley-Gödel) | 094 | `--selection clade` |\n| **Benjamini-Hochberg FDR control** — correct promotion for multiple testing | 096 | `--fdr Q` |\n| **Self-directed curriculum** — difficulty ladder over a graded suite | 097 | `--curriculum` |\n| **Multi-objective Pareto** — non-dominated (capability × parsimony) front | 100 | `--selection pareto` |\n\n### The evaluation substrate (ADR-101/102)\n\nBy default the sandbox runs the **repo's test command**, which is independent of\nthe harness surfaces — so the behavioural manifold is degenerate (measured:\n`nicheEntropy = 0`, ADR-099). `sandboxMode: 'mock'` (ADR-102) instead runs a\n**deterministic surface-driven agent loop**, so a variant's traces depend on its\nsurface content and the manifold comes alive. `sandboxMode: 'agent'` (ADR-106)\nruns a variant's **real surface code** in a child process. The real-LLM-on-real-code\nsubstrate is **no longer deferred** — it shipped (ADR-106→141) and now runs on\n**canonical SWE-bench Lite** (ADR-142+, below).\n\n### Validated results (real, reproducible — see `bench/results/`)\n\n- **Manifold goes live** (ADR-102): real `nicheEntropy 0 → 0.69`, finalScores\n  `flat 0.985 → spread 0.435–0.802` under mock mode.\n- **Self-improvement** (ADR-103): the loop evolves `contextBuilder` (window\n  30 → 70) and climbs `finalScore 0.765 → 0.985` by generation 3.\n- **Diversity beats greedy on deception** (ADR-105): on a deceptive epistatic\n  landscape across 5 seeds, greedy `score` selection crosses it **0/5**,\n  `behavioral-diversity` **5/5**, `clade` **4/5** — empirically justifying the\n  diversity machinery.\n- **Polyglot model frontier** (ADR-085): 15 models × 6 languages, execution-scored;\n  DeepSeek-V3 ($0.4/Mtok) tops quality-per-dollar — cheap beats frontier for code.\n\n### Canonical SWE-bench Lite (real, official harness — ADR-142–149)\n\n> Full reproducible evidence: [`bench/results/RESULTS.md`](bench/results/RESULTS.md) · measured best-practices: [`LEARNINGS.md`](LEARNINGS.md) · known-flaky exclusions: [`bench/swebench/KNOWN_FLAKY.md`](bench/swebench/KNOWN_FLAKY.md)\n\nRun on the **full 300** SWE-bench Lite (test) instances, scored by the **official\n`swebench` Docker harness** — no cherry-picking, tight CIs. Solver = relevance-ranked\ncontext + symbol-aware localization + search/replace patch, `deepseek-chat`, ~$0.01/instance.\n\n| config | resolved | Wilson 95% CI | ADR |\n|---|---|---|---|\n| baseline (open-loop, single-shot) | 23/300 = **7.7%** | [5.2, 11.2] | 144 |\n| + LLM localization | 24/300 = **8.0%** | [5.4, 11.6] | 146 |\n| **+ closed-loop repair (test-feedback, ≤3)** | 46/300 = **15.3%** | **[11.7, 19.8]** | 149 |\n| **+ swap base → deepseek-v4-pro (cheap)** | 88/300 = **29.3%** | **[24.5, 34.7]** | 151 |\n| **+ v4-pro + Scholar hybrid** | 121/300 = **40.3%** | **[34.9, 46.0]** | 152 |\n| **+ Sage (opus-4) — single-shot 3-tier** | 175/300 = **58.3%** | **[52.7, 63.8]** | 154 |\n| agentic full-300 (v4-pro, max-15) | 104/300 = **34.7%** | [29.5, 40.2] | 153/169 |\n| + max-30 + anti-thrash | 139/300 = **46.3%** | [40.8, 52.0] | 169 |\n| + Scholar + Sage (opus-4) — agentic 3-tier | 166/300 = **55.3%** | [49.7, 60.9] | 169 |\n| **+ Sage swapped to opus-4.8 (full tail) — HEADLINE** | 205/300 = **68.3%** | **[62.9, 73.3]** | 172 |\n\n**The harness, not the model, is the dominant lever — and it compounds.** Closed-loop repair\n~doubles a cheap model for free (7.7% → 15.3%, disjoint CIs); a newer cheap base lifts it again\n(→29.3%); and **N-tier cheap→frontier escalation reaches a batch-verified, independently-reproduced\n58.3%** [52.7, 63.8] — 7.6× the open-loop baseline — at ~$0.74/instance blended (vs $1–20 for\nfrontier-on-everything). The mid-arc \"ceiling at 15.3%\" was real for a *fixed* model but **not a\nparadigm limit**. A separate **agentic ReAct loop** (ADR-153 — read/grep/ls/edit/run_tests/submit;\nimplemented + unit-tested) reaches **31.3%** on v4-pro — competitive with single-shot+repair and ~3×\ncheaper per instance; the 65–88% SOTA tier is the next arc (stronger step models / richer tooling).\nHonest caveats throughout: only batch-eval numbers reported (in-loop drifts 1.5–5×), the local-$0\nceiling is capability-floor-bound (14b+repair = 6.7%). Full evidence: `bench/results/RESULTS.md`.\n\n**Update (2026-06-22) — new best 68.3%; the 58.3% ceiling was model-bound.** The full-300 agentic loop\nmeasures **34.7%** (max-15) → **46.3%** (max-30 + anti-thrash) → **55.3%** (agentic 3-tier, opus-4 Sage).\nThe agentic 3-tier *tied* but didn't beat single-shot 58.3% — until we swapped the Sage model to\n**opus-4.8** (newer, *cheaper* ~$0.65/inst), which recovered **35%** of the residual tail opus-4 could\nnot → **new best 68.3%** [59.1, 69.9] (ADR-172; lower bound, full pass projects ~71%). Takeaway:\ncheap-base + tiered escalation **scales with frontier Sage quality** — not exhausted. Difficulty-routing\nwas measured null (ADR-169 E2, AUC 0.505). Next: stronger Sage + the stateful-PTY agent loop (ADR-170).\n\n### Conformant cascade — Verified-500 + LiveCodeBench (2026-06-26)\n\nThese two are **fully conformant** (the solver never sees gold/acceptance tests in-loop; official\nharnesses for scoring only) and live on the [Cost-Pareto leaderboard](https://ruvnet.github.io/agent-harness-generator/cost-pareto.html).\n\n| benchmark | config | resolve | n | Wilson 95% CI | cost |\n|---|---|---|---|---|---|\n| **SWE-bench Verified** | GLM→Opus empty-patch cascade | **55.6%** (278/500) | 500 | **[51.2, 59.9]** | ~$0.15/inst (est.) |\n| SWE-bench Lite | GLM→Opus empty-patch cascade | 51.3% | 300 | — | ~$0.27/inst |\n| **LiveCodeBench** (release_v5 ≥2024-12-01) | single-shot | **44%** | 100 | — | — |\n| **LiveCodeBench** (release_v5 ≥2024-12-01) | cost-cascade (→reasoner) | **62%** | 100 | — | — |\n\n**The cheap cascade generalizes.** The GLM→Opus empty-patch cascade (escalate only the empties to a\nfrontier model) measured **55.6%** on the full **SWE-bench Verified (500)** — official `swebench` gold\neval, Wilson 95% CI [51.2, 59.9], conformant — which **beats** the Lite cascade's 51.3%, consistent with\nVerified being human-validated/cleaner than Lite. The same pattern now holds on **both** splits at ~56×\ncheaper than frontier-on-everything. Still below frontier leaders (70–79%) on raw resolve; this is the\ncheapest path to the ~55% tier. Cost ~$0.15/instance is an **estimate** (per-instance cost not captured\nin predictions). (LEARNINGS §47.)\n\n**LiveCodeBench (contamination-resistant codegen).** On a balanced n=100 from release_v5's ≥2024-12-01\nwindow (post-cutoff by construction), eval-validated against the official `lcb_runner`: **single-shot 44%**,\n**cost-cascade 62%** (escalate the hard tail to a reasoning model). Honest caveats: the deepseek snapshot's\nexact cutoff is **unpinned**; the cascade lift is **partly run-to-run (temp-0) variance** — the clean\nattributable lift is **+8** on the escalated tail; n=100 is **directional, not 1:1** with the official\nwhole-release figure (~34%). (LEARNINGS §46b.)\n\n## Research appendix: where no-test autonomous repair tops out (ADR-177)\n\nThe numbers above are **Test-Driven Repair** — the product — where the acceptance test is available\nin-loop (the real CI/CD case). We *also* ran a rigorous, leaderboard-**conformant** study of the harder\nquestion: *how far can autonomous repair get with **no** test, writing its own?* We report it in full\nbecause the boundary is the engineering result.\n\n**Setup:** the agent never sees the gold tests; it writes its own `reproduce_bug.py` (Test-Critic, ADR-174),\nMCTS-searches patches gated by that self-test, scored once at the end by the gold harness. Gold-graded\n25-instance Lite pilots (Wilson CIs wide at this n; directions are clear):\n\n| config | conformant resolve | $/inst |\n|---|---|---|\n| cheap (DeepSeek, any lever) | **12–16%** | $0.02–0.08 |\n| qwen3-coder-30b | 0–4% | — |\n| Opus-sniper on the cheap tail | 16% (**0 lift**) | $1.01 |\n| **Opus best-of-3 coding** | **33%** | $3.49 |\n\n**Findings (LEARNINGS §10–12, ADR-173–177):**\n- **The coder binds, not the oracle.** A strong (Opus) self-test lifts a cheap coder only 12→16% (noise);\n  every cheap lever (oracle, model-swap, asymmetric sniper, plan-then-edit) is null — all resolve the\n  *same easy instances*.\n- **Goodhart is structural.** Driving up the self-test pass-rate (7→23/25) added **zero** gold resolves —\n  agents overfit a self-written proxy. Only frontier **best-of-k** *diversity* converts.\n- **The scaffold is the ceiling.** Even Opus caps at 33% here (vs its 76.8% Verified via a different\n  harness) — so MCTS+self-repro itself is the limit, independent of model tier.\n\n**Conclusion:** \"leaderboard-SOTA at pennies\" via a no-test cheap-model pipeline is **falsified by our own\nclean data** — a result we publish rather than bury. The product is **TDR with your test (68.3%)**;\nno-test conformant repair is a real but bounded capability (~16–33%), not a top-10 entry. Reaching a\nconformant top-10 (≥45%) would require a different scaffold class (mini-SWE-agent-v2 idioms), out of scope\nfor this release. Full evidence: ADRs 173–177, `LEARNINGS.md` §10–12, tracking issue #45.\n\n## Darwin Shield — the defensive security application (ADR-155, v0.3.0)\n\nThe same thesis — **freeze the model, evolve the harness, prove everything by replay** — applied to a\n*different task*: **defensive vulnerability discovery**. Exported as `security` from this package\n(`src/security/`); run the benchmark with `metaharness-darwin security bench` or `npm run bench:shield`.\n\n- **Evolving genome** (planner / contextPolicy / reviewerCount / retryBudget / fuzzBudget / tools)\n  with bounded mutation + crossover; `safetyProfile` is **immutable**. Three fixed baselines\n  (static / LLM single-pass / fixed agent) to beat.\n- **Safety layer is load-bearing**: scope gate, exploit redactor, unsafe-output gate —\n  `exploitCodeAllowed` is a hard `false`; any unsafe output is an immediate **−1.00 fitness** term.\n  This is a *defensive* harness (find + prove + patch vulnerabilities), not an exploit generator.\n- **Real oracles**: a Semgrep detector + a property fuzzer + an in-loop judge; with Semgrep present,\n  the security suite runs **hundreds of tests**. Receipts are byte-identical (deterministic replay).\n- **DARWIN-SHIELD-BENCH** (pop 16 × 50 cycles) passes every ADR-155 gate on the seeded corpus:\n  TPR +150% vs the fixed harness, FPR −100%, patch-pass 100%, repro 100%, **0 unsafe outputs**,\n  cost ≤ 2×.\n\nSee also the sibling package **[`@metaharness/projects`](../projects/)** (ADR-156…167) — the\nborrowed-pattern integration program backing this work.\n\n## Status\n\n**Working, empirically validated on the mock substrate, canonical SWE-bench Lite, *and* the\nDARWIN-SHIELD security benchmark.** The `DeterministicMutator` is seeded + signature-preserving; the\n`OpenRouterMutator` (ADR-085) is the production LLM `CodeGenerator`, behind the *same*\n`validateGeneratedCode` gate. SWE-bench is **measured end-to-end**: 7.7% open-loop → 15.3% repair →\n29.3% v4-pro → 40.3% 2-tier → **58.3% 3-tier** (ADR-154, verified + reproduced), plus an agentic ReAct\nloop at 31.3% (ADR-153) and a $0 local track (ADR-150). The defensive **Darwin Shield** application\n(ADR-155) ships in v0.3.0. Darwin Mode also ships **integrated into the `metaharness` scaffolder** —\n`npx metaharness <name>` produces a harness with `npm run evolve` out of the box (ADR-147).\n\n## License\n\nMIT © rUv. See ADRs\n[070](../../docs/adrs/ADR-070-darwin-mode-self-improving-harness.md) ·\n[071](../../docs/adrs/ADR-071-darwin-mutation-surfaces-safety-allowlist.md) ·\n[072](../../docs/adrs/ADR-072-darwin-scoring-and-promotion.md) ·\n[073](../../docs/adrs/ADR-073-darwin-archive-and-selection.md) ·\n[074](../../docs/adrs/ADR-074-darwin-ruvector-memory-ruflo-fabric.md) ·\n[075](../../docs/adrs/ADR-075-darwin-prototype-roadmap-and-acceptance.md),\nand the [repository](https://github.com/ruvnet/agent-harness-generator).\n","readmeFilename":"README.md"}