oh-my-knowledge 0.33.0 → 0.35.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (739) hide show
  1. package/README.md +21 -5
  2. package/README.zh.md +25 -9
  3. package/dist/analysis/coverage-analyzer.d.ts +0 -1
  4. package/dist/analysis/coverage-analyzer.js +0 -1
  5. package/dist/analysis/failure-clusterer.d.ts +0 -1
  6. package/dist/analysis/failure-clusterer.js +0 -1
  7. package/dist/analysis/gap-analyzer.d.ts +0 -1
  8. package/dist/analysis/gap-analyzer.js +0 -1
  9. package/dist/analysis/hedging-classifier.d.ts +0 -1
  10. package/dist/analysis/hedging-classifier.js +1 -2
  11. package/dist/analysis/report-diagnostics.d.ts +0 -1
  12. package/dist/analysis/report-diagnostics.js +0 -1
  13. package/dist/analysis/sample-diagnostics.d.ts +1 -2
  14. package/dist/analysis/sample-diagnostics.js +18 -19
  15. package/dist/analysis/saturation.d.ts +0 -1
  16. package/dist/analysis/saturation.js +0 -1
  17. package/dist/assets/agent-skills/omk/SKILL.md +197 -0
  18. package/dist/assets/agent-skills/omk/references/commands.md +556 -0
  19. package/dist/authoring/evolver.d.ts +130 -6
  20. package/dist/authoring/evolver.js +233 -35
  21. package/dist/authoring/generator.d.ts +0 -1
  22. package/dist/authoring/generator.js +0 -1
  23. package/dist/authoring/sample-fixer.d.ts +0 -1
  24. package/dist/authoring/sample-fixer.js +0 -1
  25. package/dist/cli/commands/doctor.d.ts +0 -1
  26. package/dist/cli/commands/doctor.js +2 -3
  27. package/dist/cli/commands/eval/gold/compare.d.ts +0 -1
  28. package/dist/cli/commands/eval/gold/compare.js +0 -1
  29. package/dist/cli/commands/eval/gold/index.d.ts +0 -1
  30. package/dist/cli/commands/eval/gold/index.js +0 -1
  31. package/dist/cli/commands/eval/gold/init.d.ts +0 -1
  32. package/dist/cli/commands/eval/gold/init.js +0 -1
  33. package/dist/cli/commands/eval/gold/validate.d.ts +0 -1
  34. package/dist/cli/commands/eval/gold/validate.js +0 -1
  35. package/dist/cli/commands/eval/index.d.ts +3 -1
  36. package/dist/cli/commands/eval/index.js +56 -11
  37. package/dist/cli/commands/evolve.d.ts +6 -1
  38. package/dist/cli/commands/evolve.js +77 -6
  39. package/dist/cli/commands/init.d.ts +0 -1
  40. package/dist/cli/commands/init.js +0 -1
  41. package/dist/cli/commands/install.d.ts +24 -0
  42. package/dist/cli/commands/install.js +436 -0
  43. package/dist/cli/commands/observe/inbox.d.ts +0 -1
  44. package/dist/cli/commands/observe/inbox.js +0 -1
  45. package/dist/cli/commands/observe/index.d.ts +0 -1
  46. package/dist/cli/commands/observe/index.js +0 -1
  47. package/dist/cli/commands/observe/ingest.d.ts +0 -1
  48. package/dist/cli/commands/observe/ingest.js +0 -1
  49. package/dist/cli/commands/observe/show.d.ts +0 -1
  50. package/dist/cli/commands/observe/show.js +0 -1
  51. package/dist/cli/commands/sample.d.ts +1 -2
  52. package/dist/cli/commands/sample.js +27 -15
  53. package/dist/cli/commands/studio.d.ts +0 -1
  54. package/dist/cli/commands/studio.js +0 -1
  55. package/dist/cli/index.d.ts +0 -1
  56. package/dist/cli/index.js +1 -2
  57. package/dist/cli/lib/cli-exit.d.ts +0 -1
  58. package/dist/cli/lib/cli-exit.js +0 -1
  59. package/dist/cli/lib/cmd-flags.d.ts +7 -1
  60. package/dist/cli/lib/cmd-flags.js +0 -1
  61. package/dist/cli/lib/i18n-dict/common.d.ts +0 -1
  62. package/dist/cli/lib/i18n-dict/common.js +0 -1
  63. package/dist/cli/lib/i18n-dict/evolve.d.ts +1 -2
  64. package/dist/cli/lib/i18n-dict/evolve.js +16 -1
  65. package/dist/cli/lib/i18n-dict/gen.d.ts +0 -1
  66. package/dist/cli/lib/i18n-dict/gen.js +0 -1
  67. package/dist/cli/lib/i18n-dict/help.d.ts +0 -1
  68. package/dist/cli/lib/i18n-dict/help.js +0 -1
  69. package/dist/cli/lib/i18n-dict/init.d.ts +0 -1
  70. package/dist/cli/lib/i18n-dict/init.js +0 -1
  71. package/dist/cli/lib/i18n-dict/install.d.ts +3 -0
  72. package/dist/cli/lib/i18n-dict/install.js +110 -0
  73. package/dist/cli/lib/i18n-dict/run.d.ts +1 -2
  74. package/dist/cli/lib/i18n-dict/run.js +8 -1
  75. package/dist/cli/lib/i18n-dict/types.d.ts +0 -1
  76. package/dist/cli/lib/i18n-dict/types.js +0 -1
  77. package/dist/cli/lib/i18n-dict.d.ts +2 -2
  78. package/dist/cli/lib/i18n-dict.js +2 -1
  79. package/dist/cli/lib/i18n.d.ts +0 -1
  80. package/dist/cli/lib/i18n.js +0 -1
  81. package/dist/cli/lib/parse-run-config/judge-models.d.ts +0 -1
  82. package/dist/cli/lib/parse-run-config/judge-models.js +0 -1
  83. package/dist/cli/lib/parse-run-config/samples-discovery.d.ts +0 -1
  84. package/dist/cli/lib/parse-run-config/samples-discovery.js +1 -4
  85. package/dist/cli/lib/parse-run-config/variant-resolution.d.ts +0 -1
  86. package/dist/cli/lib/parse-run-config/variant-resolution.js +41 -8
  87. package/dist/cli/lib/parse-run-config.d.ts +0 -1
  88. package/dist/cli/lib/parse-run-config.js +1 -2
  89. package/dist/cli/lib/progress.d.ts +0 -1
  90. package/dist/cli/lib/progress.js +0 -1
  91. package/dist/cli/lib/resolve-skill-input.d.ts +0 -1
  92. package/dist/cli/lib/resolve-skill-input.js +0 -1
  93. package/dist/cli/lib/run-tally.d.ts +0 -1
  94. package/dist/cli/lib/run-tally.js +1 -2
  95. package/dist/cli/lib/shared.d.ts +0 -1
  96. package/dist/cli/lib/shared.js +1 -2
  97. package/dist/cli/lib/update-check.d.ts +0 -1
  98. package/dist/cli/lib/update-check.js +0 -1
  99. package/dist/cli/lib/update-fetch-worker.d.ts +0 -1
  100. package/dist/cli/lib/update-fetch-worker.js +0 -1
  101. package/dist/cli/oclif/base-command.d.ts +0 -1
  102. package/dist/cli/oclif/base-command.js +0 -1
  103. package/dist/cli/oclif/help.d.ts +0 -1
  104. package/dist/cli/oclif/help.js +0 -1
  105. package/dist/cli/oclif/i18n.d.ts +0 -1
  106. package/dist/cli/oclif/i18n.js +0 -1
  107. package/dist/cli/oclif/parsers.d.ts +0 -1
  108. package/dist/cli/oclif/parsers.js +0 -1
  109. package/dist/cli/oclif/projection.d.ts +0 -1
  110. package/dist/cli/oclif/projection.js +0 -1
  111. package/dist/cli/oclif/run.d.ts +0 -1
  112. package/dist/cli/oclif/run.js +0 -1
  113. package/dist/diagnosis/observe-mapper.d.ts +2 -3
  114. package/dist/diagnosis/observe-mapper.js +6 -7
  115. package/dist/diagnosis/observe-producer.d.ts +0 -1
  116. package/dist/diagnosis/observe-producer.js +3 -4
  117. package/dist/diagnosis/studio-projection.d.ts +0 -1
  118. package/dist/diagnosis/studio-projection.js +0 -1
  119. package/dist/diagnosis/types.d.ts +0 -1
  120. package/dist/diagnosis/types.js +0 -1
  121. package/dist/doctor/fixer.d.ts +0 -1
  122. package/dist/doctor/fixer.js +0 -1
  123. package/dist/doctor/health/builtin-dimensions.d.ts +0 -1
  124. package/dist/doctor/health/builtin-dimensions.js +0 -1
  125. package/dist/doctor/health/composer.d.ts +0 -1
  126. package/dist/doctor/health/composer.js +1 -2
  127. package/dist/doctor/health/dimension-registry.d.ts +0 -1
  128. package/dist/doctor/health/dimension-registry.js +0 -1
  129. package/dist/doctor/health/dimension-spec.d.ts +0 -1
  130. package/dist/doctor/health/dimension-spec.js +0 -1
  131. package/dist/doctor/health/load-custom-dimensions.d.ts +0 -1
  132. package/dist/doctor/health/load-custom-dimensions.js +0 -1
  133. package/dist/doctor/health/parser.d.ts +0 -1
  134. package/dist/doctor/health/parser.js +0 -1
  135. package/dist/doctor/health/prompt-builder.d.ts +0 -1
  136. package/dist/doctor/health/prompt-builder.js +0 -1
  137. package/dist/doctor/health/register.d.ts +0 -1
  138. package/dist/doctor/health/register.js +0 -1
  139. package/dist/doctor/index.d.ts +0 -1
  140. package/dist/doctor/index.js +6 -7
  141. package/dist/doctor/messages.d.ts +0 -1
  142. package/dist/doctor/messages.js +1 -2
  143. package/dist/doctor/preflight.d.ts +0 -1
  144. package/dist/doctor/preflight.js +0 -1
  145. package/dist/doctor/renderer.d.ts +0 -1
  146. package/dist/doctor/renderer.js +0 -1
  147. package/dist/doctor/rules.d.ts +0 -1
  148. package/dist/doctor/rules.js +0 -1
  149. package/dist/eval-core/bootstrap.d.ts +0 -1
  150. package/dist/eval-core/bootstrap.js +0 -1
  151. package/dist/eval-core/cache.d.ts +11 -4
  152. package/dist/eval-core/cache.js +13 -6
  153. package/dist/eval-core/comparability.d.ts +0 -1
  154. package/dist/eval-core/comparability.js +3 -4
  155. package/dist/eval-core/dependency-checker.d.ts +0 -1
  156. package/dist/eval-core/dependency-checker.js +5 -3
  157. package/dist/eval-core/evaluation-execution.d.ts +0 -1
  158. package/dist/eval-core/evaluation-execution.js +4 -2
  159. package/dist/eval-core/evaluation-job.d.ts +0 -1
  160. package/dist/eval-core/evaluation-job.js +0 -1
  161. package/dist/eval-core/evaluation-reporting.d.ts +0 -1
  162. package/dist/eval-core/evaluation-reporting.js +18 -8
  163. package/dist/eval-core/execution-strategy.d.ts +0 -1
  164. package/dist/eval-core/execution-strategy.js +8 -3
  165. package/dist/eval-core/fact-checker.d.ts +0 -1
  166. package/dist/eval-core/fact-checker.js +0 -1
  167. package/dist/eval-core/layer-gates.d.ts +0 -1
  168. package/dist/eval-core/layer-gates.js +0 -1
  169. package/dist/eval-core/mocks-runtime.d.ts +0 -1
  170. package/dist/eval-core/mocks-runtime.js +0 -1
  171. package/dist/eval-core/schema.d.ts +0 -1
  172. package/dist/eval-core/schema.js +0 -1
  173. package/dist/eval-core/statistics.d.ts +0 -1
  174. package/dist/eval-core/statistics.js +0 -1
  175. package/dist/eval-core/task-planner.d.ts +0 -1
  176. package/dist/eval-core/task-planner.js +2 -2
  177. package/dist/eval-core/verdict.d.ts +0 -1
  178. package/dist/eval-core/verdict.js +0 -1
  179. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +18 -3
  180. package/dist/eval-workflows/batch-evaluation-workflow.js +31 -15
  181. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +0 -1
  182. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +0 -1
  183. package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +0 -1
  184. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +0 -1
  185. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +0 -1
  186. package/dist/eval-workflows/evaluation-pipeline/run-state.js +0 -1
  187. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +0 -1
  188. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +0 -1
  189. package/dist/eval-workflows/evaluation-pipeline.d.ts +0 -1
  190. package/dist/eval-workflows/evaluation-pipeline.js +0 -1
  191. package/dist/eval-workflows/evaluation-preparation.d.ts +1 -5
  192. package/dist/eval-workflows/evaluation-preparation.js +22 -44
  193. package/dist/eval-workflows/messages.d.ts +0 -1
  194. package/dist/eval-workflows/messages.js +0 -1
  195. package/dist/eval-workflows/run-evaluation.d.ts +3 -8
  196. package/dist/eval-workflows/run-evaluation.js +9 -19
  197. package/dist/executors/anthropic-api.d.ts +0 -1
  198. package/dist/executors/anthropic-api.js +1 -2
  199. package/dist/executors/claude-cli.d.ts +0 -1
  200. package/dist/executors/claude-cli.js +0 -1
  201. package/dist/executors/claude-sdk-trace.d.ts +0 -1
  202. package/dist/executors/claude-sdk-trace.js +0 -1
  203. package/dist/executors/claude-sdk.d.ts +0 -1
  204. package/dist/executors/claude-sdk.js +0 -1
  205. package/dist/executors/codex-cli-trace.d.ts +0 -1
  206. package/dist/executors/codex-cli-trace.js +0 -1
  207. package/dist/executors/codex-cli.d.ts +0 -1
  208. package/dist/executors/codex-cli.js +0 -1
  209. package/dist/executors/codex-sdk.d.ts +0 -1
  210. package/dist/executors/codex-sdk.js +0 -1
  211. package/dist/executors/gemini.d.ts +0 -1
  212. package/dist/executors/gemini.js +0 -1
  213. package/dist/executors/index.d.ts +0 -1
  214. package/dist/executors/index.js +0 -1
  215. package/dist/executors/openai-api.d.ts +0 -1
  216. package/dist/executors/openai-api.js +0 -1
  217. package/dist/executors/runtime-fingerprint.d.ts +0 -1
  218. package/dist/executors/runtime-fingerprint.js +2 -3
  219. package/dist/executors/script.d.ts +0 -1
  220. package/dist/executors/script.js +0 -1
  221. package/dist/executors/shared.d.ts +0 -2
  222. package/dist/executors/shared.js +0 -1
  223. package/dist/grading/assertions.d.ts +0 -1
  224. package/dist/grading/assertions.js +0 -1
  225. package/dist/grading/debias-validate.d.ts +0 -1
  226. package/dist/grading/debias-validate.js +0 -1
  227. package/dist/grading/diagnostic.d.ts +0 -1
  228. package/dist/grading/diagnostic.js +0 -1
  229. package/dist/grading/gold-cli.d.ts +0 -1
  230. package/dist/grading/gold-cli.js +0 -1
  231. package/dist/grading/gold-dataset.d.ts +0 -1
  232. package/dist/grading/gold-dataset.js +0 -1
  233. package/dist/grading/human-gold.d.ts +0 -1
  234. package/dist/grading/human-gold.js +0 -1
  235. package/dist/grading/index.d.ts +0 -1
  236. package/dist/grading/index.js +0 -1
  237. package/dist/grading/judge.d.ts +0 -1
  238. package/dist/grading/judge.js +0 -1
  239. package/dist/grading/layered-scores.d.ts +0 -1
  240. package/dist/grading/layered-scores.js +0 -1
  241. package/dist/inputs/content-hash.d.ts +28 -0
  242. package/dist/inputs/content-hash.js +106 -0
  243. package/dist/inputs/eval-config.d.ts +0 -1
  244. package/dist/inputs/eval-config.js +44 -11
  245. package/dist/inputs/load-samples.d.ts +0 -1
  246. package/dist/inputs/load-samples.js +0 -1
  247. package/dist/inputs/materialize-copy.d.ts +37 -0
  248. package/dist/inputs/materialize-copy.js +193 -0
  249. package/dist/inputs/mcp-resolver.d.ts +0 -1
  250. package/dist/inputs/mcp-resolver.js +0 -1
  251. package/dist/inputs/skill-loader.d.ts +145 -12
  252. package/dist/inputs/skill-loader.js +476 -60
  253. package/dist/inputs/source-resolver.d.ts +35 -0
  254. package/dist/inputs/source-resolver.js +105 -0
  255. package/dist/inputs/url-fetcher.d.ts +0 -1
  256. package/dist/inputs/url-fetcher.js +0 -1
  257. package/dist/managed/evidence.d.ts +22 -0
  258. package/dist/managed/evidence.js +143 -0
  259. package/dist/managed/index.d.ts +6 -0
  260. package/dist/managed/index.js +6 -0
  261. package/dist/managed/store.d.ts +72 -0
  262. package/dist/managed/store.js +224 -0
  263. package/dist/observability/experience-frontmatter.d.ts +0 -1
  264. package/dist/observability/experience-frontmatter.js +0 -1
  265. package/dist/observability/experience.d.ts +1 -2
  266. package/dist/observability/experience.js +1 -2
  267. package/dist/observability/feedback-matchers.d.ts +0 -1
  268. package/dist/observability/feedback-matchers.js +0 -1
  269. package/dist/observability/feedback-projection.d.ts +0 -1
  270. package/dist/observability/feedback-projection.js +0 -1
  271. package/dist/observability/inbox-view-model.d.ts +0 -1
  272. package/dist/observability/inbox-view-model.js +0 -1
  273. package/dist/observability/inbox.d.ts +0 -1
  274. package/dist/observability/inbox.js +2 -3
  275. package/dist/observability/problem-patterns.d.ts +0 -1
  276. package/dist/observability/problem-patterns.js +0 -1
  277. package/dist/observability/resolved-review.d.ts +0 -1
  278. package/dist/observability/resolved-review.js +4 -5
  279. package/dist/observability/review-state.d.ts +0 -1
  280. package/dist/observability/review-state.js +3 -4
  281. package/dist/observability/skill-chain-advisories.d.ts +0 -1
  282. package/dist/observability/skill-chain-advisories.js +0 -1
  283. package/dist/observability/skill-chain.d.ts +0 -1
  284. package/dist/observability/skill-chain.js +5 -6
  285. package/dist/observability/skill-health-analyzer.d.ts +0 -1
  286. package/dist/observability/skill-health-analyzer.js +0 -1
  287. package/dist/observability/soft-standards/constants.d.ts +0 -1
  288. package/dist/observability/soft-standards/constants.js +0 -1
  289. package/dist/observability/soft-standards/index.d.ts +0 -1
  290. package/dist/observability/soft-standards/index.js +0 -1
  291. package/dist/observability/soft-standards/llm-extractor.d.ts +0 -1
  292. package/dist/observability/soft-standards/llm-extractor.js +7 -8
  293. package/dist/observability/soft-standards/runtime-evaluator.d.ts +0 -1
  294. package/dist/observability/soft-standards/runtime-evaluator.js +2 -3
  295. package/dist/observability/soft-standards/skill-standards-store.d.ts +0 -1
  296. package/dist/observability/soft-standards/skill-standards-store.js +5 -6
  297. package/dist/observability/soft-standards/types.d.ts +10 -11
  298. package/dist/observability/soft-standards/types.js +0 -1
  299. package/dist/observability/text-signals.d.ts +0 -1
  300. package/dist/observability/text-signals.js +0 -1
  301. package/dist/observability/trace-adapter.d.ts +0 -1
  302. package/dist/observability/trace-adapter.js +0 -1
  303. package/dist/observability/trace-attribution.d.ts +0 -1
  304. package/dist/observability/trace-attribution.js +0 -1
  305. package/dist/observability/trace-segmenter.d.ts +0 -1
  306. package/dist/observability/trace-segmenter.js +0 -1
  307. package/dist/observability/trace-source.d.ts +0 -1
  308. package/dist/observability/trace-source.js +0 -1
  309. package/dist/renderer/html-renderer.d.ts +0 -1
  310. package/dist/renderer/html-renderer.js +4 -5
  311. package/dist/renderer/layout.d.ts +0 -1
  312. package/dist/renderer/layout.js +2 -3
  313. package/dist/renderer/observation-inbox/helpers.d.ts +0 -1
  314. package/dist/renderer/observation-inbox/helpers.js +0 -1
  315. package/dist/renderer/observation-inbox/styles.d.ts +0 -1
  316. package/dist/renderer/observation-inbox/styles.js +0 -1
  317. package/dist/renderer/observation-inbox-renderer.d.ts +0 -1
  318. package/dist/renderer/observation-inbox-renderer.js +13 -14
  319. package/dist/renderer/skill-detail-renderer.d.ts +0 -1
  320. package/dist/renderer/skill-detail-renderer.js +3 -4
  321. package/dist/renderer/skill-health-renderer.d.ts +0 -1
  322. package/dist/renderer/skill-health-renderer.js +0 -1
  323. package/dist/renderer/skill-list-renderer.d.ts +0 -1
  324. package/dist/renderer/skill-list-renderer.js +0 -1
  325. package/dist/renderer/summary.d.ts +0 -1
  326. package/dist/renderer/summary.js +1 -2
  327. package/dist/renderer/table.d.ts +0 -1
  328. package/dist/renderer/table.js +0 -1
  329. package/dist/renderer/test-view.d.ts +0 -1
  330. package/dist/renderer/test-view.js +0 -1
  331. package/dist/renderer/trends.d.ts +0 -1
  332. package/dist/renderer/trends.js +0 -1
  333. package/dist/server/job-store.d.ts +0 -1
  334. package/dist/server/job-store.js +0 -1
  335. package/dist/server/report-server.d.ts +0 -1
  336. package/dist/server/report-server.js +1 -2
  337. package/dist/server/report-store.d.ts +1 -2
  338. package/dist/server/report-store.js +6 -7
  339. package/dist/server/skill-index.d.ts +0 -1
  340. package/dist/server/skill-index.js +4 -5
  341. package/dist/server/skill-insights.d.ts +0 -1
  342. package/dist/server/skill-insights.js +8 -9
  343. package/dist/shared/hard-rules.d.ts +0 -1
  344. package/dist/shared/hard-rules.js +0 -1
  345. package/dist/shared/llm-prompts/index.d.ts +0 -1
  346. package/dist/shared/llm-prompts/index.js +0 -1
  347. package/dist/shared/llm-prompts/skill-health.d.ts +0 -1
  348. package/dist/shared/llm-prompts/skill-health.js +0 -1
  349. package/dist/shared/time.d.ts +0 -1
  350. package/dist/shared/time.js +0 -1
  351. package/dist/shared/tool-search.d.ts +0 -1
  352. package/dist/shared/tool-search.js +0 -1
  353. package/dist/types/dependencies.d.ts +0 -1
  354. package/dist/types/dependencies.js +0 -1
  355. package/dist/types/diagnosis.d.ts +0 -1
  356. package/dist/types/diagnosis.js +0 -1
  357. package/dist/types/doctor.d.ts +4 -5
  358. package/dist/types/doctor.js +2 -3
  359. package/dist/types/eval.d.ts +13 -2
  360. package/dist/types/eval.js +0 -1
  361. package/dist/types/executor.d.ts +1 -2
  362. package/dist/types/executor.js +0 -1
  363. package/dist/types/index.d.ts +1 -1
  364. package/dist/types/index.js +1 -1
  365. package/dist/types/judge.d.ts +0 -1
  366. package/dist/types/judge.js +0 -1
  367. package/dist/types/managed.d.ts +110 -0
  368. package/dist/types/managed.js +1 -0
  369. package/dist/types/observability.d.ts +5 -6
  370. package/dist/types/observability.js +0 -1
  371. package/dist/types/report.d.ts +19 -6
  372. package/dist/types/report.js +0 -1
  373. package/dist/types/shared.d.ts +0 -1
  374. package/dist/types/shared.js +0 -1
  375. package/dist/types/skill-index.d.ts +0 -1
  376. package/dist/types/skill-index.js +0 -1
  377. package/dist/types/storage.d.ts +0 -1
  378. package/dist/types/storage.js +0 -1
  379. package/dist/util/safe-slice.d.ts +0 -1
  380. package/dist/util/safe-slice.js +0 -1
  381. package/package.json +9 -4
  382. package/dist/analysis/coverage-analyzer.d.ts.map +0 -1
  383. package/dist/analysis/coverage-analyzer.js.map +0 -1
  384. package/dist/analysis/failure-clusterer.d.ts.map +0 -1
  385. package/dist/analysis/failure-clusterer.js.map +0 -1
  386. package/dist/analysis/gap-analyzer.d.ts.map +0 -1
  387. package/dist/analysis/gap-analyzer.js.map +0 -1
  388. package/dist/analysis/hedging-classifier.d.ts.map +0 -1
  389. package/dist/analysis/hedging-classifier.js.map +0 -1
  390. package/dist/analysis/report-diagnostics.d.ts.map +0 -1
  391. package/dist/analysis/report-diagnostics.js.map +0 -1
  392. package/dist/analysis/sample-diagnostics.d.ts.map +0 -1
  393. package/dist/analysis/sample-diagnostics.js.map +0 -1
  394. package/dist/analysis/saturation.d.ts.map +0 -1
  395. package/dist/analysis/saturation.js.map +0 -1
  396. package/dist/authoring/evolver.d.ts.map +0 -1
  397. package/dist/authoring/evolver.js.map +0 -1
  398. package/dist/authoring/generator.d.ts.map +0 -1
  399. package/dist/authoring/generator.js.map +0 -1
  400. package/dist/authoring/sample-fixer.d.ts.map +0 -1
  401. package/dist/authoring/sample-fixer.js.map +0 -1
  402. package/dist/cli/commands/doctor.d.ts.map +0 -1
  403. package/dist/cli/commands/doctor.js.map +0 -1
  404. package/dist/cli/commands/eval/gold/compare.d.ts.map +0 -1
  405. package/dist/cli/commands/eval/gold/compare.js.map +0 -1
  406. package/dist/cli/commands/eval/gold/index.d.ts.map +0 -1
  407. package/dist/cli/commands/eval/gold/index.js.map +0 -1
  408. package/dist/cli/commands/eval/gold/init.d.ts.map +0 -1
  409. package/dist/cli/commands/eval/gold/init.js.map +0 -1
  410. package/dist/cli/commands/eval/gold/validate.d.ts.map +0 -1
  411. package/dist/cli/commands/eval/gold/validate.js.map +0 -1
  412. package/dist/cli/commands/eval/index.d.ts.map +0 -1
  413. package/dist/cli/commands/eval/index.js.map +0 -1
  414. package/dist/cli/commands/evolve.d.ts.map +0 -1
  415. package/dist/cli/commands/evolve.js.map +0 -1
  416. package/dist/cli/commands/init.d.ts.map +0 -1
  417. package/dist/cli/commands/init.js.map +0 -1
  418. package/dist/cli/commands/observe/inbox.d.ts.map +0 -1
  419. package/dist/cli/commands/observe/inbox.js.map +0 -1
  420. package/dist/cli/commands/observe/index.d.ts.map +0 -1
  421. package/dist/cli/commands/observe/index.js.map +0 -1
  422. package/dist/cli/commands/observe/ingest.d.ts.map +0 -1
  423. package/dist/cli/commands/observe/ingest.js.map +0 -1
  424. package/dist/cli/commands/observe/show.d.ts.map +0 -1
  425. package/dist/cli/commands/observe/show.js.map +0 -1
  426. package/dist/cli/commands/sample.d.ts.map +0 -1
  427. package/dist/cli/commands/sample.js.map +0 -1
  428. package/dist/cli/commands/studio.d.ts.map +0 -1
  429. package/dist/cli/commands/studio.js.map +0 -1
  430. package/dist/cli/index.d.ts.map +0 -1
  431. package/dist/cli/index.js.map +0 -1
  432. package/dist/cli/lib/cli-exit.d.ts.map +0 -1
  433. package/dist/cli/lib/cli-exit.js.map +0 -1
  434. package/dist/cli/lib/cmd-flags.d.ts.map +0 -1
  435. package/dist/cli/lib/cmd-flags.js.map +0 -1
  436. package/dist/cli/lib/i18n-dict/common.d.ts.map +0 -1
  437. package/dist/cli/lib/i18n-dict/common.js.map +0 -1
  438. package/dist/cli/lib/i18n-dict/evolve.d.ts.map +0 -1
  439. package/dist/cli/lib/i18n-dict/evolve.js.map +0 -1
  440. package/dist/cli/lib/i18n-dict/gen.d.ts.map +0 -1
  441. package/dist/cli/lib/i18n-dict/gen.js.map +0 -1
  442. package/dist/cli/lib/i18n-dict/help.d.ts.map +0 -1
  443. package/dist/cli/lib/i18n-dict/help.js.map +0 -1
  444. package/dist/cli/lib/i18n-dict/init.d.ts.map +0 -1
  445. package/dist/cli/lib/i18n-dict/init.js.map +0 -1
  446. package/dist/cli/lib/i18n-dict/run.d.ts.map +0 -1
  447. package/dist/cli/lib/i18n-dict/run.js.map +0 -1
  448. package/dist/cli/lib/i18n-dict/types.d.ts.map +0 -1
  449. package/dist/cli/lib/i18n-dict/types.js.map +0 -1
  450. package/dist/cli/lib/i18n-dict.d.ts.map +0 -1
  451. package/dist/cli/lib/i18n-dict.js.map +0 -1
  452. package/dist/cli/lib/i18n.d.ts.map +0 -1
  453. package/dist/cli/lib/i18n.js.map +0 -1
  454. package/dist/cli/lib/parse-run-config/judge-models.d.ts.map +0 -1
  455. package/dist/cli/lib/parse-run-config/judge-models.js.map +0 -1
  456. package/dist/cli/lib/parse-run-config/samples-discovery.d.ts.map +0 -1
  457. package/dist/cli/lib/parse-run-config/samples-discovery.js.map +0 -1
  458. package/dist/cli/lib/parse-run-config/variant-resolution.d.ts.map +0 -1
  459. package/dist/cli/lib/parse-run-config/variant-resolution.js.map +0 -1
  460. package/dist/cli/lib/parse-run-config.d.ts.map +0 -1
  461. package/dist/cli/lib/parse-run-config.js.map +0 -1
  462. package/dist/cli/lib/progress.d.ts.map +0 -1
  463. package/dist/cli/lib/progress.js.map +0 -1
  464. package/dist/cli/lib/resolve-skill-input.d.ts.map +0 -1
  465. package/dist/cli/lib/resolve-skill-input.js.map +0 -1
  466. package/dist/cli/lib/run-tally.d.ts.map +0 -1
  467. package/dist/cli/lib/run-tally.js.map +0 -1
  468. package/dist/cli/lib/shared.d.ts.map +0 -1
  469. package/dist/cli/lib/shared.js.map +0 -1
  470. package/dist/cli/lib/update-check.d.ts.map +0 -1
  471. package/dist/cli/lib/update-check.js.map +0 -1
  472. package/dist/cli/lib/update-fetch-worker.d.ts.map +0 -1
  473. package/dist/cli/lib/update-fetch-worker.js.map +0 -1
  474. package/dist/cli/oclif/base-command.d.ts.map +0 -1
  475. package/dist/cli/oclif/base-command.js.map +0 -1
  476. package/dist/cli/oclif/help.d.ts.map +0 -1
  477. package/dist/cli/oclif/help.js.map +0 -1
  478. package/dist/cli/oclif/i18n.d.ts.map +0 -1
  479. package/dist/cli/oclif/i18n.js.map +0 -1
  480. package/dist/cli/oclif/parsers.d.ts.map +0 -1
  481. package/dist/cli/oclif/parsers.js.map +0 -1
  482. package/dist/cli/oclif/projection.d.ts.map +0 -1
  483. package/dist/cli/oclif/projection.js.map +0 -1
  484. package/dist/cli/oclif/run.d.ts.map +0 -1
  485. package/dist/cli/oclif/run.js.map +0 -1
  486. package/dist/diagnosis/observe-mapper.d.ts.map +0 -1
  487. package/dist/diagnosis/observe-mapper.js.map +0 -1
  488. package/dist/diagnosis/observe-producer.d.ts.map +0 -1
  489. package/dist/diagnosis/observe-producer.js.map +0 -1
  490. package/dist/diagnosis/studio-projection.d.ts.map +0 -1
  491. package/dist/diagnosis/studio-projection.js.map +0 -1
  492. package/dist/diagnosis/types.d.ts.map +0 -1
  493. package/dist/diagnosis/types.js.map +0 -1
  494. package/dist/doctor/fixer.d.ts.map +0 -1
  495. package/dist/doctor/fixer.js.map +0 -1
  496. package/dist/doctor/health/builtin-dimensions.d.ts.map +0 -1
  497. package/dist/doctor/health/builtin-dimensions.js.map +0 -1
  498. package/dist/doctor/health/composer.d.ts.map +0 -1
  499. package/dist/doctor/health/composer.js.map +0 -1
  500. package/dist/doctor/health/dimension-registry.d.ts.map +0 -1
  501. package/dist/doctor/health/dimension-registry.js.map +0 -1
  502. package/dist/doctor/health/dimension-spec.d.ts.map +0 -1
  503. package/dist/doctor/health/dimension-spec.js.map +0 -1
  504. package/dist/doctor/health/load-custom-dimensions.d.ts.map +0 -1
  505. package/dist/doctor/health/load-custom-dimensions.js.map +0 -1
  506. package/dist/doctor/health/parser.d.ts.map +0 -1
  507. package/dist/doctor/health/parser.js.map +0 -1
  508. package/dist/doctor/health/prompt-builder.d.ts.map +0 -1
  509. package/dist/doctor/health/prompt-builder.js.map +0 -1
  510. package/dist/doctor/health/register.d.ts.map +0 -1
  511. package/dist/doctor/health/register.js.map +0 -1
  512. package/dist/doctor/index.d.ts.map +0 -1
  513. package/dist/doctor/index.js.map +0 -1
  514. package/dist/doctor/messages.d.ts.map +0 -1
  515. package/dist/doctor/messages.js.map +0 -1
  516. package/dist/doctor/preflight.d.ts.map +0 -1
  517. package/dist/doctor/preflight.js.map +0 -1
  518. package/dist/doctor/renderer.d.ts.map +0 -1
  519. package/dist/doctor/renderer.js.map +0 -1
  520. package/dist/doctor/rules.d.ts.map +0 -1
  521. package/dist/doctor/rules.js.map +0 -1
  522. package/dist/eval-core/bootstrap.d.ts.map +0 -1
  523. package/dist/eval-core/bootstrap.js.map +0 -1
  524. package/dist/eval-core/cache.d.ts.map +0 -1
  525. package/dist/eval-core/cache.js.map +0 -1
  526. package/dist/eval-core/comparability.d.ts.map +0 -1
  527. package/dist/eval-core/comparability.js.map +0 -1
  528. package/dist/eval-core/dependency-checker.d.ts.map +0 -1
  529. package/dist/eval-core/dependency-checker.js.map +0 -1
  530. package/dist/eval-core/evaluation-execution.d.ts.map +0 -1
  531. package/dist/eval-core/evaluation-execution.js.map +0 -1
  532. package/dist/eval-core/evaluation-job.d.ts.map +0 -1
  533. package/dist/eval-core/evaluation-job.js.map +0 -1
  534. package/dist/eval-core/evaluation-reporting.d.ts.map +0 -1
  535. package/dist/eval-core/evaluation-reporting.js.map +0 -1
  536. package/dist/eval-core/execution-strategy.d.ts.map +0 -1
  537. package/dist/eval-core/execution-strategy.js.map +0 -1
  538. package/dist/eval-core/fact-checker.d.ts.map +0 -1
  539. package/dist/eval-core/fact-checker.js.map +0 -1
  540. package/dist/eval-core/layer-gates.d.ts.map +0 -1
  541. package/dist/eval-core/layer-gates.js.map +0 -1
  542. package/dist/eval-core/mocks-runtime.d.ts.map +0 -1
  543. package/dist/eval-core/mocks-runtime.js.map +0 -1
  544. package/dist/eval-core/schema.d.ts.map +0 -1
  545. package/dist/eval-core/schema.js.map +0 -1
  546. package/dist/eval-core/statistics.d.ts.map +0 -1
  547. package/dist/eval-core/statistics.js.map +0 -1
  548. package/dist/eval-core/task-planner.d.ts.map +0 -1
  549. package/dist/eval-core/task-planner.js.map +0 -1
  550. package/dist/eval-core/verdict.d.ts.map +0 -1
  551. package/dist/eval-core/verdict.js.map +0 -1
  552. package/dist/eval-workflows/batch-evaluation-workflow.d.ts.map +0 -1
  553. package/dist/eval-workflows/batch-evaluation-workflow.js.map +0 -1
  554. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts.map +0 -1
  555. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js.map +0 -1
  556. package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts.map +0 -1
  557. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js.map +0 -1
  558. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts.map +0 -1
  559. package/dist/eval-workflows/evaluation-pipeline/run-state.js.map +0 -1
  560. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts.map +0 -1
  561. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js.map +0 -1
  562. package/dist/eval-workflows/evaluation-pipeline.d.ts.map +0 -1
  563. package/dist/eval-workflows/evaluation-pipeline.js.map +0 -1
  564. package/dist/eval-workflows/evaluation-preparation.d.ts.map +0 -1
  565. package/dist/eval-workflows/evaluation-preparation.js.map +0 -1
  566. package/dist/eval-workflows/messages.d.ts.map +0 -1
  567. package/dist/eval-workflows/messages.js.map +0 -1
  568. package/dist/eval-workflows/run-evaluation.d.ts.map +0 -1
  569. package/dist/eval-workflows/run-evaluation.js.map +0 -1
  570. package/dist/executors/anthropic-api.d.ts.map +0 -1
  571. package/dist/executors/anthropic-api.js.map +0 -1
  572. package/dist/executors/claude-cli.d.ts.map +0 -1
  573. package/dist/executors/claude-cli.js.map +0 -1
  574. package/dist/executors/claude-sdk-trace.d.ts.map +0 -1
  575. package/dist/executors/claude-sdk-trace.js.map +0 -1
  576. package/dist/executors/claude-sdk.d.ts.map +0 -1
  577. package/dist/executors/claude-sdk.js.map +0 -1
  578. package/dist/executors/codex-cli-trace.d.ts.map +0 -1
  579. package/dist/executors/codex-cli-trace.js.map +0 -1
  580. package/dist/executors/codex-cli.d.ts.map +0 -1
  581. package/dist/executors/codex-cli.js.map +0 -1
  582. package/dist/executors/codex-sdk.d.ts.map +0 -1
  583. package/dist/executors/codex-sdk.js.map +0 -1
  584. package/dist/executors/gemini.d.ts.map +0 -1
  585. package/dist/executors/gemini.js.map +0 -1
  586. package/dist/executors/index.d.ts.map +0 -1
  587. package/dist/executors/index.js.map +0 -1
  588. package/dist/executors/openai-api.d.ts.map +0 -1
  589. package/dist/executors/openai-api.js.map +0 -1
  590. package/dist/executors/runtime-fingerprint.d.ts.map +0 -1
  591. package/dist/executors/runtime-fingerprint.js.map +0 -1
  592. package/dist/executors/script.d.ts.map +0 -1
  593. package/dist/executors/script.js.map +0 -1
  594. package/dist/executors/shared.d.ts.map +0 -1
  595. package/dist/executors/shared.js.map +0 -1
  596. package/dist/grading/assertions.d.ts.map +0 -1
  597. package/dist/grading/assertions.js.map +0 -1
  598. package/dist/grading/debias-validate.d.ts.map +0 -1
  599. package/dist/grading/debias-validate.js.map +0 -1
  600. package/dist/grading/diagnostic.d.ts.map +0 -1
  601. package/dist/grading/diagnostic.js.map +0 -1
  602. package/dist/grading/gold-cli.d.ts.map +0 -1
  603. package/dist/grading/gold-cli.js.map +0 -1
  604. package/dist/grading/gold-dataset.d.ts.map +0 -1
  605. package/dist/grading/gold-dataset.js.map +0 -1
  606. package/dist/grading/human-gold.d.ts.map +0 -1
  607. package/dist/grading/human-gold.js.map +0 -1
  608. package/dist/grading/index.d.ts.map +0 -1
  609. package/dist/grading/index.js.map +0 -1
  610. package/dist/grading/judge.d.ts.map +0 -1
  611. package/dist/grading/judge.js.map +0 -1
  612. package/dist/grading/layered-scores.d.ts.map +0 -1
  613. package/dist/grading/layered-scores.js.map +0 -1
  614. package/dist/inputs/eval-config.d.ts.map +0 -1
  615. package/dist/inputs/eval-config.js.map +0 -1
  616. package/dist/inputs/load-samples.d.ts.map +0 -1
  617. package/dist/inputs/load-samples.js.map +0 -1
  618. package/dist/inputs/mcp-resolver.d.ts.map +0 -1
  619. package/dist/inputs/mcp-resolver.js.map +0 -1
  620. package/dist/inputs/skill-loader.d.ts.map +0 -1
  621. package/dist/inputs/skill-loader.js.map +0 -1
  622. package/dist/inputs/url-fetcher.d.ts.map +0 -1
  623. package/dist/inputs/url-fetcher.js.map +0 -1
  624. package/dist/observability/experience-frontmatter.d.ts.map +0 -1
  625. package/dist/observability/experience-frontmatter.js.map +0 -1
  626. package/dist/observability/experience.d.ts.map +0 -1
  627. package/dist/observability/experience.js.map +0 -1
  628. package/dist/observability/feedback-matchers.d.ts.map +0 -1
  629. package/dist/observability/feedback-matchers.js.map +0 -1
  630. package/dist/observability/feedback-projection.d.ts.map +0 -1
  631. package/dist/observability/feedback-projection.js.map +0 -1
  632. package/dist/observability/inbox-view-model.d.ts.map +0 -1
  633. package/dist/observability/inbox-view-model.js.map +0 -1
  634. package/dist/observability/inbox.d.ts.map +0 -1
  635. package/dist/observability/inbox.js.map +0 -1
  636. package/dist/observability/problem-patterns.d.ts.map +0 -1
  637. package/dist/observability/problem-patterns.js.map +0 -1
  638. package/dist/observability/resolved-review.d.ts.map +0 -1
  639. package/dist/observability/resolved-review.js.map +0 -1
  640. package/dist/observability/review-state.d.ts.map +0 -1
  641. package/dist/observability/review-state.js.map +0 -1
  642. package/dist/observability/skill-chain-advisories.d.ts.map +0 -1
  643. package/dist/observability/skill-chain-advisories.js.map +0 -1
  644. package/dist/observability/skill-chain.d.ts.map +0 -1
  645. package/dist/observability/skill-chain.js.map +0 -1
  646. package/dist/observability/skill-health-analyzer.d.ts.map +0 -1
  647. package/dist/observability/skill-health-analyzer.js.map +0 -1
  648. package/dist/observability/soft-standards/constants.d.ts.map +0 -1
  649. package/dist/observability/soft-standards/constants.js.map +0 -1
  650. package/dist/observability/soft-standards/index.d.ts.map +0 -1
  651. package/dist/observability/soft-standards/index.js.map +0 -1
  652. package/dist/observability/soft-standards/llm-extractor.d.ts.map +0 -1
  653. package/dist/observability/soft-standards/llm-extractor.js.map +0 -1
  654. package/dist/observability/soft-standards/runtime-evaluator.d.ts.map +0 -1
  655. package/dist/observability/soft-standards/runtime-evaluator.js.map +0 -1
  656. package/dist/observability/soft-standards/skill-standards-store.d.ts.map +0 -1
  657. package/dist/observability/soft-standards/skill-standards-store.js.map +0 -1
  658. package/dist/observability/soft-standards/types.d.ts.map +0 -1
  659. package/dist/observability/soft-standards/types.js.map +0 -1
  660. package/dist/observability/text-signals.d.ts.map +0 -1
  661. package/dist/observability/text-signals.js.map +0 -1
  662. package/dist/observability/trace-adapter.d.ts.map +0 -1
  663. package/dist/observability/trace-adapter.js.map +0 -1
  664. package/dist/observability/trace-attribution.d.ts.map +0 -1
  665. package/dist/observability/trace-attribution.js.map +0 -1
  666. package/dist/observability/trace-segmenter.d.ts.map +0 -1
  667. package/dist/observability/trace-segmenter.js.map +0 -1
  668. package/dist/observability/trace-source.d.ts.map +0 -1
  669. package/dist/observability/trace-source.js.map +0 -1
  670. package/dist/renderer/html-renderer.d.ts.map +0 -1
  671. package/dist/renderer/html-renderer.js.map +0 -1
  672. package/dist/renderer/layout.d.ts.map +0 -1
  673. package/dist/renderer/layout.js.map +0 -1
  674. package/dist/renderer/observation-inbox/helpers.d.ts.map +0 -1
  675. package/dist/renderer/observation-inbox/helpers.js.map +0 -1
  676. package/dist/renderer/observation-inbox/styles.d.ts.map +0 -1
  677. package/dist/renderer/observation-inbox/styles.js.map +0 -1
  678. package/dist/renderer/observation-inbox-renderer.d.ts.map +0 -1
  679. package/dist/renderer/observation-inbox-renderer.js.map +0 -1
  680. package/dist/renderer/skill-detail-renderer.d.ts.map +0 -1
  681. package/dist/renderer/skill-detail-renderer.js.map +0 -1
  682. package/dist/renderer/skill-health-renderer.d.ts.map +0 -1
  683. package/dist/renderer/skill-health-renderer.js.map +0 -1
  684. package/dist/renderer/skill-list-renderer.d.ts.map +0 -1
  685. package/dist/renderer/skill-list-renderer.js.map +0 -1
  686. package/dist/renderer/summary.d.ts.map +0 -1
  687. package/dist/renderer/summary.js.map +0 -1
  688. package/dist/renderer/table.d.ts.map +0 -1
  689. package/dist/renderer/table.js.map +0 -1
  690. package/dist/renderer/test-view.d.ts.map +0 -1
  691. package/dist/renderer/test-view.js.map +0 -1
  692. package/dist/renderer/trends.d.ts.map +0 -1
  693. package/dist/renderer/trends.js.map +0 -1
  694. package/dist/server/job-store.d.ts.map +0 -1
  695. package/dist/server/job-store.js.map +0 -1
  696. package/dist/server/report-server.d.ts.map +0 -1
  697. package/dist/server/report-server.js.map +0 -1
  698. package/dist/server/report-store.d.ts.map +0 -1
  699. package/dist/server/report-store.js.map +0 -1
  700. package/dist/server/skill-index.d.ts.map +0 -1
  701. package/dist/server/skill-index.js.map +0 -1
  702. package/dist/server/skill-insights.d.ts.map +0 -1
  703. package/dist/server/skill-insights.js.map +0 -1
  704. package/dist/shared/hard-rules.d.ts.map +0 -1
  705. package/dist/shared/hard-rules.js.map +0 -1
  706. package/dist/shared/llm-prompts/index.d.ts.map +0 -1
  707. package/dist/shared/llm-prompts/index.js.map +0 -1
  708. package/dist/shared/llm-prompts/skill-health.d.ts.map +0 -1
  709. package/dist/shared/llm-prompts/skill-health.js.map +0 -1
  710. package/dist/shared/time.d.ts.map +0 -1
  711. package/dist/shared/time.js.map +0 -1
  712. package/dist/shared/tool-search.d.ts.map +0 -1
  713. package/dist/shared/tool-search.js.map +0 -1
  714. package/dist/types/dependencies.d.ts.map +0 -1
  715. package/dist/types/dependencies.js.map +0 -1
  716. package/dist/types/diagnosis.d.ts.map +0 -1
  717. package/dist/types/diagnosis.js.map +0 -1
  718. package/dist/types/doctor.d.ts.map +0 -1
  719. package/dist/types/doctor.js.map +0 -1
  720. package/dist/types/eval.d.ts.map +0 -1
  721. package/dist/types/eval.js.map +0 -1
  722. package/dist/types/executor.d.ts.map +0 -1
  723. package/dist/types/executor.js.map +0 -1
  724. package/dist/types/index.d.ts.map +0 -1
  725. package/dist/types/index.js.map +0 -1
  726. package/dist/types/judge.d.ts.map +0 -1
  727. package/dist/types/judge.js.map +0 -1
  728. package/dist/types/observability.d.ts.map +0 -1
  729. package/dist/types/observability.js.map +0 -1
  730. package/dist/types/report.d.ts.map +0 -1
  731. package/dist/types/report.js.map +0 -1
  732. package/dist/types/shared.d.ts.map +0 -1
  733. package/dist/types/shared.js.map +0 -1
  734. package/dist/types/skill-index.d.ts.map +0 -1
  735. package/dist/types/skill-index.js.map +0 -1
  736. package/dist/types/storage.d.ts.map +0 -1
  737. package/dist/types/storage.js.map +0 -1
  738. package/dist/util/safe-slice.d.ts.map +0 -1
  739. package/dist/util/safe-slice.js.map +0 -1
@@ -21,6 +21,19 @@ interface HoldoutSplit {
21
21
  trainIds: Set<string>;
22
22
  holdoutIds: Set<string>;
23
23
  }
24
+ /** A train / val / test partition. `val` drives the accept decision; `test` is
25
+ * locked — never seen during the loop, read once at the end for an unbiased
26
+ * generalization score. */
27
+ interface TrainValTestSplit {
28
+ trainIds: Set<string>;
29
+ valIds: Set<string>;
30
+ testIds: Set<string>;
31
+ }
32
+ /** Below this many decision (val) samples the bootstrap diff CI almost never
33
+ * excludes 0 for realistic effect sizes, so the significance gate would reject
34
+ * every candidate. Under that floor evolve degrades to the point-estimate accept
35
+ * and flags `gate.underpowered`. */
36
+ export declare const MIN_GATE_SAMPLES = 8;
24
37
  /**
25
38
  * Deterministically split sample ids into train / holdout by `ratio` (fraction
26
39
  * held out). Holdout members are picked at an even stride so the partition is
@@ -29,6 +42,44 @@ interface HoldoutSplit {
29
42
  * MIN_HOLDOUT_SUBSET — the caller then scores on the full set.
30
43
  */
31
44
  export declare function splitHoldout(sampleIds: string[], ratio: number): HoldoutSplit | null;
45
+ /**
46
+ * Deterministically split sample ids into train / val / test. `val` is carved
47
+ * first at an even stride; `test` is carved at an even stride over what remains,
48
+ * so the three sets are disjoint and stable across rounds/runs (no RNG). Returns
49
+ * null when either ratio ≤ 0 or any of the three sides would drop below
50
+ * MIN_HOLDOUT_SUBSET — the caller then degrades to a 2-way (or full-set) split.
51
+ */
52
+ export declare function splitTrainValTest(sampleIds: string[], valRatio: number, testRatio: number): TrainValTestSplit | null;
53
+ export interface AcceptDecision {
54
+ accepted: boolean;
55
+ /** Diff CI (candidate − best) when the gate ran; absent when it degraded. */
56
+ diffCI?: {
57
+ low: number;
58
+ high: number;
59
+ estimate: number;
60
+ significant: boolean;
61
+ };
62
+ /** True when the gate was requested but the decision set was below MIN_GATE_SAMPLES,
63
+ * so the decision degraded to the point-estimate comparison. */
64
+ underpowered: boolean;
65
+ }
66
+ /**
67
+ * The accept decision for one round. With the significance gate on and enough
68
+ * decision samples, a candidate is accepted only when it is **significantly** above
69
+ * the current best's fresh re-eval (`bootstrapDiffCI(...).significant && estimate > 0`)
70
+ * AND its decision score actually beats the recorded best (`pointCand > pointBest`).
71
+ * The second clause preserves evolve's monotonic invariant: `bestScore` never
72
+ * decreases. Without it, an unlucky (noise-low) re-eval of the current best could let
73
+ * a candidate that is significantly above that re-eval — yet still below the recorded
74
+ * best — win and overwrite the best downward. Off, or under-powered, the gate degrades
75
+ * to the legacy point-estimate comparison alone. Pure (modulo the seeded bootstrap) so
76
+ * the core behavior is unit-testable.
77
+ */
78
+ export declare function decideAccept(bestScores: number[], candScores: number[], pointBest: number, pointCand: number, opts: {
79
+ significanceGate: boolean;
80
+ alpha: number;
81
+ seed: number;
82
+ }): AcceptDecision;
32
83
  /**
33
84
  * A view of `report` whose results are restricted to `sampleIds`. Used to keep the
34
85
  * holdout split out of the sample-fixer: under an active holdout, only training-split
@@ -38,7 +89,22 @@ export declare function splitHoldout(sampleIds: string[], ratio: number): Holdou
38
89
  */
39
90
  export declare function restrictReportToSamples(report: Report, sampleIds: Set<string>): Report;
40
91
  export declare function allNonTripwireAssertionsPass(report: Report, variantKey: string): boolean;
41
- export declare function buildImprovementPrompt(skillContent: string, score: number, weakSamples: WeakSample[]): string;
92
+ interface EditDelta {
93
+ /** Symmetric line difference (added + removed unique lines) over original line count. */
94
+ ratio: number;
95
+ /** Absolute count of added + removed unique lines. */
96
+ changedLines: number;
97
+ /** Compact `+`/`-` summary of the changed lines, truncated. */
98
+ summary: string;
99
+ }
100
+ /**
101
+ * How a candidate differs from the current best, by trimmed non-empty line sets.
102
+ * `ratio` drives the edit budget; `summary` feeds the rejected-edit memory so the
103
+ * improver doesn't re-propose changes that already failed. Order-insensitive and
104
+ * O(n) over small skill files.
105
+ */
106
+ export declare function computeEditDelta(before: string, after: string, maxSummaryLines?: number): EditDelta;
107
+ export declare function buildImprovementPrompt(skillContent: string, score: number, weakSamples: WeakSample[], rejectedEdits?: string[]): string;
42
108
  /** @deprecated Use ProgressCallback from evaluation-core.ts */
43
109
  export type EvolveProgressInfo = Parameters<ProgressCallback>[0];
44
110
  export interface EvolveRoundProgressInfo {
@@ -54,6 +120,10 @@ export interface EvolveRoundProgressInfo {
54
120
  costReported?: boolean;
55
121
  error?: string;
56
122
  reused?: boolean;
123
+ /** When the significance gate ran: whether the candidate's gain was significant.
124
+ * False on a rejected round means "score rose but within noise" — lets the CLI
125
+ * explain an otherwise-confusing `(+0.0x) ✗ REJECT`. Undefined = gate didn't run. */
126
+ significant?: boolean;
57
127
  }
58
128
  interface EvolveOptions {
59
129
  skillPath: string;
@@ -88,20 +158,56 @@ interface EvolveOptions {
88
158
  * so the skill is never tuned to the samples that judge it. Too small a split
89
159
  * (either side < MIN_HOLDOUT_SUBSET) falls back to full-set scoring + a warning. */
90
160
  holdoutRatio?: number;
161
+ /** Statistically gate acceptance: a candidate is accepted only when its
162
+ * per-sample composite is **significantly** above the current best on the
163
+ * decision (val) set — `bootstrapDiffCI(...).significant && estimate > 0` —
164
+ * not merely numerically higher. Default true (rejecting improvements
165
+ * indistinguishable from judge noise is the point). Below MIN_GATE_SAMPLES
166
+ * decision samples the gate is underpowered and degrades to the point-estimate
167
+ * comparison + a warning. Set false to force the legacy point-estimate accept. */
168
+ significanceGate?: boolean;
169
+ /** Significance level for the accept gate's diff CI. Default 0.05 (95% CI). */
170
+ significanceAlpha?: number;
171
+ /** Fraction of samples locked away as a **test** set (0..1). Default 0 = off.
172
+ * Only honored alongside `holdoutRatio` > 0 (test needs a separate val set to
173
+ * decide on). The test split is never seen during the loop — not by weak-sample
174
+ * extraction, not by the accept gate — and is read exactly once at the end for an
175
+ * unbiased `generalizationScore`. Too small a 3-way split degrades to 2-way. */
176
+ testRatio?: number;
177
+ /** Max fraction of skill lines a single round may change before the candidate is
178
+ * rejected **without paying for evaluation**. Default 0.2 (matches the "≤20%"
179
+ * the improvement prompt already asks for — this enforces it). A small floor
180
+ * always permits a handful of lines so tiny skills aren't frozen. Set 0 to disable. */
181
+ editBudget?: number;
182
+ /** Feed rejected candidate edits back into the next round's improvement prompt
183
+ * ("these were tried and did not help — don't repeat them"). Default true. */
184
+ rejectMemory?: boolean;
91
185
  onProgress?: ProgressCallback | null;
92
186
  onRoundProgress?: ((progress: EvolveRoundProgressInfo) => void) | null;
93
187
  }
94
188
  interface TrajectoryEntry {
95
189
  round: number;
96
- /** Accept-decision score: holdout composite when holdout is active, else full-set. */
190
+ /** Accept-decision score: val composite when a holdout split is active, else full-set. */
97
191
  score: number;
98
192
  delta: number;
99
193
  accepted: boolean;
100
194
  costUSD: number;
101
- /** Present when holdout is active: the training-split composite (improvement signal). */
195
+ /** Present when a holdout split is active: the training-split composite (improvement signal). */
102
196
  trainScore?: number;
103
- /** Present when holdout is active: the holdout-split composite (== score). */
197
+ /** Present when a holdout split is active: the val-split composite (== score). */
104
198
  holdoutScore?: number;
199
+ /** Significance-gate diff CI (candidate − current best) on the decision set, when
200
+ * the gate was powered enough to run. `significant` 决定接受。 */
201
+ diffCI?: {
202
+ low: number;
203
+ high: number;
204
+ estimate: number;
205
+ significant: boolean;
206
+ };
207
+ /** Fraction of skill lines this candidate changed vs the current best. */
208
+ editRatio?: number;
209
+ /** True when the candidate was rejected by the edit budget before evaluation. */
210
+ rejectedPreEval?: boolean;
105
211
  }
106
212
  export interface EvolveResult {
107
213
  startScore: number;
@@ -127,6 +233,25 @@ export interface EvolveResult {
127
233
  holdoutCount: number;
128
234
  disabled?: boolean;
129
235
  };
236
+ /** Locked-test split summary. `disabled` is true when `--test-ratio` was requested
237
+ * but the 3-way split was too small, so evolve fell back to a 2-way holdout and
238
+ * produced no generalization score. */
239
+ test?: {
240
+ ratio: number;
241
+ count: number;
242
+ disabled?: boolean;
243
+ };
244
+ /** Unbiased composite of the best skill on the locked test set — the headline honest
245
+ * number. Present only when a 3-way split was active. The test set never influenced
246
+ * selection or weak-sample extraction, so this is an out-of-sample estimate. */
247
+ generalizationScore?: number;
248
+ /** Accept-gate summary. `underpowered` = the decision set was below MIN_GATE_SAMPLES
249
+ * at least once, so the gate degraded to the point-estimate comparison + warned. */
250
+ gate?: {
251
+ enabled: boolean;
252
+ alpha: number;
253
+ underpowered?: boolean;
254
+ };
130
255
  trajectory: TrajectoryEntry[];
131
256
  bestSkillPath: string;
132
257
  allVersions: string[];
@@ -138,6 +263,5 @@ export interface RoundReport {
138
263
  report: Report;
139
264
  }
140
265
  export declare function mergeEvolveReports(roundReports: RoundReport[], skillName: string, totalCostUSD: number, samples?: Sample[], skillPath?: string): Report;
141
- export declare function evolveSkill({ skillPath, samplesPath, rounds, target, stopOnAssertionsPass, autoFixSamples, sampleFixMaxAttempts, reuseLatestEval, model, judgeModels, improveModel, improveMode, executorName, concurrency, timeoutMs, skipConnectivity, effort, noDiagnostic, skipDoctor, holdoutRatio, onProgress, onRoundProgress, }: EvolveOptions): Promise<EvolveResult>;
266
+ export declare function evolveSkill({ skillPath, samplesPath, rounds, target, stopOnAssertionsPass, autoFixSamples, sampleFixMaxAttempts, reuseLatestEval, model, judgeModels, improveModel, improveMode, executorName, concurrency, timeoutMs, skipConnectivity, effort, noDiagnostic, skipDoctor, holdoutRatio, significanceGate, significanceAlpha, testRatio, editBudget, rejectMemory, onProgress, onRoundProgress, }: EvolveOptions): Promise<EvolveResult>;
142
267
  export {};
143
- //# sourceMappingURL=evolver.d.ts.map
@@ -6,7 +6,9 @@ import { persistReport, DEFAULT_OUTPUT_DIR, generateRunId, hashString } from '..
6
6
  import { createFileStore } from '../server/report-store.js';
7
7
  import { analyzeResults } from '../analysis/report-diagnostics.js';
8
8
  import { loadSamples } from '../inputs/load-samples.js';
9
+ import { hashArtifactSource } from '../inputs/content-hash.js';
9
10
  import { buildVariantSummary } from '../eval-core/schema.js';
11
+ import { bootstrapDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
10
12
  import { fixSamples } from './sample-fixer.js';
11
13
  const IMPROVE_SYSTEM_PROMPT = `你是一个 AI 提示词改进专家。你的任务是分析评测结果中的薄弱环节,针对性地改进 skill(系统提示词),使其在评测中获得更高的分数。
12
14
 
@@ -87,9 +89,13 @@ function singleVariantReport(report, variantKey) {
87
89
  async function findReusableBaselineReport(opts) {
88
90
  const store = createFileStore(DEFAULT_OUTPUT_DIR);
89
91
  const { samples } = loadSamples(opts.samplesPath);
90
- const artifactHash = hashString(opts.skillContent);
92
+ const artifactHash = opts.artifactHash;
91
93
  const reports = await store.findByArtifactHash(artifactHash);
92
94
  for (const report of reports) {
95
+ // schemaVersion < 2 的报告 artifactHashes 是旧文本哈,与当前树哈不同空间:即便值偶合也不该复用
96
+ // (口径不同会让 lineage 串错身份)。直接跳过,让旧 baseline 重跑出树哈报告。
97
+ if ((report.meta.schemaVersion ?? 0) < 2)
98
+ continue;
93
99
  if (report.meta.model !== opts.model || report.meta.executor !== opts.executorName)
94
100
  continue;
95
101
  if ((report.meta.effort ?? undefined) !== (opts.effort ?? undefined))
@@ -156,9 +162,25 @@ export function extractWeakSamples(report, variantKey, count = 5, sampleIdFilter
156
162
  .sort((a, b) => a.compositeScore - b.compositeScore)
157
163
  .slice(0, count);
158
164
  }
159
- /** Below this many samples on either side, a holdout split is too small to be
160
- * meaningful — evolve falls back to full-set scoring and warns. */
165
+ /** Below this many samples on any side, a split is too small to be meaningful —
166
+ * evolve falls back to full-set scoring and warns. */
161
167
  const MIN_HOLDOUT_SUBSET = 3;
168
+ /** Below this many decision (val) samples the bootstrap diff CI almost never
169
+ * excludes 0 for realistic effect sizes, so the significance gate would reject
170
+ * every candidate. Under that floor evolve degrades to the point-estimate accept
171
+ * and flags `gate.underpowered`. */
172
+ export const MIN_GATE_SAMPLES = 8;
173
+ /** Pick `count` ids at an even stride across `ids` (deterministic, no RNG) so the
174
+ * picked subset is representative of the ordering and stable across rounds/runs. */
175
+ function pickByStride(ids, count) {
176
+ const picked = new Set();
177
+ if (count <= 0)
178
+ return picked;
179
+ const stride = ids.length / count;
180
+ for (let k = 0; k < count; k++)
181
+ picked.add(ids[Math.floor(k * stride)]);
182
+ return picked;
183
+ }
162
184
  /**
163
185
  * Deterministically split sample ids into train / holdout by `ratio` (fraction
164
186
  * held out). Holdout members are picked at an even stride so the partition is
@@ -173,14 +195,31 @@ export function splitHoldout(sampleIds, ratio) {
173
195
  const trainCount = sampleIds.length - holdoutCount;
174
196
  if (holdoutCount < MIN_HOLDOUT_SUBSET || trainCount < MIN_HOLDOUT_SUBSET)
175
197
  return null;
176
- const stride = sampleIds.length / holdoutCount;
177
- const holdoutIds = new Set();
178
- for (let k = 0; k < holdoutCount; k++) {
179
- holdoutIds.add(sampleIds[Math.floor(k * stride)]);
180
- }
198
+ const holdoutIds = pickByStride(sampleIds, holdoutCount);
181
199
  const trainIds = new Set(sampleIds.filter((id) => !holdoutIds.has(id)));
182
200
  return { trainIds, holdoutIds };
183
201
  }
202
+ /**
203
+ * Deterministically split sample ids into train / val / test. `val` is carved
204
+ * first at an even stride; `test` is carved at an even stride over what remains,
205
+ * so the three sets are disjoint and stable across rounds/runs (no RNG). Returns
206
+ * null when either ratio ≤ 0 or any of the three sides would drop below
207
+ * MIN_HOLDOUT_SUBSET — the caller then degrades to a 2-way (or full-set) split.
208
+ */
209
+ export function splitTrainValTest(sampleIds, valRatio, testRatio) {
210
+ if (!(valRatio > 0) || !(testRatio > 0) || sampleIds.length === 0)
211
+ return null;
212
+ const valCount = Math.round(sampleIds.length * valRatio);
213
+ const testCount = Math.round(sampleIds.length * testRatio);
214
+ const trainCount = sampleIds.length - valCount - testCount;
215
+ if (valCount < MIN_HOLDOUT_SUBSET || testCount < MIN_HOLDOUT_SUBSET || trainCount < MIN_HOLDOUT_SUBSET)
216
+ return null;
217
+ const valIds = pickByStride(sampleIds, valCount);
218
+ const remaining = sampleIds.filter((id) => !valIds.has(id));
219
+ const testIds = pickByStride(remaining, testCount);
220
+ const trainIds = new Set(sampleIds.filter((id) => !valIds.has(id) && !testIds.has(id)));
221
+ return { trainIds, valIds, testIds };
222
+ }
184
223
  /**
185
224
  * Mean composite over the subset of a report's results whose sample_id is in
186
225
  * `ids`, using the same aggregation as the full-run summary
@@ -200,6 +239,47 @@ function subsetCompositeScore(report, variantKey, ids) {
200
239
  return 0;
201
240
  return buildVariantSummary(entries).avgCompositeScore ?? 0;
202
241
  }
242
+ /**
243
+ * Per-sample composite scores over the subset of a report's results whose
244
+ * sample_id is in `ids`, in result order. Feeds `bootstrapDiffCI` for the
245
+ * significance accept gate — the array (not the mean) is what the bootstrap
246
+ * resamples. Entries without a numeric compositeScore are skipped.
247
+ */
248
+ function perSampleComposite(report, variantKey, ids) {
249
+ const scores = [];
250
+ for (const r of report.results) {
251
+ if (!ids.has(r.sample_id))
252
+ continue;
253
+ const v = r.variants[variantKey];
254
+ if (v && typeof v.compositeScore === 'number')
255
+ scores.push(v.compositeScore);
256
+ }
257
+ return scores;
258
+ }
259
+ /**
260
+ * The accept decision for one round. With the significance gate on and enough
261
+ * decision samples, a candidate is accepted only when it is **significantly** above
262
+ * the current best's fresh re-eval (`bootstrapDiffCI(...).significant && estimate > 0`)
263
+ * AND its decision score actually beats the recorded best (`pointCand > pointBest`).
264
+ * The second clause preserves evolve's monotonic invariant: `bestScore` never
265
+ * decreases. Without it, an unlucky (noise-low) re-eval of the current best could let
266
+ * a candidate that is significantly above that re-eval — yet still below the recorded
267
+ * best — win and overwrite the best downward. Off, or under-powered, the gate degrades
268
+ * to the legacy point-estimate comparison alone. Pure (modulo the seeded bootstrap) so
269
+ * the core behavior is unit-testable.
270
+ */
271
+ export function decideAccept(bestScores, candScores, pointBest, pointCand, opts) {
272
+ const powered = bestScores.length >= MIN_GATE_SAMPLES && candScores.length >= MIN_GATE_SAMPLES;
273
+ if (opts.significanceGate && powered) {
274
+ const diff = bootstrapDiffCI(bestScores, candScores, opts.alpha, DEFAULT_BOOTSTRAP_SAMPLES, opts.seed);
275
+ return {
276
+ accepted: diff.significant && diff.estimate > 0 && pointCand > pointBest,
277
+ diffCI: { low: diff.low, high: diff.high, estimate: diff.estimate, significant: diff.significant },
278
+ underpowered: false,
279
+ };
280
+ }
281
+ return { accepted: pointCand > pointBest, underpowered: opts.significanceGate && !powered };
282
+ }
203
283
  /**
204
284
  * A view of `report` whose results are restricted to `sampleIds`. Used to keep the
205
285
  * holdout split out of the sample-fixer: under an active holdout, only training-split
@@ -336,7 +416,36 @@ async function autoFixSamplesAfterSkillRound(opts) {
336
416
  }
337
417
  return { fixedCount: result.fixedCount, costUSD: result.costUSD };
338
418
  }
339
- export function buildImprovementPrompt(skillContent, score, weakSamples) {
419
+ /** Number of skill lines below which the edit budget never trips — so a tiny skill
420
+ * isn't frozen by a percentage threshold that a few lines already blow past. */
421
+ const EDIT_BUDGET_FLOOR_LINES = 10;
422
+ /**
423
+ * How a candidate differs from the current best, by trimmed non-empty line sets.
424
+ * `ratio` drives the edit budget; `summary` feeds the rejected-edit memory so the
425
+ * improver doesn't re-propose changes that already failed. Order-insensitive and
426
+ * O(n) over small skill files.
427
+ */
428
+ export function computeEditDelta(before, after, maxSummaryLines = 12) {
429
+ const beforeArr = before.split('\n').map((l) => l.trim()).filter(Boolean);
430
+ const afterArr = after.split('\n').map((l) => l.trim()).filter(Boolean);
431
+ const beforeSet = new Set(beforeArr);
432
+ const afterSet = new Set(afterArr);
433
+ const added = [...afterSet].filter((l) => !beforeSet.has(l));
434
+ const removed = [...beforeSet].filter((l) => !afterSet.has(l));
435
+ const changedLines = added.length + removed.length;
436
+ const ratio = changedLines / Math.max(beforeArr.length, 1);
437
+ const parts = [];
438
+ for (const l of added.slice(0, maxSummaryLines))
439
+ parts.push(`+ ${l}`);
440
+ if (added.length > maxSummaryLines)
441
+ parts.push(`+ …(其余 +${added.length - maxSummaryLines} 行)`);
442
+ for (const l of removed.slice(0, maxSummaryLines))
443
+ parts.push(`- ${l}`);
444
+ if (removed.length > maxSummaryLines)
445
+ parts.push(`- …(其余 -${removed.length - maxSummaryLines} 行)`);
446
+ return { ratio, changedLines, summary: parts.join('\n') || '(无文本差异)' };
447
+ }
448
+ export function buildImprovementPrompt(skillContent, score, weakSamples, rejectedEdits) {
340
449
  const weakDetails = weakSamples.map((s) => {
341
450
  const parts = [`### ${s.sample_id}(${s.compositeScore}/5.0)`];
342
451
  if (s.llmReason)
@@ -365,13 +474,16 @@ export function buildImprovementPrompt(skillContent, score, weakSamples) {
365
474
  }
366
475
  return parts.join('\n');
367
476
  }).join('\n\n');
477
+ const rejectedSection = rejectedEdits && rejectedEdits.length > 0
478
+ ? `\n\n## 已试过且未带来显著提升的改法(不要重复)\n\n${rejectedEdits.join('\n\n')}`
479
+ : '';
368
480
  return `## 当前 Skill(平均分: ${score.toFixed(2)}/5.0)
369
481
 
370
482
  ${skillContent}
371
483
 
372
484
  ## 低分用例分析
373
485
 
374
- ${weakDetails || '(无低分用例)'}`;
486
+ ${weakDetails || '(无低分用例)'}${rejectedSection}`;
375
487
  }
376
488
  function buildImprovementSuffix(mode, candidatePath) {
377
489
  if (mode === 'agent') {
@@ -434,7 +546,7 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
434
546
  }
435
547
  const runId = `evolve-${skillName}-${generateRunId([skillName]).split('-').slice(-2).join('-')}`;
436
548
  const report = {
437
- kind: 'evaluation',
549
+ reportKind: 'evaluation',
438
550
  id: runId,
439
551
  meta: {
440
552
  ...firstReport.meta,
@@ -458,7 +570,7 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
458
570
  report.analysis = analyzeResults(report, { samples });
459
571
  return report;
460
572
  }
461
- export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model = DEFAULT_MODEL, judgeModels, improveModel = DEFAULT_MODEL, improveMode = 'agent', executorName = 'claude', concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, holdoutRatio = 0, onProgress = null, onRoundProgress = null, }) {
573
+ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model = DEFAULT_MODEL, judgeModels, improveModel = DEFAULT_MODEL, improveMode = 'agent', executorName = 'claude', concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, holdoutRatio = 0, significanceGate = true, significanceAlpha = DEFAULT_BOOTSTRAP_ALPHA, testRatio = 0, editBudget = 0.2, rejectMemory = true, onProgress = null, onRoundProgress = null, }) {
462
574
  if (judgeModels && judgeModels.length > 1) {
463
575
  throw new Error('evolveSkill does not support multi-judge ensemble (received '
464
576
  + `${judgeModels.length} judges). Pass a single-judge array, e.g. `
@@ -478,28 +590,72 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
478
590
  if (!existsSync(absSamplesPath))
479
591
  throw new Error(`samples file not found: ${absSamplesPath}`);
480
592
  mkdirSync(evolveDir, { recursive: true });
481
- // Holdout split (opt-in). Computed once over the canonical sample order so it's
482
- // stable across rounds. When active, accept decisions use the holdout composite
483
- // and weak-sample extraction only sees the training split — the skill is never
484
- // tuned to the samples that decide whether it's accepted.
593
+ // Split (opt-in). Computed once over the canonical sample order so it's stable
594
+ // across rounds. `val` drives the accept decision; weak-sample extraction and the
595
+ // sample-fixer only ever see `train`; `test` is locked away — never seen during the
596
+ // loop — and read once at the end for an unbiased generalization score. With no
597
+ // holdout the decision runs on the full set (legacy). A too-small 3-way split
598
+ // degrades to 2-way, then to full-set.
485
599
  const allSampleIds = loadSamples(absSamplesPath).samples.map((s) => s.sample_id);
486
- const holdoutSplit = splitHoldout(allSampleIds, holdoutRatio);
600
+ const threeWay = (testRatio > 0 && holdoutRatio > 0) ? splitTrainValTest(allSampleIds, holdoutRatio, testRatio) : null;
601
+ const twoWay = (!threeWay && holdoutRatio > 0) ? splitHoldout(allSampleIds, holdoutRatio) : null;
602
+ const split = threeWay
603
+ ? { trainIds: threeWay.trainIds, valIds: threeWay.valIds, testIds: threeWay.testIds }
604
+ : twoWay
605
+ ? { trainIds: twoWay.trainIds, valIds: twoWay.holdoutIds, testIds: null }
606
+ : null;
487
607
  const holdoutInfo = holdoutRatio > 0
488
608
  ? {
489
609
  ratio: holdoutRatio,
490
- trainCount: holdoutSplit?.trainIds.size ?? allSampleIds.length,
491
- holdoutCount: holdoutSplit?.holdoutIds.size ?? 0,
492
- ...(holdoutSplit ? {} : { disabled: true }),
610
+ trainCount: split?.trainIds.size ?? allSampleIds.length,
611
+ holdoutCount: split?.valIds.size ?? 0,
612
+ ...(split ? {} : { disabled: true }),
493
613
  }
494
614
  : undefined;
495
- // Accept-decision score for a report's variant: holdout composite when the split
496
- // is active, otherwise the full-set composite (legacy behavior).
497
- const decisionScore = (report, key) => holdoutSplit
498
- ? subsetCompositeScore(report, key, holdoutSplit.holdoutIds)
615
+ // test 被请求(配了 --holdout-ratio)但 3-way 太小回退 → 标 disabled,别让用户
616
+ // 以为拿到了 locked-test 泛化分。
617
+ const testRequested = testRatio > 0 && holdoutRatio > 0;
618
+ const testInfo = threeWay
619
+ ? { ratio: testRatio, count: threeWay.testIds.size }
620
+ : testRequested ? { ratio: testRatio, count: 0, disabled: true } : undefined;
621
+ // Deterministic seed so the gate's CIs are reproducible across reruns (and
622
+ // assertable in tests). Derived from skill identity + sample count, parsed to a uint32.
623
+ const gateSeed = parseInt(hashString(`${skillName}:${allSampleIds.length}`).slice(0, 8), 16) >>> 0;
624
+ let gateUnderpowered = false;
625
+ // Accept-decision score for a report's variant: val composite when a split is
626
+ // active, otherwise the full-set composite (legacy behavior).
627
+ const decisionScore = (report, key) => split
628
+ ? subsetCompositeScore(report, key, split.valIds)
499
629
  : (report.summary[key]?.avgCompositeScore ?? 0);
500
- const trainScoreOf = (report, key) => holdoutSplit ? subsetCompositeScore(report, key, holdoutSplit.trainIds) : undefined;
630
+ const trainScoreOf = (report, key) => split ? subsetCompositeScore(report, key, split.trainIds) : undefined;
501
631
  // Per-round trajectory tail: train / holdout breakdown, only when split active.
502
- const splitScores = (report, key, decision) => holdoutSplit ? { trainScore: trainScoreOf(report, key), holdoutScore: decision } : {};
632
+ const splitScores = (report, key, decision) => split ? { trainScore: trainScoreOf(report, key), holdoutScore: decision } : {};
633
+ // Unbiased generalization: the best skill's composite on the locked test set,
634
+ // read once at the very end. Present only under a valid 3-way split; the test
635
+ // set never influenced selection or weak-sample extraction.
636
+ const buildGeneralization = () => {
637
+ if (!testInfo)
638
+ return {};
639
+ if (!threeWay || !split?.testIds)
640
+ return { test: testInfo }; // requested but degraded → disabled, no score
641
+ const best = roundReports.find((r) => r.round === bestRound)?.report;
642
+ if (!best)
643
+ return { test: testInfo };
644
+ const key = Object.keys(best.summary)[0];
645
+ return { test: testInfo, generalizationScore: Number(subsetCompositeScore(best, key, split.testIds).toFixed(4)) };
646
+ };
647
+ const gateInfo = () => ({ enabled: significanceGate, alpha: significanceAlpha, ...(gateUnderpowered ? { underpowered: true } : {}) });
648
+ // Rejected-edit memory (most recent K). Fed back into the next round's improvement
649
+ // prompt so the improver doesn't re-propose changes that already failed to help.
650
+ const rejectedEdits = [];
651
+ const REJECT_MEMORY_K = 3;
652
+ const rememberRejected = (round, summary, reason) => {
653
+ if (!rejectMemory)
654
+ return;
655
+ rejectedEdits.push(`【第 ${round} 轮被拒(${reason})】\n${summary}`);
656
+ if (rejectedEdits.length > REJECT_MEMORY_K)
657
+ rejectedEdits.shift();
658
+ };
503
659
  // Save original as r0
504
660
  let currentBest = readFileSync(absSkillPath, 'utf-8').trim();
505
661
  const r0Path = join(evolveDir, `${skillName}.r0.md`);
@@ -518,10 +674,14 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
518
674
  let stopReason = 'rounds';
519
675
  // 给定一个 report 看任一 variant 的 exec/judge cost 是否未报告
520
676
  const reportHasUnreportedCost = (rep) => Object.values(rep.summary).some((v) => v.execCostReported === false || v.judgeCostReported === false);
521
- // Round 0: baseline evaluation
677
+ // Round 0: baseline evaluation。复用查询键走整树哈,与 eval 报告口径一致:round 0 的 currentBest 即
678
+ // 磁盘上原始 skill,树哈取自磁盘(dir-skill 哈整目录、单文件 .md 哈单文件)。evolve 只改 SKILL.md 正文、
679
+ // 不动 references/ 资产,故磁盘树哈即该 baseline 的权威指纹。
680
+ const baselineIsDirSkill = basename(absSkillPath) === 'SKILL.md';
681
+ const baselineArtifactHash = hashArtifactSource(baselineIsDirSkill ? skillDir : absSkillPath, baselineIsDirSkill);
522
682
  let baselineReport = reuseLatestEval
523
683
  ? await findReusableBaselineReport({
524
- skillContent: currentBest,
684
+ artifactHash: baselineArtifactHash,
525
685
  samplesPath: absSamplesPath,
526
686
  model,
527
687
  executorName,
@@ -567,6 +727,8 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
567
727
  ...(reusedBaselineReportId && { reusedBaselineReportId }),
568
728
  ...(totalCostReported ? {} : { costReported: false }),
569
729
  ...(holdoutInfo ? { holdout: holdoutInfo } : {}),
730
+ ...buildGeneralization(),
731
+ gate: gateInfo(),
570
732
  trajectory,
571
733
  bestSkillPath: allVersions[bestRound],
572
734
  allVersions,
@@ -593,10 +755,10 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
593
755
  totalCostReported = false;
594
756
  }
595
757
  const lastVariantKey = Object.keys(lastReport.summary)[0];
596
- const weakSamples = extractWeakSamples(lastReport, lastVariantKey, 5, holdoutSplit?.trainIds);
758
+ const weakSamples = extractWeakSamples(lastReport, lastVariantKey, 5, split?.trainIds);
597
759
  // Generate improvement
598
760
  const candidatePath = join(evolveDir, `${skillName}.r${round}.md`);
599
- const basePrompt = buildImprovementPrompt(currentBest, bestScore, weakSamples);
761
+ const basePrompt = buildImprovementPrompt(currentBest, bestScore, weakSamples, rejectMemory ? rejectedEdits : undefined);
600
762
  const executor = createExecutor(executorName);
601
763
  let candidateContent;
602
764
  let improveCostUSD;
@@ -650,6 +812,24 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
650
812
  if (!improveCostReported)
651
813
  totalCostReported = false;
652
814
  allVersions.push(candidatePath);
815
+ // Edit budget: reject oversized rewrites BEFORE paying for evaluation. The
816
+ // improvement prompt already asks for ≤ editBudget of lines changed — this
817
+ // enforces it. A small floor still lets tiny skills change a handful of lines.
818
+ const editDelta = computeEditDelta(currentBest, candidateContent);
819
+ if (editBudget > 0 && editDelta.ratio > editBudget && editDelta.changedLines > EDIT_BUDGET_FLOOR_LINES) {
820
+ totalCostUSD += improveCostUSD;
821
+ consecutiveRejects++;
822
+ const reason = `改动过大 ${(editDelta.ratio * 100).toFixed(0)}%(预算 ${(editBudget * 100).toFixed(0)}%),评测前判拒`;
823
+ rememberRejected(round, editDelta.summary, reason);
824
+ trajectory.push({ round, score: bestScore, delta: 0, accepted: false, costUSD: improveCostUSD, editRatio: Number(editDelta.ratio.toFixed(4)), rejectedPreEval: true });
825
+ if (onRoundProgress)
826
+ onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: bestScore, delta: 0, accepted: false, costUSD: improveCostUSD, costReported: improveCostReported });
827
+ if (consecutiveRejects >= 2) {
828
+ stopReason = 'consecutive-rejects';
829
+ break;
830
+ }
831
+ continue;
832
+ }
653
833
  let preEvalSampleFixCost = 0;
654
834
  if (autoFixSamples) {
655
835
  const sampleFix = await autoFixSamplesAfterSkillRound({
@@ -657,7 +837,7 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
657
837
  skillContent: candidateContent,
658
838
  // Under an active holdout, the sample-fixer may only see training-split samples —
659
839
  // never the holdout samples that drive the accept decision (leak guard).
660
- report: holdoutSplit ? restrictReportToSamples(lastReport, holdoutSplit.trainIds) : lastReport,
840
+ report: split ? restrictReportToSamples(lastReport, split.trainIds) : lastReport,
661
841
  treatmentKey: lastVariantKey,
662
842
  executorName,
663
843
  model: improveModel,
@@ -681,7 +861,22 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
681
861
  if (!roundCostReported)
682
862
  totalCostReported = false;
683
863
  totalCostUSD += improveCostUSD + candidateReport.meta.totalCostUSD;
684
- const accepted = candidateScore > bestScore;
864
+ // Significance accept gate: accept only when the candidate is *significantly*
865
+ // above the current best on the decision (val) set, not merely numerically higher
866
+ // — rejecting gains indistinguishable from judge noise. `lastReport` is the current
867
+ // best's fresh eval and `candidateReport` the candidate's, over the same samples;
868
+ // bootstrapDiffCI resamples the two arrays independently (conservative — not a paired
869
+ // bootstrap). Under-powered decision sets degrade to the legacy point-estimate accept
870
+ // (note: that path compares the prior-round best scalar, not this fresh re-eval) and
871
+ // flag `gate.underpowered`.
872
+ const valIds = split ? split.valIds : new Set(allSampleIds);
873
+ const bestScores = perSampleComposite(lastReport, lastVariantKey, valIds);
874
+ const candScores = perSampleComposite(candidateReport, candidateVariantKey, valIds);
875
+ const decision = decideAccept(bestScores, candScores, bestScore, candidateScore, { significanceGate, alpha: significanceAlpha, seed: gateSeed });
876
+ const accepted = decision.accepted;
877
+ const diffCI = decision.diffCI;
878
+ if (decision.underpowered)
879
+ gateUnderpowered = true;
685
880
  if (accepted) {
686
881
  currentBest = candidateContent;
687
882
  bestScore = candidateScore;
@@ -690,13 +885,15 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
690
885
  }
691
886
  else {
692
887
  consecutiveRejects++;
888
+ const reason = diffCI ? (diffCI.estimate > 0 ? '提升不显著' : '方向为负') : '未超过当前最优';
889
+ rememberRejected(round, editDelta.summary, reason);
693
890
  }
694
891
  if (accepted)
695
892
  roundReports.push({ round, accepted, report: candidateReport });
696
893
  const roundDelta = candidateScore - trajectory[trajectory.length - 1].score;
697
- trajectory.push({ round, score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, ...splitScores(candidateReport, candidateVariantKey, candidateScore) });
894
+ trajectory.push({ round, score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, ...splitScores(candidateReport, candidateVariantKey, candidateScore), ...(diffCI ? { diffCI } : {}), editRatio: Number(editDelta.ratio.toFixed(4)) });
698
895
  if (onRoundProgress)
699
- onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, costReported: roundCostReported });
896
+ onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, costReported: roundCostReported, ...(diffCI ? { significant: diffCI.significant } : {}) });
700
897
  // Early stop
701
898
  if (stopOnAssertionsPass && accepted && allNonTripwireAssertionsPass(candidateReport, candidateVariantKey)) {
702
899
  stopReason = 'assertions-pass';
@@ -735,6 +932,8 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
735
932
  ...(reusedBaselineReportId && { reusedBaselineReportId }),
736
933
  ...(totalCostReported ? {} : { costReported: false }),
737
934
  ...(holdoutInfo ? { holdout: holdoutInfo } : {}),
935
+ ...buildGeneralization(),
936
+ gate: gateInfo(),
738
937
  trajectory,
739
938
  bestSkillPath: allVersions[bestRound],
740
939
  allVersions,
@@ -761,4 +960,3 @@ async function evaluate(skillFilePath, { samplesPath, skillDir, model, judgeMode
761
960
  });
762
961
  return report;
763
962
  }
764
- //# sourceMappingURL=evolver.js.map
@@ -63,4 +63,3 @@ export declare function sanitizeGeneratedSamples(samples: Sample[], opts?: {
63
63
  stripped: string[];
64
64
  };
65
65
  export {};
66
- //# sourceMappingURL=generator.d.ts.map
@@ -830,4 +830,3 @@ export function sanitizeGeneratedSamples(samples, opts = {}) {
830
830
  }
831
831
  return { stripped };
832
832
  }
833
- //# sourceMappingURL=generator.js.map
@@ -48,4 +48,3 @@ export interface FixSamplesResult {
48
48
  }>;
49
49
  }
50
50
  export declare function fixSamples(options: FixSamplesOptions): Promise<FixSamplesResult>;
51
- //# sourceMappingURL=sample-fixer.d.ts.map
@@ -210,4 +210,3 @@ ${sampleSections}
210
210
  };
211
211
  }
212
212
  }
213
- //# sourceMappingURL=sample-fixer.js.map
@@ -25,4 +25,3 @@ export default class Doctor extends BaseCommand {
25
25
  run(): Promise<void>;
26
26
  }
27
27
  export declare function pruneDoctorHistory(dir: string, skillName: string, maxKeep: number): void;
28
- //# sourceMappingURL=doctor.d.ts.map