oh-my-knowledge 0.32.0 → 0.34.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (729) hide show
  1. package/README.md +29 -14
  2. package/README.zh.md +32 -17
  3. package/dist/analysis/coverage-analyzer.d.ts +0 -1
  4. package/dist/analysis/coverage-analyzer.js +0 -1
  5. package/dist/analysis/failure-clusterer.d.ts +0 -1
  6. package/dist/analysis/failure-clusterer.js +0 -1
  7. package/dist/analysis/gap-analyzer.d.ts +0 -1
  8. package/dist/analysis/gap-analyzer.js +0 -1
  9. package/dist/analysis/hedging-classifier.d.ts +0 -1
  10. package/dist/analysis/hedging-classifier.js +1 -2
  11. package/dist/analysis/report-diagnostics.d.ts +0 -1
  12. package/dist/analysis/report-diagnostics.js +0 -1
  13. package/dist/analysis/sample-diagnostics.d.ts +1 -2
  14. package/dist/analysis/sample-diagnostics.js +18 -19
  15. package/dist/analysis/saturation.d.ts +8 -1
  16. package/dist/analysis/saturation.js +12 -5
  17. package/dist/assets/agent-skills/omk/SKILL.md +197 -0
  18. package/dist/assets/agent-skills/omk/references/commands.md +547 -0
  19. package/dist/authoring/evolver.d.ts +169 -4
  20. package/dist/authoring/evolver.js +287 -15
  21. package/dist/authoring/generator.d.ts +29 -2
  22. package/dist/authoring/generator.js +113 -1
  23. package/dist/authoring/sample-fixer.d.ts +0 -1
  24. package/dist/authoring/sample-fixer.js +0 -1
  25. package/dist/cli/commands/doctor.d.ts +2 -1
  26. package/dist/cli/commands/doctor.js +24 -6
  27. package/dist/cli/commands/eval/gold/compare.d.ts +0 -1
  28. package/dist/cli/commands/eval/gold/compare.js +0 -1
  29. package/dist/cli/commands/eval/gold/index.d.ts +0 -1
  30. package/dist/cli/commands/eval/gold/index.js +0 -1
  31. package/dist/cli/commands/eval/gold/init.d.ts +0 -1
  32. package/dist/cli/commands/eval/gold/init.js +0 -1
  33. package/dist/cli/commands/eval/gold/validate.d.ts +0 -1
  34. package/dist/cli/commands/eval/gold/validate.js +0 -1
  35. package/dist/cli/commands/eval/index.d.ts +2 -1
  36. package/dist/cli/commands/eval/index.js +23 -11
  37. package/dist/cli/commands/evolve.d.ts +7 -1
  38. package/dist/cli/commands/evolve.js +93 -6
  39. package/dist/cli/commands/init.d.ts +0 -1
  40. package/dist/cli/commands/init.js +0 -1
  41. package/dist/cli/commands/install.d.ts +22 -0
  42. package/dist/cli/commands/install.js +411 -0
  43. package/dist/cli/commands/observe/inbox.d.ts +0 -1
  44. package/dist/cli/commands/observe/inbox.js +0 -1
  45. package/dist/cli/commands/observe/index.d.ts +0 -1
  46. package/dist/cli/commands/observe/index.js +6 -3
  47. package/dist/cli/commands/observe/ingest.d.ts +0 -1
  48. package/dist/cli/commands/observe/ingest.js +0 -1
  49. package/dist/cli/commands/observe/show.d.ts +0 -1
  50. package/dist/cli/commands/observe/show.js +0 -1
  51. package/dist/cli/commands/sample.d.ts +2 -1
  52. package/dist/cli/commands/sample.js +100 -9
  53. package/dist/cli/commands/studio.d.ts +0 -1
  54. package/dist/cli/commands/studio.js +0 -1
  55. package/dist/cli/index.d.ts +0 -1
  56. package/dist/cli/index.js +1 -2
  57. package/dist/cli/lib/cli-exit.d.ts +0 -1
  58. package/dist/cli/lib/cli-exit.js +0 -1
  59. package/dist/cli/lib/cmd-flags.d.ts +11 -1
  60. package/dist/cli/lib/cmd-flags.js +0 -1
  61. package/dist/cli/lib/i18n-dict/common.d.ts +1 -2
  62. package/dist/cli/lib/i18n-dict/common.js +18 -3
  63. package/dist/cli/lib/i18n-dict/evolve.d.ts +1 -2
  64. package/dist/cli/lib/i18n-dict/evolve.js +24 -1
  65. package/dist/cli/lib/i18n-dict/gen.d.ts +0 -1
  66. package/dist/cli/lib/i18n-dict/gen.js +0 -1
  67. package/dist/cli/lib/i18n-dict/help.d.ts +0 -1
  68. package/dist/cli/lib/i18n-dict/help.js +0 -1
  69. package/dist/cli/lib/i18n-dict/init.d.ts +0 -1
  70. package/dist/cli/lib/i18n-dict/init.js +0 -1
  71. package/dist/cli/lib/i18n-dict/install.d.ts +3 -0
  72. package/dist/cli/lib/i18n-dict/install.js +90 -0
  73. package/dist/cli/lib/i18n-dict/run.d.ts +0 -1
  74. package/dist/cli/lib/i18n-dict/run.js +0 -1
  75. package/dist/cli/lib/i18n-dict/types.d.ts +0 -1
  76. package/dist/cli/lib/i18n-dict/types.js +0 -1
  77. package/dist/cli/lib/i18n-dict.d.ts +2 -2
  78. package/dist/cli/lib/i18n-dict.js +2 -1
  79. package/dist/cli/lib/i18n.d.ts +0 -1
  80. package/dist/cli/lib/i18n.js +0 -1
  81. package/dist/cli/lib/parse-run-config/judge-models.d.ts +0 -1
  82. package/dist/cli/lib/parse-run-config/judge-models.js +0 -1
  83. package/dist/cli/lib/parse-run-config/samples-discovery.d.ts +0 -1
  84. package/dist/cli/lib/parse-run-config/samples-discovery.js +1 -4
  85. package/dist/cli/lib/parse-run-config/variant-resolution.d.ts +8 -4
  86. package/dist/cli/lib/parse-run-config/variant-resolution.js +46 -14
  87. package/dist/cli/lib/parse-run-config.d.ts +0 -1
  88. package/dist/cli/lib/parse-run-config.js +1 -2
  89. package/dist/cli/lib/progress.d.ts +0 -1
  90. package/dist/cli/lib/progress.js +0 -1
  91. package/dist/cli/lib/resolve-skill-input.d.ts +0 -1
  92. package/dist/cli/lib/resolve-skill-input.js +7 -6
  93. package/dist/cli/lib/run-tally.d.ts +0 -1
  94. package/dist/cli/lib/run-tally.js +1 -2
  95. package/dist/cli/lib/shared.d.ts +0 -1
  96. package/dist/cli/lib/shared.js +1 -2
  97. package/dist/cli/lib/update-check.d.ts +65 -1
  98. package/dist/cli/lib/update-check.js +220 -32
  99. package/dist/cli/lib/update-fetch-worker.d.ts +1 -0
  100. package/dist/cli/lib/update-fetch-worker.js +34 -0
  101. package/dist/cli/oclif/base-command.d.ts +0 -1
  102. package/dist/cli/oclif/base-command.js +0 -1
  103. package/dist/cli/oclif/help.d.ts +0 -1
  104. package/dist/cli/oclif/help.js +0 -1
  105. package/dist/cli/oclif/i18n.d.ts +0 -1
  106. package/dist/cli/oclif/i18n.js +0 -1
  107. package/dist/cli/oclif/parsers.d.ts +0 -1
  108. package/dist/cli/oclif/parsers.js +0 -1
  109. package/dist/cli/oclif/projection.d.ts +0 -1
  110. package/dist/cli/oclif/projection.js +0 -1
  111. package/dist/cli/oclif/run.d.ts +0 -1
  112. package/dist/cli/oclif/run.js +0 -1
  113. package/dist/diagnosis/observe-mapper.d.ts +2 -3
  114. package/dist/diagnosis/observe-mapper.js +6 -7
  115. package/dist/diagnosis/observe-producer.d.ts +0 -1
  116. package/dist/diagnosis/observe-producer.js +3 -4
  117. package/dist/diagnosis/studio-projection.d.ts +0 -1
  118. package/dist/diagnosis/studio-projection.js +0 -1
  119. package/dist/diagnosis/types.d.ts +0 -1
  120. package/dist/diagnosis/types.js +0 -1
  121. package/dist/doctor/fixer.d.ts +0 -1
  122. package/dist/doctor/fixer.js +0 -1
  123. package/dist/doctor/health/builtin-dimensions.d.ts +0 -1
  124. package/dist/doctor/health/builtin-dimensions.js +0 -1
  125. package/dist/doctor/health/composer.d.ts +0 -1
  126. package/dist/doctor/health/composer.js +1 -2
  127. package/dist/doctor/health/dimension-registry.d.ts +0 -1
  128. package/dist/doctor/health/dimension-registry.js +0 -1
  129. package/dist/doctor/health/dimension-spec.d.ts +0 -1
  130. package/dist/doctor/health/dimension-spec.js +0 -1
  131. package/dist/doctor/health/load-custom-dimensions.d.ts +1 -0
  132. package/dist/doctor/health/load-custom-dimensions.js +30 -0
  133. package/dist/doctor/health/parser.d.ts +0 -1
  134. package/dist/doctor/health/parser.js +0 -1
  135. package/dist/doctor/health/prompt-builder.d.ts +0 -1
  136. package/dist/doctor/health/prompt-builder.js +0 -1
  137. package/dist/doctor/health/register.d.ts +0 -1
  138. package/dist/doctor/health/register.js +0 -1
  139. package/dist/doctor/index.d.ts +0 -1
  140. package/dist/doctor/index.js +1 -2
  141. package/dist/doctor/messages.d.ts +0 -1
  142. package/dist/doctor/messages.js +1 -2
  143. package/dist/doctor/preflight.d.ts +0 -1
  144. package/dist/doctor/preflight.js +0 -1
  145. package/dist/doctor/renderer.d.ts +0 -1
  146. package/dist/doctor/renderer.js +0 -1
  147. package/dist/doctor/rules.d.ts +0 -1
  148. package/dist/doctor/rules.js +0 -1
  149. package/dist/eval-core/bootstrap.d.ts +8 -1
  150. package/dist/eval-core/bootstrap.js +11 -4
  151. package/dist/eval-core/cache.d.ts +0 -1
  152. package/dist/eval-core/cache.js +0 -1
  153. package/dist/eval-core/comparability.d.ts +0 -1
  154. package/dist/eval-core/comparability.js +3 -4
  155. package/dist/eval-core/dependency-checker.d.ts +0 -1
  156. package/dist/eval-core/dependency-checker.js +2 -2
  157. package/dist/eval-core/evaluation-execution.d.ts +0 -1
  158. package/dist/eval-core/evaluation-execution.js +0 -1
  159. package/dist/eval-core/evaluation-job.d.ts +0 -1
  160. package/dist/eval-core/evaluation-job.js +0 -1
  161. package/dist/eval-core/evaluation-reporting.d.ts +0 -1
  162. package/dist/eval-core/evaluation-reporting.js +9 -8
  163. package/dist/eval-core/execution-strategy.d.ts +0 -1
  164. package/dist/eval-core/execution-strategy.js +6 -3
  165. package/dist/eval-core/fact-checker.d.ts +0 -1
  166. package/dist/eval-core/fact-checker.js +0 -1
  167. package/dist/eval-core/layer-gates.d.ts +0 -1
  168. package/dist/eval-core/layer-gates.js +0 -1
  169. package/dist/eval-core/mocks-runtime.d.ts +0 -1
  170. package/dist/eval-core/mocks-runtime.js +0 -1
  171. package/dist/eval-core/schema.d.ts +0 -1
  172. package/dist/eval-core/schema.js +0 -1
  173. package/dist/eval-core/statistics.d.ts +0 -1
  174. package/dist/eval-core/statistics.js +0 -1
  175. package/dist/eval-core/task-planner.d.ts +0 -1
  176. package/dist/eval-core/task-planner.js +0 -1
  177. package/dist/eval-core/verdict.d.ts +21 -4
  178. package/dist/eval-core/verdict.js +54 -5
  179. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +18 -3
  180. package/dist/eval-workflows/batch-evaluation-workflow.js +31 -15
  181. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +0 -1
  182. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +0 -1
  183. package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +0 -1
  184. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +0 -1
  185. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +0 -1
  186. package/dist/eval-workflows/evaluation-pipeline/run-state.js +0 -1
  187. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +0 -1
  188. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +0 -1
  189. package/dist/eval-workflows/evaluation-pipeline.d.ts +0 -1
  190. package/dist/eval-workflows/evaluation-pipeline.js +0 -1
  191. package/dist/eval-workflows/evaluation-preparation.d.ts +1 -5
  192. package/dist/eval-workflows/evaluation-preparation.js +20 -16
  193. package/dist/eval-workflows/messages.d.ts +0 -1
  194. package/dist/eval-workflows/messages.js +0 -1
  195. package/dist/eval-workflows/run-evaluation.d.ts +4 -9
  196. package/dist/eval-workflows/run-evaluation.js +16 -26
  197. package/dist/executors/anthropic-api.d.ts +0 -1
  198. package/dist/executors/anthropic-api.js +1 -2
  199. package/dist/executors/claude-cli.d.ts +0 -1
  200. package/dist/executors/claude-cli.js +0 -1
  201. package/dist/executors/claude-sdk-trace.d.ts +0 -1
  202. package/dist/executors/claude-sdk-trace.js +0 -1
  203. package/dist/executors/claude-sdk.d.ts +0 -1
  204. package/dist/executors/claude-sdk.js +0 -1
  205. package/dist/executors/codex-cli-trace.d.ts +0 -1
  206. package/dist/executors/codex-cli-trace.js +0 -1
  207. package/dist/executors/codex-cli.d.ts +0 -1
  208. package/dist/executors/codex-cli.js +0 -1
  209. package/dist/executors/codex-sdk.d.ts +0 -1
  210. package/dist/executors/codex-sdk.js +0 -1
  211. package/dist/executors/gemini.d.ts +0 -1
  212. package/dist/executors/gemini.js +0 -1
  213. package/dist/executors/index.d.ts +0 -1
  214. package/dist/executors/index.js +0 -1
  215. package/dist/executors/openai-api.d.ts +0 -1
  216. package/dist/executors/openai-api.js +0 -1
  217. package/dist/executors/runtime-fingerprint.d.ts +0 -1
  218. package/dist/executors/runtime-fingerprint.js +2 -3
  219. package/dist/executors/script.d.ts +0 -1
  220. package/dist/executors/script.js +0 -1
  221. package/dist/executors/shared.d.ts +0 -2
  222. package/dist/executors/shared.js +0 -1
  223. package/dist/grading/assertions.d.ts +0 -1
  224. package/dist/grading/assertions.js +0 -1
  225. package/dist/grading/debias-validate.d.ts +0 -1
  226. package/dist/grading/debias-validate.js +0 -1
  227. package/dist/grading/diagnostic.d.ts +0 -1
  228. package/dist/grading/diagnostic.js +0 -1
  229. package/dist/grading/gold-cli.d.ts +0 -1
  230. package/dist/grading/gold-cli.js +0 -1
  231. package/dist/grading/gold-dataset.d.ts +0 -1
  232. package/dist/grading/gold-dataset.js +0 -1
  233. package/dist/grading/human-gold.d.ts +0 -1
  234. package/dist/grading/human-gold.js +0 -1
  235. package/dist/grading/index.d.ts +0 -1
  236. package/dist/grading/index.js +0 -1
  237. package/dist/grading/judge.d.ts +0 -1
  238. package/dist/grading/judge.js +0 -1
  239. package/dist/grading/layered-scores.d.ts +0 -1
  240. package/dist/grading/layered-scores.js +0 -1
  241. package/dist/inputs/eval-config.d.ts +0 -1
  242. package/dist/inputs/eval-config.js +3 -2
  243. package/dist/inputs/load-samples.d.ts +0 -1
  244. package/dist/inputs/load-samples.js +0 -1
  245. package/dist/inputs/mcp-resolver.d.ts +0 -1
  246. package/dist/inputs/mcp-resolver.js +0 -1
  247. package/dist/inputs/skill-loader.d.ts +99 -7
  248. package/dist/inputs/skill-loader.js +325 -45
  249. package/dist/inputs/source-resolver.d.ts +28 -0
  250. package/dist/inputs/source-resolver.js +125 -0
  251. package/dist/inputs/url-fetcher.d.ts +0 -1
  252. package/dist/inputs/url-fetcher.js +0 -1
  253. package/dist/managed/index.d.ts +5 -0
  254. package/dist/managed/index.js +5 -0
  255. package/dist/managed/store.d.ts +76 -0
  256. package/dist/managed/store.js +260 -0
  257. package/dist/observability/experience-frontmatter.d.ts +0 -1
  258. package/dist/observability/experience-frontmatter.js +0 -1
  259. package/dist/observability/experience.d.ts +1 -2
  260. package/dist/observability/experience.js +1 -2
  261. package/dist/observability/feedback-matchers.d.ts +0 -1
  262. package/dist/observability/feedback-matchers.js +0 -1
  263. package/dist/observability/feedback-projection.d.ts +0 -1
  264. package/dist/observability/feedback-projection.js +0 -1
  265. package/dist/observability/inbox-view-model.d.ts +0 -1
  266. package/dist/observability/inbox-view-model.js +0 -1
  267. package/dist/observability/inbox.d.ts +0 -1
  268. package/dist/observability/inbox.js +2 -3
  269. package/dist/observability/problem-patterns.d.ts +0 -1
  270. package/dist/observability/problem-patterns.js +0 -1
  271. package/dist/observability/resolved-review.d.ts +0 -1
  272. package/dist/observability/resolved-review.js +4 -5
  273. package/dist/observability/review-state.d.ts +0 -1
  274. package/dist/observability/review-state.js +3 -4
  275. package/dist/observability/skill-chain-advisories.d.ts +0 -1
  276. package/dist/observability/skill-chain-advisories.js +0 -1
  277. package/dist/observability/skill-chain.d.ts +0 -1
  278. package/dist/observability/skill-chain.js +5 -6
  279. package/dist/observability/skill-health-analyzer.d.ts +13 -1
  280. package/dist/observability/skill-health-analyzer.js +17 -3
  281. package/dist/observability/soft-standards/constants.d.ts +0 -1
  282. package/dist/observability/soft-standards/constants.js +0 -1
  283. package/dist/observability/soft-standards/index.d.ts +0 -1
  284. package/dist/observability/soft-standards/index.js +0 -1
  285. package/dist/observability/soft-standards/llm-extractor.d.ts +0 -1
  286. package/dist/observability/soft-standards/llm-extractor.js +7 -8
  287. package/dist/observability/soft-standards/runtime-evaluator.d.ts +0 -1
  288. package/dist/observability/soft-standards/runtime-evaluator.js +2 -3
  289. package/dist/observability/soft-standards/skill-standards-store.d.ts +0 -1
  290. package/dist/observability/soft-standards/skill-standards-store.js +5 -6
  291. package/dist/observability/soft-standards/types.d.ts +10 -11
  292. package/dist/observability/soft-standards/types.js +0 -1
  293. package/dist/observability/text-signals.d.ts +0 -1
  294. package/dist/observability/text-signals.js +0 -1
  295. package/dist/observability/trace-adapter.d.ts +0 -1
  296. package/dist/observability/trace-adapter.js +0 -1
  297. package/dist/observability/trace-attribution.d.ts +0 -1
  298. package/dist/observability/trace-attribution.js +0 -1
  299. package/dist/observability/trace-segmenter.d.ts +0 -1
  300. package/dist/observability/trace-segmenter.js +0 -1
  301. package/dist/observability/trace-source.d.ts +0 -1
  302. package/dist/observability/trace-source.js +0 -1
  303. package/dist/renderer/html-renderer.d.ts +0 -1
  304. package/dist/renderer/html-renderer.js +4 -5
  305. package/dist/renderer/layout.d.ts +0 -1
  306. package/dist/renderer/layout.js +2 -1
  307. package/dist/renderer/observation-inbox/helpers.d.ts +0 -1
  308. package/dist/renderer/observation-inbox/helpers.js +0 -1
  309. package/dist/renderer/observation-inbox/styles.d.ts +0 -1
  310. package/dist/renderer/observation-inbox/styles.js +0 -1
  311. package/dist/renderer/observation-inbox-renderer.d.ts +0 -1
  312. package/dist/renderer/observation-inbox-renderer.js +13 -14
  313. package/dist/renderer/skill-detail-renderer.d.ts +0 -1
  314. package/dist/renderer/skill-detail-renderer.js +101 -23
  315. package/dist/renderer/skill-health-renderer.d.ts +0 -1
  316. package/dist/renderer/skill-health-renderer.js +33 -5
  317. package/dist/renderer/skill-list-renderer.d.ts +0 -1
  318. package/dist/renderer/skill-list-renderer.js +18 -7
  319. package/dist/renderer/summary.d.ts +0 -1
  320. package/dist/renderer/summary.js +21 -10
  321. package/dist/renderer/table.d.ts +0 -1
  322. package/dist/renderer/table.js +0 -1
  323. package/dist/renderer/test-view.d.ts +0 -1
  324. package/dist/renderer/test-view.js +0 -1
  325. package/dist/renderer/trends.d.ts +0 -1
  326. package/dist/renderer/trends.js +0 -1
  327. package/dist/server/job-store.d.ts +0 -1
  328. package/dist/server/job-store.js +0 -1
  329. package/dist/server/report-server.d.ts +0 -1
  330. package/dist/server/report-server.js +10 -4
  331. package/dist/server/report-store.d.ts +1 -2
  332. package/dist/server/report-store.js +6 -7
  333. package/dist/server/skill-index.d.ts +0 -1
  334. package/dist/server/skill-index.js +7 -5
  335. package/dist/server/skill-insights.d.ts +0 -1
  336. package/dist/server/skill-insights.js +41 -14
  337. package/dist/shared/hard-rules.d.ts +0 -1
  338. package/dist/shared/hard-rules.js +0 -1
  339. package/dist/shared/llm-prompts/index.d.ts +0 -1
  340. package/dist/shared/llm-prompts/index.js +0 -1
  341. package/dist/shared/llm-prompts/skill-health.d.ts +0 -1
  342. package/dist/shared/llm-prompts/skill-health.js +0 -1
  343. package/dist/shared/time.d.ts +0 -1
  344. package/dist/shared/time.js +0 -1
  345. package/dist/shared/tool-search.d.ts +0 -1
  346. package/dist/shared/tool-search.js +0 -1
  347. package/dist/types/dependencies.d.ts +0 -1
  348. package/dist/types/dependencies.js +0 -1
  349. package/dist/types/diagnosis.d.ts +0 -1
  350. package/dist/types/diagnosis.js +0 -1
  351. package/dist/types/doctor.d.ts +4 -5
  352. package/dist/types/doctor.js +2 -3
  353. package/dist/types/eval.d.ts +2 -1
  354. package/dist/types/eval.js +0 -1
  355. package/dist/types/executor.d.ts +1 -2
  356. package/dist/types/executor.js +0 -1
  357. package/dist/types/index.d.ts +1 -1
  358. package/dist/types/index.js +1 -1
  359. package/dist/types/judge.d.ts +0 -1
  360. package/dist/types/judge.js +0 -1
  361. package/dist/types/managed.d.ts +85 -0
  362. package/dist/types/managed.js +1 -0
  363. package/dist/types/observability.d.ts +5 -6
  364. package/dist/types/observability.js +0 -1
  365. package/dist/types/report.d.ts +2 -3
  366. package/dist/types/report.js +0 -1
  367. package/dist/types/shared.d.ts +0 -1
  368. package/dist/types/shared.js +0 -1
  369. package/dist/types/skill-index.d.ts +3 -1
  370. package/dist/types/skill-index.js +0 -1
  371. package/dist/types/storage.d.ts +0 -1
  372. package/dist/types/storage.js +0 -1
  373. package/dist/util/safe-slice.d.ts +0 -1
  374. package/dist/util/safe-slice.js +0 -1
  375. package/package.json +10 -5
  376. package/dist/analysis/coverage-analyzer.d.ts.map +0 -1
  377. package/dist/analysis/coverage-analyzer.js.map +0 -1
  378. package/dist/analysis/failure-clusterer.d.ts.map +0 -1
  379. package/dist/analysis/failure-clusterer.js.map +0 -1
  380. package/dist/analysis/gap-analyzer.d.ts.map +0 -1
  381. package/dist/analysis/gap-analyzer.js.map +0 -1
  382. package/dist/analysis/hedging-classifier.d.ts.map +0 -1
  383. package/dist/analysis/hedging-classifier.js.map +0 -1
  384. package/dist/analysis/report-diagnostics.d.ts.map +0 -1
  385. package/dist/analysis/report-diagnostics.js.map +0 -1
  386. package/dist/analysis/sample-diagnostics.d.ts.map +0 -1
  387. package/dist/analysis/sample-diagnostics.js.map +0 -1
  388. package/dist/analysis/saturation.d.ts.map +0 -1
  389. package/dist/analysis/saturation.js.map +0 -1
  390. package/dist/authoring/evolver.d.ts.map +0 -1
  391. package/dist/authoring/evolver.js.map +0 -1
  392. package/dist/authoring/generator.d.ts.map +0 -1
  393. package/dist/authoring/generator.js.map +0 -1
  394. package/dist/authoring/sample-fixer.d.ts.map +0 -1
  395. package/dist/authoring/sample-fixer.js.map +0 -1
  396. package/dist/cli/commands/doctor.d.ts.map +0 -1
  397. package/dist/cli/commands/doctor.js.map +0 -1
  398. package/dist/cli/commands/eval/gold/compare.d.ts.map +0 -1
  399. package/dist/cli/commands/eval/gold/compare.js.map +0 -1
  400. package/dist/cli/commands/eval/gold/index.d.ts.map +0 -1
  401. package/dist/cli/commands/eval/gold/index.js.map +0 -1
  402. package/dist/cli/commands/eval/gold/init.d.ts.map +0 -1
  403. package/dist/cli/commands/eval/gold/init.js.map +0 -1
  404. package/dist/cli/commands/eval/gold/validate.d.ts.map +0 -1
  405. package/dist/cli/commands/eval/gold/validate.js.map +0 -1
  406. package/dist/cli/commands/eval/index.d.ts.map +0 -1
  407. package/dist/cli/commands/eval/index.js.map +0 -1
  408. package/dist/cli/commands/evolve.d.ts.map +0 -1
  409. package/dist/cli/commands/evolve.js.map +0 -1
  410. package/dist/cli/commands/init.d.ts.map +0 -1
  411. package/dist/cli/commands/init.js.map +0 -1
  412. package/dist/cli/commands/observe/inbox.d.ts.map +0 -1
  413. package/dist/cli/commands/observe/inbox.js.map +0 -1
  414. package/dist/cli/commands/observe/index.d.ts.map +0 -1
  415. package/dist/cli/commands/observe/index.js.map +0 -1
  416. package/dist/cli/commands/observe/ingest.d.ts.map +0 -1
  417. package/dist/cli/commands/observe/ingest.js.map +0 -1
  418. package/dist/cli/commands/observe/show.d.ts.map +0 -1
  419. package/dist/cli/commands/observe/show.js.map +0 -1
  420. package/dist/cli/commands/sample.d.ts.map +0 -1
  421. package/dist/cli/commands/sample.js.map +0 -1
  422. package/dist/cli/commands/studio.d.ts.map +0 -1
  423. package/dist/cli/commands/studio.js.map +0 -1
  424. package/dist/cli/index.d.ts.map +0 -1
  425. package/dist/cli/index.js.map +0 -1
  426. package/dist/cli/lib/cli-exit.d.ts.map +0 -1
  427. package/dist/cli/lib/cli-exit.js.map +0 -1
  428. package/dist/cli/lib/cmd-flags.d.ts.map +0 -1
  429. package/dist/cli/lib/cmd-flags.js.map +0 -1
  430. package/dist/cli/lib/i18n-dict/common.d.ts.map +0 -1
  431. package/dist/cli/lib/i18n-dict/common.js.map +0 -1
  432. package/dist/cli/lib/i18n-dict/evolve.d.ts.map +0 -1
  433. package/dist/cli/lib/i18n-dict/evolve.js.map +0 -1
  434. package/dist/cli/lib/i18n-dict/gen.d.ts.map +0 -1
  435. package/dist/cli/lib/i18n-dict/gen.js.map +0 -1
  436. package/dist/cli/lib/i18n-dict/help.d.ts.map +0 -1
  437. package/dist/cli/lib/i18n-dict/help.js.map +0 -1
  438. package/dist/cli/lib/i18n-dict/init.d.ts.map +0 -1
  439. package/dist/cli/lib/i18n-dict/init.js.map +0 -1
  440. package/dist/cli/lib/i18n-dict/run.d.ts.map +0 -1
  441. package/dist/cli/lib/i18n-dict/run.js.map +0 -1
  442. package/dist/cli/lib/i18n-dict/types.d.ts.map +0 -1
  443. package/dist/cli/lib/i18n-dict/types.js.map +0 -1
  444. package/dist/cli/lib/i18n-dict.d.ts.map +0 -1
  445. package/dist/cli/lib/i18n-dict.js.map +0 -1
  446. package/dist/cli/lib/i18n.d.ts.map +0 -1
  447. package/dist/cli/lib/i18n.js.map +0 -1
  448. package/dist/cli/lib/parse-run-config/judge-models.d.ts.map +0 -1
  449. package/dist/cli/lib/parse-run-config/judge-models.js.map +0 -1
  450. package/dist/cli/lib/parse-run-config/samples-discovery.d.ts.map +0 -1
  451. package/dist/cli/lib/parse-run-config/samples-discovery.js.map +0 -1
  452. package/dist/cli/lib/parse-run-config/variant-resolution.d.ts.map +0 -1
  453. package/dist/cli/lib/parse-run-config/variant-resolution.js.map +0 -1
  454. package/dist/cli/lib/parse-run-config.d.ts.map +0 -1
  455. package/dist/cli/lib/parse-run-config.js.map +0 -1
  456. package/dist/cli/lib/progress.d.ts.map +0 -1
  457. package/dist/cli/lib/progress.js.map +0 -1
  458. package/dist/cli/lib/resolve-skill-input.d.ts.map +0 -1
  459. package/dist/cli/lib/resolve-skill-input.js.map +0 -1
  460. package/dist/cli/lib/run-tally.d.ts.map +0 -1
  461. package/dist/cli/lib/run-tally.js.map +0 -1
  462. package/dist/cli/lib/shared.d.ts.map +0 -1
  463. package/dist/cli/lib/shared.js.map +0 -1
  464. package/dist/cli/lib/update-check.d.ts.map +0 -1
  465. package/dist/cli/lib/update-check.js.map +0 -1
  466. package/dist/cli/oclif/base-command.d.ts.map +0 -1
  467. package/dist/cli/oclif/base-command.js.map +0 -1
  468. package/dist/cli/oclif/help.d.ts.map +0 -1
  469. package/dist/cli/oclif/help.js.map +0 -1
  470. package/dist/cli/oclif/i18n.d.ts.map +0 -1
  471. package/dist/cli/oclif/i18n.js.map +0 -1
  472. package/dist/cli/oclif/parsers.d.ts.map +0 -1
  473. package/dist/cli/oclif/parsers.js.map +0 -1
  474. package/dist/cli/oclif/projection.d.ts.map +0 -1
  475. package/dist/cli/oclif/projection.js.map +0 -1
  476. package/dist/cli/oclif/run.d.ts.map +0 -1
  477. package/dist/cli/oclif/run.js.map +0 -1
  478. package/dist/diagnosis/observe-mapper.d.ts.map +0 -1
  479. package/dist/diagnosis/observe-mapper.js.map +0 -1
  480. package/dist/diagnosis/observe-producer.d.ts.map +0 -1
  481. package/dist/diagnosis/observe-producer.js.map +0 -1
  482. package/dist/diagnosis/studio-projection.d.ts.map +0 -1
  483. package/dist/diagnosis/studio-projection.js.map +0 -1
  484. package/dist/diagnosis/types.d.ts.map +0 -1
  485. package/dist/diagnosis/types.js.map +0 -1
  486. package/dist/doctor/fixer.d.ts.map +0 -1
  487. package/dist/doctor/fixer.js.map +0 -1
  488. package/dist/doctor/health/builtin-dimensions.d.ts.map +0 -1
  489. package/dist/doctor/health/builtin-dimensions.js.map +0 -1
  490. package/dist/doctor/health/composer.d.ts.map +0 -1
  491. package/dist/doctor/health/composer.js.map +0 -1
  492. package/dist/doctor/health/dimension-registry.d.ts.map +0 -1
  493. package/dist/doctor/health/dimension-registry.js.map +0 -1
  494. package/dist/doctor/health/dimension-spec.d.ts.map +0 -1
  495. package/dist/doctor/health/dimension-spec.js.map +0 -1
  496. package/dist/doctor/health/parser.d.ts.map +0 -1
  497. package/dist/doctor/health/parser.js.map +0 -1
  498. package/dist/doctor/health/prompt-builder.d.ts.map +0 -1
  499. package/dist/doctor/health/prompt-builder.js.map +0 -1
  500. package/dist/doctor/health/register.d.ts.map +0 -1
  501. package/dist/doctor/health/register.js.map +0 -1
  502. package/dist/doctor/index.d.ts.map +0 -1
  503. package/dist/doctor/index.js.map +0 -1
  504. package/dist/doctor/messages.d.ts.map +0 -1
  505. package/dist/doctor/messages.js.map +0 -1
  506. package/dist/doctor/preflight.d.ts.map +0 -1
  507. package/dist/doctor/preflight.js.map +0 -1
  508. package/dist/doctor/renderer.d.ts.map +0 -1
  509. package/dist/doctor/renderer.js.map +0 -1
  510. package/dist/doctor/rules.d.ts.map +0 -1
  511. package/dist/doctor/rules.js.map +0 -1
  512. package/dist/eval-core/bootstrap.d.ts.map +0 -1
  513. package/dist/eval-core/bootstrap.js.map +0 -1
  514. package/dist/eval-core/cache.d.ts.map +0 -1
  515. package/dist/eval-core/cache.js.map +0 -1
  516. package/dist/eval-core/comparability.d.ts.map +0 -1
  517. package/dist/eval-core/comparability.js.map +0 -1
  518. package/dist/eval-core/dependency-checker.d.ts.map +0 -1
  519. package/dist/eval-core/dependency-checker.js.map +0 -1
  520. package/dist/eval-core/evaluation-execution.d.ts.map +0 -1
  521. package/dist/eval-core/evaluation-execution.js.map +0 -1
  522. package/dist/eval-core/evaluation-job.d.ts.map +0 -1
  523. package/dist/eval-core/evaluation-job.js.map +0 -1
  524. package/dist/eval-core/evaluation-reporting.d.ts.map +0 -1
  525. package/dist/eval-core/evaluation-reporting.js.map +0 -1
  526. package/dist/eval-core/execution-strategy.d.ts.map +0 -1
  527. package/dist/eval-core/execution-strategy.js.map +0 -1
  528. package/dist/eval-core/fact-checker.d.ts.map +0 -1
  529. package/dist/eval-core/fact-checker.js.map +0 -1
  530. package/dist/eval-core/layer-gates.d.ts.map +0 -1
  531. package/dist/eval-core/layer-gates.js.map +0 -1
  532. package/dist/eval-core/mocks-runtime.d.ts.map +0 -1
  533. package/dist/eval-core/mocks-runtime.js.map +0 -1
  534. package/dist/eval-core/schema.d.ts.map +0 -1
  535. package/dist/eval-core/schema.js.map +0 -1
  536. package/dist/eval-core/statistics.d.ts.map +0 -1
  537. package/dist/eval-core/statistics.js.map +0 -1
  538. package/dist/eval-core/task-planner.d.ts.map +0 -1
  539. package/dist/eval-core/task-planner.js.map +0 -1
  540. package/dist/eval-core/verdict.d.ts.map +0 -1
  541. package/dist/eval-core/verdict.js.map +0 -1
  542. package/dist/eval-workflows/batch-evaluation-workflow.d.ts.map +0 -1
  543. package/dist/eval-workflows/batch-evaluation-workflow.js.map +0 -1
  544. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts.map +0 -1
  545. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js.map +0 -1
  546. package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts.map +0 -1
  547. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js.map +0 -1
  548. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts.map +0 -1
  549. package/dist/eval-workflows/evaluation-pipeline/run-state.js.map +0 -1
  550. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts.map +0 -1
  551. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js.map +0 -1
  552. package/dist/eval-workflows/evaluation-pipeline.d.ts.map +0 -1
  553. package/dist/eval-workflows/evaluation-pipeline.js.map +0 -1
  554. package/dist/eval-workflows/evaluation-preparation.d.ts.map +0 -1
  555. package/dist/eval-workflows/evaluation-preparation.js.map +0 -1
  556. package/dist/eval-workflows/messages.d.ts.map +0 -1
  557. package/dist/eval-workflows/messages.js.map +0 -1
  558. package/dist/eval-workflows/run-evaluation.d.ts.map +0 -1
  559. package/dist/eval-workflows/run-evaluation.js.map +0 -1
  560. package/dist/executors/anthropic-api.d.ts.map +0 -1
  561. package/dist/executors/anthropic-api.js.map +0 -1
  562. package/dist/executors/claude-cli.d.ts.map +0 -1
  563. package/dist/executors/claude-cli.js.map +0 -1
  564. package/dist/executors/claude-sdk-trace.d.ts.map +0 -1
  565. package/dist/executors/claude-sdk-trace.js.map +0 -1
  566. package/dist/executors/claude-sdk.d.ts.map +0 -1
  567. package/dist/executors/claude-sdk.js.map +0 -1
  568. package/dist/executors/codex-cli-trace.d.ts.map +0 -1
  569. package/dist/executors/codex-cli-trace.js.map +0 -1
  570. package/dist/executors/codex-cli.d.ts.map +0 -1
  571. package/dist/executors/codex-cli.js.map +0 -1
  572. package/dist/executors/codex-sdk.d.ts.map +0 -1
  573. package/dist/executors/codex-sdk.js.map +0 -1
  574. package/dist/executors/gemini.d.ts.map +0 -1
  575. package/dist/executors/gemini.js.map +0 -1
  576. package/dist/executors/index.d.ts.map +0 -1
  577. package/dist/executors/index.js.map +0 -1
  578. package/dist/executors/openai-api.d.ts.map +0 -1
  579. package/dist/executors/openai-api.js.map +0 -1
  580. package/dist/executors/runtime-fingerprint.d.ts.map +0 -1
  581. package/dist/executors/runtime-fingerprint.js.map +0 -1
  582. package/dist/executors/script.d.ts.map +0 -1
  583. package/dist/executors/script.js.map +0 -1
  584. package/dist/executors/shared.d.ts.map +0 -1
  585. package/dist/executors/shared.js.map +0 -1
  586. package/dist/grading/assertions.d.ts.map +0 -1
  587. package/dist/grading/assertions.js.map +0 -1
  588. package/dist/grading/debias-validate.d.ts.map +0 -1
  589. package/dist/grading/debias-validate.js.map +0 -1
  590. package/dist/grading/diagnostic.d.ts.map +0 -1
  591. package/dist/grading/diagnostic.js.map +0 -1
  592. package/dist/grading/gold-cli.d.ts.map +0 -1
  593. package/dist/grading/gold-cli.js.map +0 -1
  594. package/dist/grading/gold-dataset.d.ts.map +0 -1
  595. package/dist/grading/gold-dataset.js.map +0 -1
  596. package/dist/grading/human-gold.d.ts.map +0 -1
  597. package/dist/grading/human-gold.js.map +0 -1
  598. package/dist/grading/index.d.ts.map +0 -1
  599. package/dist/grading/index.js.map +0 -1
  600. package/dist/grading/judge.d.ts.map +0 -1
  601. package/dist/grading/judge.js.map +0 -1
  602. package/dist/grading/layered-scores.d.ts.map +0 -1
  603. package/dist/grading/layered-scores.js.map +0 -1
  604. package/dist/inputs/eval-config.d.ts.map +0 -1
  605. package/dist/inputs/eval-config.js.map +0 -1
  606. package/dist/inputs/load-samples.d.ts.map +0 -1
  607. package/dist/inputs/load-samples.js.map +0 -1
  608. package/dist/inputs/mcp-resolver.d.ts.map +0 -1
  609. package/dist/inputs/mcp-resolver.js.map +0 -1
  610. package/dist/inputs/skill-loader.d.ts.map +0 -1
  611. package/dist/inputs/skill-loader.js.map +0 -1
  612. package/dist/inputs/url-fetcher.d.ts.map +0 -1
  613. package/dist/inputs/url-fetcher.js.map +0 -1
  614. package/dist/observability/experience-frontmatter.d.ts.map +0 -1
  615. package/dist/observability/experience-frontmatter.js.map +0 -1
  616. package/dist/observability/experience.d.ts.map +0 -1
  617. package/dist/observability/experience.js.map +0 -1
  618. package/dist/observability/feedback-matchers.d.ts.map +0 -1
  619. package/dist/observability/feedback-matchers.js.map +0 -1
  620. package/dist/observability/feedback-projection.d.ts.map +0 -1
  621. package/dist/observability/feedback-projection.js.map +0 -1
  622. package/dist/observability/inbox-view-model.d.ts.map +0 -1
  623. package/dist/observability/inbox-view-model.js.map +0 -1
  624. package/dist/observability/inbox.d.ts.map +0 -1
  625. package/dist/observability/inbox.js.map +0 -1
  626. package/dist/observability/problem-patterns.d.ts.map +0 -1
  627. package/dist/observability/problem-patterns.js.map +0 -1
  628. package/dist/observability/resolved-review.d.ts.map +0 -1
  629. package/dist/observability/resolved-review.js.map +0 -1
  630. package/dist/observability/review-state.d.ts.map +0 -1
  631. package/dist/observability/review-state.js.map +0 -1
  632. package/dist/observability/skill-chain-advisories.d.ts.map +0 -1
  633. package/dist/observability/skill-chain-advisories.js.map +0 -1
  634. package/dist/observability/skill-chain.d.ts.map +0 -1
  635. package/dist/observability/skill-chain.js.map +0 -1
  636. package/dist/observability/skill-health-analyzer.d.ts.map +0 -1
  637. package/dist/observability/skill-health-analyzer.js.map +0 -1
  638. package/dist/observability/soft-standards/constants.d.ts.map +0 -1
  639. package/dist/observability/soft-standards/constants.js.map +0 -1
  640. package/dist/observability/soft-standards/index.d.ts.map +0 -1
  641. package/dist/observability/soft-standards/index.js.map +0 -1
  642. package/dist/observability/soft-standards/llm-extractor.d.ts.map +0 -1
  643. package/dist/observability/soft-standards/llm-extractor.js.map +0 -1
  644. package/dist/observability/soft-standards/runtime-evaluator.d.ts.map +0 -1
  645. package/dist/observability/soft-standards/runtime-evaluator.js.map +0 -1
  646. package/dist/observability/soft-standards/skill-standards-store.d.ts.map +0 -1
  647. package/dist/observability/soft-standards/skill-standards-store.js.map +0 -1
  648. package/dist/observability/soft-standards/types.d.ts.map +0 -1
  649. package/dist/observability/soft-standards/types.js.map +0 -1
  650. package/dist/observability/text-signals.d.ts.map +0 -1
  651. package/dist/observability/text-signals.js.map +0 -1
  652. package/dist/observability/trace-adapter.d.ts.map +0 -1
  653. package/dist/observability/trace-adapter.js.map +0 -1
  654. package/dist/observability/trace-attribution.d.ts.map +0 -1
  655. package/dist/observability/trace-attribution.js.map +0 -1
  656. package/dist/observability/trace-segmenter.d.ts.map +0 -1
  657. package/dist/observability/trace-segmenter.js.map +0 -1
  658. package/dist/observability/trace-source.d.ts.map +0 -1
  659. package/dist/observability/trace-source.js.map +0 -1
  660. package/dist/renderer/html-renderer.d.ts.map +0 -1
  661. package/dist/renderer/html-renderer.js.map +0 -1
  662. package/dist/renderer/layout.d.ts.map +0 -1
  663. package/dist/renderer/layout.js.map +0 -1
  664. package/dist/renderer/observation-inbox/helpers.d.ts.map +0 -1
  665. package/dist/renderer/observation-inbox/helpers.js.map +0 -1
  666. package/dist/renderer/observation-inbox/styles.d.ts.map +0 -1
  667. package/dist/renderer/observation-inbox/styles.js.map +0 -1
  668. package/dist/renderer/observation-inbox-renderer.d.ts.map +0 -1
  669. package/dist/renderer/observation-inbox-renderer.js.map +0 -1
  670. package/dist/renderer/skill-detail-renderer.d.ts.map +0 -1
  671. package/dist/renderer/skill-detail-renderer.js.map +0 -1
  672. package/dist/renderer/skill-health-renderer.d.ts.map +0 -1
  673. package/dist/renderer/skill-health-renderer.js.map +0 -1
  674. package/dist/renderer/skill-list-renderer.d.ts.map +0 -1
  675. package/dist/renderer/skill-list-renderer.js.map +0 -1
  676. package/dist/renderer/summary.d.ts.map +0 -1
  677. package/dist/renderer/summary.js.map +0 -1
  678. package/dist/renderer/table.d.ts.map +0 -1
  679. package/dist/renderer/table.js.map +0 -1
  680. package/dist/renderer/test-view.d.ts.map +0 -1
  681. package/dist/renderer/test-view.js.map +0 -1
  682. package/dist/renderer/trends.d.ts.map +0 -1
  683. package/dist/renderer/trends.js.map +0 -1
  684. package/dist/server/job-store.d.ts.map +0 -1
  685. package/dist/server/job-store.js.map +0 -1
  686. package/dist/server/report-server.d.ts.map +0 -1
  687. package/dist/server/report-server.js.map +0 -1
  688. package/dist/server/report-store.d.ts.map +0 -1
  689. package/dist/server/report-store.js.map +0 -1
  690. package/dist/server/skill-index.d.ts.map +0 -1
  691. package/dist/server/skill-index.js.map +0 -1
  692. package/dist/server/skill-insights.d.ts.map +0 -1
  693. package/dist/server/skill-insights.js.map +0 -1
  694. package/dist/shared/hard-rules.d.ts.map +0 -1
  695. package/dist/shared/hard-rules.js.map +0 -1
  696. package/dist/shared/llm-prompts/index.d.ts.map +0 -1
  697. package/dist/shared/llm-prompts/index.js.map +0 -1
  698. package/dist/shared/llm-prompts/skill-health.d.ts.map +0 -1
  699. package/dist/shared/llm-prompts/skill-health.js.map +0 -1
  700. package/dist/shared/time.d.ts.map +0 -1
  701. package/dist/shared/time.js.map +0 -1
  702. package/dist/shared/tool-search.d.ts.map +0 -1
  703. package/dist/shared/tool-search.js.map +0 -1
  704. package/dist/types/dependencies.d.ts.map +0 -1
  705. package/dist/types/dependencies.js.map +0 -1
  706. package/dist/types/diagnosis.d.ts.map +0 -1
  707. package/dist/types/diagnosis.js.map +0 -1
  708. package/dist/types/doctor.d.ts.map +0 -1
  709. package/dist/types/doctor.js.map +0 -1
  710. package/dist/types/eval.d.ts.map +0 -1
  711. package/dist/types/eval.js.map +0 -1
  712. package/dist/types/executor.d.ts.map +0 -1
  713. package/dist/types/executor.js.map +0 -1
  714. package/dist/types/index.d.ts.map +0 -1
  715. package/dist/types/index.js.map +0 -1
  716. package/dist/types/judge.d.ts.map +0 -1
  717. package/dist/types/judge.js.map +0 -1
  718. package/dist/types/observability.d.ts.map +0 -1
  719. package/dist/types/observability.js.map +0 -1
  720. package/dist/types/report.d.ts.map +0 -1
  721. package/dist/types/report.js.map +0 -1
  722. package/dist/types/shared.d.ts.map +0 -1
  723. package/dist/types/shared.js.map +0 -1
  724. package/dist/types/skill-index.d.ts.map +0 -1
  725. package/dist/types/skill-index.js.map +0 -1
  726. package/dist/types/storage.d.ts.map +0 -1
  727. package/dist/types/storage.js.map +0 -1
  728. package/dist/util/safe-slice.d.ts.map +0 -1
  729. package/dist/util/safe-slice.js.map +0 -1
@@ -15,9 +15,96 @@ interface WeakSample {
15
15
  none?: string;
16
16
  };
17
17
  }
18
- export declare function extractWeakSamples(report: Report, variantKey: string, count?: number): WeakSample[];
18
+ export declare function extractWeakSamples(report: Report, variantKey: string, count?: number, sampleIdFilter?: Set<string>): WeakSample[];
19
+ /** A train / holdout partition of a sample set. */
20
+ interface HoldoutSplit {
21
+ trainIds: Set<string>;
22
+ holdoutIds: Set<string>;
23
+ }
24
+ /** A train / val / test partition. `val` drives the accept decision; `test` is
25
+ * locked — never seen during the loop, read once at the end for an unbiased
26
+ * generalization score. */
27
+ interface TrainValTestSplit {
28
+ trainIds: Set<string>;
29
+ valIds: Set<string>;
30
+ testIds: Set<string>;
31
+ }
32
+ /** Below this many decision (val) samples the bootstrap diff CI almost never
33
+ * excludes 0 for realistic effect sizes, so the significance gate would reject
34
+ * every candidate. Under that floor evolve degrades to the point-estimate accept
35
+ * and flags `gate.underpowered`. */
36
+ export declare const MIN_GATE_SAMPLES = 8;
37
+ /**
38
+ * Deterministically split sample ids into train / holdout by `ratio` (fraction
39
+ * held out). Holdout members are picked at an even stride so the partition is
40
+ * representative of the ordering, and the split is stable across rounds and runs
41
+ * (no RNG). Returns null when ratio ≤ 0 or either side would drop below
42
+ * MIN_HOLDOUT_SUBSET — the caller then scores on the full set.
43
+ */
44
+ export declare function splitHoldout(sampleIds: string[], ratio: number): HoldoutSplit | null;
45
+ /**
46
+ * Deterministically split sample ids into train / val / test. `val` is carved
47
+ * first at an even stride; `test` is carved at an even stride over what remains,
48
+ * so the three sets are disjoint and stable across rounds/runs (no RNG). Returns
49
+ * null when either ratio ≤ 0 or any of the three sides would drop below
50
+ * MIN_HOLDOUT_SUBSET — the caller then degrades to a 2-way (or full-set) split.
51
+ */
52
+ export declare function splitTrainValTest(sampleIds: string[], valRatio: number, testRatio: number): TrainValTestSplit | null;
53
+ export interface AcceptDecision {
54
+ accepted: boolean;
55
+ /** Diff CI (candidate − best) when the gate ran; absent when it degraded. */
56
+ diffCI?: {
57
+ low: number;
58
+ high: number;
59
+ estimate: number;
60
+ significant: boolean;
61
+ };
62
+ /** True when the gate was requested but the decision set was below MIN_GATE_SAMPLES,
63
+ * so the decision degraded to the point-estimate comparison. */
64
+ underpowered: boolean;
65
+ }
66
+ /**
67
+ * The accept decision for one round. With the significance gate on and enough
68
+ * decision samples, a candidate is accepted only when it is **significantly** above
69
+ * the current best's fresh re-eval (`bootstrapDiffCI(...).significant && estimate > 0`)
70
+ * AND its decision score actually beats the recorded best (`pointCand > pointBest`).
71
+ * The second clause preserves evolve's monotonic invariant: `bestScore` never
72
+ * decreases. Without it, an unlucky (noise-low) re-eval of the current best could let
73
+ * a candidate that is significantly above that re-eval — yet still below the recorded
74
+ * best — win and overwrite the best downward. Off, or under-powered, the gate degrades
75
+ * to the legacy point-estimate comparison alone. Pure (modulo the seeded bootstrap) so
76
+ * the core behavior is unit-testable.
77
+ */
78
+ export declare function decideAccept(bestScores: number[], candScores: number[], pointBest: number, pointCand: number, opts: {
79
+ significanceGate: boolean;
80
+ alpha: number;
81
+ seed: number;
82
+ }): AcceptDecision;
83
+ /**
84
+ * A view of `report` whose results are restricted to `sampleIds`. Used to keep the
85
+ * holdout split out of the sample-fixer: under an active holdout, only training-split
86
+ * samples may enter the --auto-fix-samples prompt or be rewritten — otherwise the
87
+ * skill's samples get tuned to the very samples that decide acceptance, reintroducing
88
+ * the leak holdout exists to prevent.
89
+ */
90
+ export declare function restrictReportToSamples(report: Report, sampleIds: Set<string>): Report;
19
91
  export declare function allNonTripwireAssertionsPass(report: Report, variantKey: string): boolean;
20
- export declare function buildImprovementPrompt(skillContent: string, score: number, weakSamples: WeakSample[]): string;
92
+ interface EditDelta {
93
+ /** Symmetric line difference (added + removed unique lines) over original line count. */
94
+ ratio: number;
95
+ /** Absolute count of added + removed unique lines. */
96
+ changedLines: number;
97
+ /** Compact `+`/`-` summary of the changed lines, truncated. */
98
+ summary: string;
99
+ }
100
+ /**
101
+ * How a candidate differs from the current best, by trimmed non-empty line sets.
102
+ * `ratio` drives the edit budget; `summary` feeds the rejected-edit memory so the
103
+ * improver doesn't re-propose changes that already failed. Order-insensitive and
104
+ * O(n) over small skill files.
105
+ */
106
+ export declare function computeEditDelta(before: string, after: string, maxSummaryLines?: number): EditDelta;
107
+ export declare function buildImprovementPrompt(skillContent: string, score: number, weakSamples: WeakSample[], rejectedEdits?: string[]): string;
21
108
  /** @deprecated Use ProgressCallback from evaluation-core.ts */
22
109
  export type EvolveProgressInfo = Parameters<ProgressCallback>[0];
23
110
  export interface EvolveRoundProgressInfo {
@@ -33,6 +120,10 @@ export interface EvolveRoundProgressInfo {
33
120
  costReported?: boolean;
34
121
  error?: string;
35
122
  reused?: boolean;
123
+ /** When the significance gate ran: whether the candidate's gain was significant.
124
+ * False on a rejected round means "score rose but within noise" — lets the CLI
125
+ * explain an otherwise-confusing `(+0.0x) ✗ REJECT`. Undefined = gate didn't run. */
126
+ significant?: boolean;
36
127
  }
37
128
  interface EvolveOptions {
38
129
  skillPath: string;
@@ -61,15 +152,62 @@ interface EvolveOptions {
61
152
  noDiagnostic?: boolean;
62
153
  /** 跳过 doctor 健康检查门禁。默认 false。 */
63
154
  skipDoctor?: boolean;
155
+ /** Fraction of samples held out for the accept decision (0..1). Default 0 = off.
156
+ * When > 0, a candidate is accepted on its **holdout** composite rather than the
157
+ * training composite, and weak-sample extraction only sees the training split —
158
+ * so the skill is never tuned to the samples that judge it. Too small a split
159
+ * (either side < MIN_HOLDOUT_SUBSET) falls back to full-set scoring + a warning. */
160
+ holdoutRatio?: number;
161
+ /** Statistically gate acceptance: a candidate is accepted only when its
162
+ * per-sample composite is **significantly** above the current best on the
163
+ * decision (val) set — `bootstrapDiffCI(...).significant && estimate > 0` —
164
+ * not merely numerically higher. Default true (rejecting improvements
165
+ * indistinguishable from judge noise is the point). Below MIN_GATE_SAMPLES
166
+ * decision samples the gate is underpowered and degrades to the point-estimate
167
+ * comparison + a warning. Set false to force the legacy point-estimate accept. */
168
+ significanceGate?: boolean;
169
+ /** Significance level for the accept gate's diff CI. Default 0.05 (95% CI). */
170
+ significanceAlpha?: number;
171
+ /** Fraction of samples locked away as a **test** set (0..1). Default 0 = off.
172
+ * Only honored alongside `holdoutRatio` > 0 (test needs a separate val set to
173
+ * decide on). The test split is never seen during the loop — not by weak-sample
174
+ * extraction, not by the accept gate — and is read exactly once at the end for an
175
+ * unbiased `generalizationScore`. Too small a 3-way split degrades to 2-way. */
176
+ testRatio?: number;
177
+ /** Max fraction of skill lines a single round may change before the candidate is
178
+ * rejected **without paying for evaluation**. Default 0.2 (matches the "≤20%"
179
+ * the improvement prompt already asks for — this enforces it). A small floor
180
+ * always permits a handful of lines so tiny skills aren't frozen. Set 0 to disable. */
181
+ editBudget?: number;
182
+ /** Feed rejected candidate edits back into the next round's improvement prompt
183
+ * ("these were tried and did not help — don't repeat them"). Default true. */
184
+ rejectMemory?: boolean;
64
185
  onProgress?: ProgressCallback | null;
65
186
  onRoundProgress?: ((progress: EvolveRoundProgressInfo) => void) | null;
66
187
  }
67
188
  interface TrajectoryEntry {
68
189
  round: number;
190
+ /** Accept-decision score: val composite when a holdout split is active, else full-set. */
69
191
  score: number;
70
192
  delta: number;
71
193
  accepted: boolean;
72
194
  costUSD: number;
195
+ /** Present when a holdout split is active: the training-split composite (improvement signal). */
196
+ trainScore?: number;
197
+ /** Present when a holdout split is active: the val-split composite (== score). */
198
+ holdoutScore?: number;
199
+ /** Significance-gate diff CI (candidate − current best) on the decision set, when
200
+ * the gate was powered enough to run. `significant` 决定接受。 */
201
+ diffCI?: {
202
+ low: number;
203
+ high: number;
204
+ estimate: number;
205
+ significant: boolean;
206
+ };
207
+ /** Fraction of skill lines this candidate changed vs the current best. */
208
+ editRatio?: number;
209
+ /** True when the candidate was rejected by the edit budget before evaluation. */
210
+ rejectedPreEval?: boolean;
73
211
  }
74
212
  export interface EvolveResult {
75
213
  startScore: number;
@@ -86,6 +224,34 @@ export interface EvolveResult {
86
224
  reusedBaselineReportId?: string;
87
225
  /** False = 任一轮的 exec / judge 不报 cost → totalCostUSD 是 lower-bound 而非真值。 */
88
226
  costReported?: boolean;
227
+ /** Holdout split summary when `--holdout-ratio` > 0. `disabled` is true when the
228
+ * split was too small and evolve fell back to full-set scoring (CLI formats the
229
+ * user-facing message bilingually). */
230
+ holdout?: {
231
+ ratio: number;
232
+ trainCount: number;
233
+ holdoutCount: number;
234
+ disabled?: boolean;
235
+ };
236
+ /** Locked-test split summary. `disabled` is true when `--test-ratio` was requested
237
+ * but the 3-way split was too small, so evolve fell back to a 2-way holdout and
238
+ * produced no generalization score. */
239
+ test?: {
240
+ ratio: number;
241
+ count: number;
242
+ disabled?: boolean;
243
+ };
244
+ /** Unbiased composite of the best skill on the locked test set — the headline honest
245
+ * number. Present only when a 3-way split was active. The test set never influenced
246
+ * selection or weak-sample extraction, so this is an out-of-sample estimate. */
247
+ generalizationScore?: number;
248
+ /** Accept-gate summary. `underpowered` = the decision set was below MIN_GATE_SAMPLES
249
+ * at least once, so the gate degraded to the point-estimate comparison + warned. */
250
+ gate?: {
251
+ enabled: boolean;
252
+ alpha: number;
253
+ underpowered?: boolean;
254
+ };
89
255
  trajectory: TrajectoryEntry[];
90
256
  bestSkillPath: string;
91
257
  allVersions: string[];
@@ -97,6 +263,5 @@ export interface RoundReport {
97
263
  report: Report;
98
264
  }
99
265
  export declare function mergeEvolveReports(roundReports: RoundReport[], skillName: string, totalCostUSD: number, samples?: Sample[], skillPath?: string): Report;
100
- export declare function evolveSkill({ skillPath, samplesPath, rounds, target, stopOnAssertionsPass, autoFixSamples, sampleFixMaxAttempts, reuseLatestEval, model, judgeModels, improveModel, improveMode, executorName, concurrency, timeoutMs, skipConnectivity, effort, noDiagnostic, skipDoctor, onProgress, onRoundProgress, }: EvolveOptions): Promise<EvolveResult>;
266
+ export declare function evolveSkill({ skillPath, samplesPath, rounds, target, stopOnAssertionsPass, autoFixSamples, sampleFixMaxAttempts, reuseLatestEval, model, judgeModels, improveModel, improveMode, executorName, concurrency, timeoutMs, skipConnectivity, effort, noDiagnostic, skipDoctor, holdoutRatio, significanceGate, significanceAlpha, testRatio, editBudget, rejectMemory, onProgress, onRoundProgress, }: EvolveOptions): Promise<EvolveResult>;
101
267
  export {};
102
- //# sourceMappingURL=evolver.d.ts.map
@@ -6,6 +6,8 @@ import { persistReport, DEFAULT_OUTPUT_DIR, generateRunId, hashString } from '..
6
6
  import { createFileStore } from '../server/report-store.js';
7
7
  import { analyzeResults } from '../analysis/report-diagnostics.js';
8
8
  import { loadSamples } from '../inputs/load-samples.js';
9
+ import { buildVariantSummary } from '../eval-core/schema.js';
10
+ import { bootstrapDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
9
11
  import { fixSamples } from './sample-fixer.js';
10
12
  const IMPROVE_SYSTEM_PROMPT = `你是一个 AI 提示词改进专家。你的任务是分析评测结果中的薄弱环节,针对性地改进 skill(系统提示词),使其在评测中获得更高的分数。
11
13
 
@@ -121,9 +123,11 @@ function readSkillName(skillPath) {
121
123
  return null;
122
124
  }
123
125
  }
124
- export function extractWeakSamples(report, variantKey, count = 5) {
126
+ export function extractWeakSamples(report, variantKey, count = 5, sampleIdFilter) {
125
127
  const weakSamples = [];
126
128
  for (const r of report.results) {
129
+ if (sampleIdFilter && !sampleIdFilter.has(r.sample_id))
130
+ continue;
127
131
  const v = r.variants[variantKey];
128
132
  if (!v || typeof v.compositeScore !== 'number')
129
133
  continue;
@@ -153,6 +157,134 @@ export function extractWeakSamples(report, variantKey, count = 5) {
153
157
  .sort((a, b) => a.compositeScore - b.compositeScore)
154
158
  .slice(0, count);
155
159
  }
160
+ /** Below this many samples on any side, a split is too small to be meaningful —
161
+ * evolve falls back to full-set scoring and warns. */
162
+ const MIN_HOLDOUT_SUBSET = 3;
163
+ /** Below this many decision (val) samples the bootstrap diff CI almost never
164
+ * excludes 0 for realistic effect sizes, so the significance gate would reject
165
+ * every candidate. Under that floor evolve degrades to the point-estimate accept
166
+ * and flags `gate.underpowered`. */
167
+ export const MIN_GATE_SAMPLES = 8;
168
+ /** Pick `count` ids at an even stride across `ids` (deterministic, no RNG) so the
169
+ * picked subset is representative of the ordering and stable across rounds/runs. */
170
+ function pickByStride(ids, count) {
171
+ const picked = new Set();
172
+ if (count <= 0)
173
+ return picked;
174
+ const stride = ids.length / count;
175
+ for (let k = 0; k < count; k++)
176
+ picked.add(ids[Math.floor(k * stride)]);
177
+ return picked;
178
+ }
179
+ /**
180
+ * Deterministically split sample ids into train / holdout by `ratio` (fraction
181
+ * held out). Holdout members are picked at an even stride so the partition is
182
+ * representative of the ordering, and the split is stable across rounds and runs
183
+ * (no RNG). Returns null when ratio ≤ 0 or either side would drop below
184
+ * MIN_HOLDOUT_SUBSET — the caller then scores on the full set.
185
+ */
186
+ export function splitHoldout(sampleIds, ratio) {
187
+ if (!(ratio > 0) || sampleIds.length === 0)
188
+ return null;
189
+ const holdoutCount = Math.round(sampleIds.length * ratio);
190
+ const trainCount = sampleIds.length - holdoutCount;
191
+ if (holdoutCount < MIN_HOLDOUT_SUBSET || trainCount < MIN_HOLDOUT_SUBSET)
192
+ return null;
193
+ const holdoutIds = pickByStride(sampleIds, holdoutCount);
194
+ const trainIds = new Set(sampleIds.filter((id) => !holdoutIds.has(id)));
195
+ return { trainIds, holdoutIds };
196
+ }
197
+ /**
198
+ * Deterministically split sample ids into train / val / test. `val` is carved
199
+ * first at an even stride; `test` is carved at an even stride over what remains,
200
+ * so the three sets are disjoint and stable across rounds/runs (no RNG). Returns
201
+ * null when either ratio ≤ 0 or any of the three sides would drop below
202
+ * MIN_HOLDOUT_SUBSET — the caller then degrades to a 2-way (or full-set) split.
203
+ */
204
+ export function splitTrainValTest(sampleIds, valRatio, testRatio) {
205
+ if (!(valRatio > 0) || !(testRatio > 0) || sampleIds.length === 0)
206
+ return null;
207
+ const valCount = Math.round(sampleIds.length * valRatio);
208
+ const testCount = Math.round(sampleIds.length * testRatio);
209
+ const trainCount = sampleIds.length - valCount - testCount;
210
+ if (valCount < MIN_HOLDOUT_SUBSET || testCount < MIN_HOLDOUT_SUBSET || trainCount < MIN_HOLDOUT_SUBSET)
211
+ return null;
212
+ const valIds = pickByStride(sampleIds, valCount);
213
+ const remaining = sampleIds.filter((id) => !valIds.has(id));
214
+ const testIds = pickByStride(remaining, testCount);
215
+ const trainIds = new Set(sampleIds.filter((id) => !valIds.has(id) && !testIds.has(id)));
216
+ return { trainIds, valIds, testIds };
217
+ }
218
+ /**
219
+ * Mean composite over the subset of a report's results whose sample_id is in
220
+ * `ids`, using the same aggregation as the full-run summary
221
+ * (`buildVariantSummary`) so train / holdout scores stay comparable to the
222
+ * headline composite. Returns 0 when the subset has no scorable entries.
223
+ */
224
+ function subsetCompositeScore(report, variantKey, ids) {
225
+ const entries = [];
226
+ for (const r of report.results) {
227
+ if (!ids.has(r.sample_id))
228
+ continue;
229
+ const v = r.variants[variantKey];
230
+ if (v)
231
+ entries.push(v);
232
+ }
233
+ if (entries.length === 0)
234
+ return 0;
235
+ return buildVariantSummary(entries).avgCompositeScore ?? 0;
236
+ }
237
+ /**
238
+ * Per-sample composite scores over the subset of a report's results whose
239
+ * sample_id is in `ids`, in result order. Feeds `bootstrapDiffCI` for the
240
+ * significance accept gate — the array (not the mean) is what the bootstrap
241
+ * resamples. Entries without a numeric compositeScore are skipped.
242
+ */
243
+ function perSampleComposite(report, variantKey, ids) {
244
+ const scores = [];
245
+ for (const r of report.results) {
246
+ if (!ids.has(r.sample_id))
247
+ continue;
248
+ const v = r.variants[variantKey];
249
+ if (v && typeof v.compositeScore === 'number')
250
+ scores.push(v.compositeScore);
251
+ }
252
+ return scores;
253
+ }
254
+ /**
255
+ * The accept decision for one round. With the significance gate on and enough
256
+ * decision samples, a candidate is accepted only when it is **significantly** above
257
+ * the current best's fresh re-eval (`bootstrapDiffCI(...).significant && estimate > 0`)
258
+ * AND its decision score actually beats the recorded best (`pointCand > pointBest`).
259
+ * The second clause preserves evolve's monotonic invariant: `bestScore` never
260
+ * decreases. Without it, an unlucky (noise-low) re-eval of the current best could let
261
+ * a candidate that is significantly above that re-eval — yet still below the recorded
262
+ * best — win and overwrite the best downward. Off, or under-powered, the gate degrades
263
+ * to the legacy point-estimate comparison alone. Pure (modulo the seeded bootstrap) so
264
+ * the core behavior is unit-testable.
265
+ */
266
+ export function decideAccept(bestScores, candScores, pointBest, pointCand, opts) {
267
+ const powered = bestScores.length >= MIN_GATE_SAMPLES && candScores.length >= MIN_GATE_SAMPLES;
268
+ if (opts.significanceGate && powered) {
269
+ const diff = bootstrapDiffCI(bestScores, candScores, opts.alpha, DEFAULT_BOOTSTRAP_SAMPLES, opts.seed);
270
+ return {
271
+ accepted: diff.significant && diff.estimate > 0 && pointCand > pointBest,
272
+ diffCI: { low: diff.low, high: diff.high, estimate: diff.estimate, significant: diff.significant },
273
+ underpowered: false,
274
+ };
275
+ }
276
+ return { accepted: pointCand > pointBest, underpowered: opts.significanceGate && !powered };
277
+ }
278
+ /**
279
+ * A view of `report` whose results are restricted to `sampleIds`. Used to keep the
280
+ * holdout split out of the sample-fixer: under an active holdout, only training-split
281
+ * samples may enter the --auto-fix-samples prompt or be rewritten — otherwise the
282
+ * skill's samples get tuned to the very samples that decide acceptance, reintroducing
283
+ * the leak holdout exists to prevent.
284
+ */
285
+ export function restrictReportToSamples(report, sampleIds) {
286
+ return { ...report, results: report.results.filter((r) => sampleIds.has(r.sample_id)) };
287
+ }
156
288
  export function allNonTripwireAssertionsPass(report, variantKey) {
157
289
  for (const entry of report.results) {
158
290
  const variant = entry.variants[variantKey];
@@ -279,7 +411,36 @@ async function autoFixSamplesAfterSkillRound(opts) {
279
411
  }
280
412
  return { fixedCount: result.fixedCount, costUSD: result.costUSD };
281
413
  }
282
- export function buildImprovementPrompt(skillContent, score, weakSamples) {
414
+ /** Number of skill lines below which the edit budget never trips — so a tiny skill
415
+ * isn't frozen by a percentage threshold that a few lines already blow past. */
416
+ const EDIT_BUDGET_FLOOR_LINES = 10;
417
+ /**
418
+ * How a candidate differs from the current best, by trimmed non-empty line sets.
419
+ * `ratio` drives the edit budget; `summary` feeds the rejected-edit memory so the
420
+ * improver doesn't re-propose changes that already failed. Order-insensitive and
421
+ * O(n) over small skill files.
422
+ */
423
+ export function computeEditDelta(before, after, maxSummaryLines = 12) {
424
+ const beforeArr = before.split('\n').map((l) => l.trim()).filter(Boolean);
425
+ const afterArr = after.split('\n').map((l) => l.trim()).filter(Boolean);
426
+ const beforeSet = new Set(beforeArr);
427
+ const afterSet = new Set(afterArr);
428
+ const added = [...afterSet].filter((l) => !beforeSet.has(l));
429
+ const removed = [...beforeSet].filter((l) => !afterSet.has(l));
430
+ const changedLines = added.length + removed.length;
431
+ const ratio = changedLines / Math.max(beforeArr.length, 1);
432
+ const parts = [];
433
+ for (const l of added.slice(0, maxSummaryLines))
434
+ parts.push(`+ ${l}`);
435
+ if (added.length > maxSummaryLines)
436
+ parts.push(`+ …(其余 +${added.length - maxSummaryLines} 行)`);
437
+ for (const l of removed.slice(0, maxSummaryLines))
438
+ parts.push(`- ${l}`);
439
+ if (removed.length > maxSummaryLines)
440
+ parts.push(`- …(其余 -${removed.length - maxSummaryLines} 行)`);
441
+ return { ratio, changedLines, summary: parts.join('\n') || '(无文本差异)' };
442
+ }
443
+ export function buildImprovementPrompt(skillContent, score, weakSamples, rejectedEdits) {
283
444
  const weakDetails = weakSamples.map((s) => {
284
445
  const parts = [`### ${s.sample_id}(${s.compositeScore}/5.0)`];
285
446
  if (s.llmReason)
@@ -308,13 +469,16 @@ export function buildImprovementPrompt(skillContent, score, weakSamples) {
308
469
  }
309
470
  return parts.join('\n');
310
471
  }).join('\n\n');
472
+ const rejectedSection = rejectedEdits && rejectedEdits.length > 0
473
+ ? `\n\n## 已试过且未带来显著提升的改法(不要重复)\n\n${rejectedEdits.join('\n\n')}`
474
+ : '';
311
475
  return `## 当前 Skill(平均分: ${score.toFixed(2)}/5.0)
312
476
 
313
477
  ${skillContent}
314
478
 
315
479
  ## 低分用例分析
316
480
 
317
- ${weakDetails || '(无低分用例)'}`;
481
+ ${weakDetails || '(无低分用例)'}${rejectedSection}`;
318
482
  }
319
483
  function buildImprovementSuffix(mode, candidatePath) {
320
484
  if (mode === 'agent') {
@@ -377,7 +541,7 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
377
541
  }
378
542
  const runId = `evolve-${skillName}-${generateRunId([skillName]).split('-').slice(-2).join('-')}`;
379
543
  const report = {
380
- kind: 'evaluation',
544
+ reportKind: 'evaluation',
381
545
  id: runId,
382
546
  meta: {
383
547
  ...firstReport.meta,
@@ -401,7 +565,7 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
401
565
  report.analysis = analyzeResults(report, { samples });
402
566
  return report;
403
567
  }
404
- export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model = DEFAULT_MODEL, judgeModels, improveModel = DEFAULT_MODEL, improveMode = 'agent', executorName = 'claude', concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, onProgress = null, onRoundProgress = null, }) {
568
+ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model = DEFAULT_MODEL, judgeModels, improveModel = DEFAULT_MODEL, improveMode = 'agent', executorName = 'claude', concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, holdoutRatio = 0, significanceGate = true, significanceAlpha = DEFAULT_BOOTSTRAP_ALPHA, testRatio = 0, editBudget = 0.2, rejectMemory = true, onProgress = null, onRoundProgress = null, }) {
405
569
  if (judgeModels && judgeModels.length > 1) {
406
570
  throw new Error('evolveSkill does not support multi-judge ensemble (received '
407
571
  + `${judgeModels.length} judges). Pass a single-judge array, e.g. `
@@ -421,6 +585,72 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
421
585
  if (!existsSync(absSamplesPath))
422
586
  throw new Error(`samples file not found: ${absSamplesPath}`);
423
587
  mkdirSync(evolveDir, { recursive: true });
588
+ // Split (opt-in). Computed once over the canonical sample order so it's stable
589
+ // across rounds. `val` drives the accept decision; weak-sample extraction and the
590
+ // sample-fixer only ever see `train`; `test` is locked away — never seen during the
591
+ // loop — and read once at the end for an unbiased generalization score. With no
592
+ // holdout the decision runs on the full set (legacy). A too-small 3-way split
593
+ // degrades to 2-way, then to full-set.
594
+ const allSampleIds = loadSamples(absSamplesPath).samples.map((s) => s.sample_id);
595
+ const threeWay = (testRatio > 0 && holdoutRatio > 0) ? splitTrainValTest(allSampleIds, holdoutRatio, testRatio) : null;
596
+ const twoWay = (!threeWay && holdoutRatio > 0) ? splitHoldout(allSampleIds, holdoutRatio) : null;
597
+ const split = threeWay
598
+ ? { trainIds: threeWay.trainIds, valIds: threeWay.valIds, testIds: threeWay.testIds }
599
+ : twoWay
600
+ ? { trainIds: twoWay.trainIds, valIds: twoWay.holdoutIds, testIds: null }
601
+ : null;
602
+ const holdoutInfo = holdoutRatio > 0
603
+ ? {
604
+ ratio: holdoutRatio,
605
+ trainCount: split?.trainIds.size ?? allSampleIds.length,
606
+ holdoutCount: split?.valIds.size ?? 0,
607
+ ...(split ? {} : { disabled: true }),
608
+ }
609
+ : undefined;
610
+ // test 被请求(配了 --holdout-ratio)但 3-way 太小回退 → 标 disabled,别让用户
611
+ // 以为拿到了 locked-test 泛化分。
612
+ const testRequested = testRatio > 0 && holdoutRatio > 0;
613
+ const testInfo = threeWay
614
+ ? { ratio: testRatio, count: threeWay.testIds.size }
615
+ : testRequested ? { ratio: testRatio, count: 0, disabled: true } : undefined;
616
+ // Deterministic seed so the gate's CIs are reproducible across reruns (and
617
+ // assertable in tests). Derived from skill identity + sample count, parsed to a uint32.
618
+ const gateSeed = parseInt(hashString(`${skillName}:${allSampleIds.length}`).slice(0, 8), 16) >>> 0;
619
+ let gateUnderpowered = false;
620
+ // Accept-decision score for a report's variant: val composite when a split is
621
+ // active, otherwise the full-set composite (legacy behavior).
622
+ const decisionScore = (report, key) => split
623
+ ? subsetCompositeScore(report, key, split.valIds)
624
+ : (report.summary[key]?.avgCompositeScore ?? 0);
625
+ const trainScoreOf = (report, key) => split ? subsetCompositeScore(report, key, split.trainIds) : undefined;
626
+ // Per-round trajectory tail: train / holdout breakdown, only when split active.
627
+ const splitScores = (report, key, decision) => split ? { trainScore: trainScoreOf(report, key), holdoutScore: decision } : {};
628
+ // Unbiased generalization: the best skill's composite on the locked test set,
629
+ // read once at the very end. Present only under a valid 3-way split; the test
630
+ // set never influenced selection or weak-sample extraction.
631
+ const buildGeneralization = () => {
632
+ if (!testInfo)
633
+ return {};
634
+ if (!threeWay || !split?.testIds)
635
+ return { test: testInfo }; // requested but degraded → disabled, no score
636
+ const best = roundReports.find((r) => r.round === bestRound)?.report;
637
+ if (!best)
638
+ return { test: testInfo };
639
+ const key = Object.keys(best.summary)[0];
640
+ return { test: testInfo, generalizationScore: Number(subsetCompositeScore(best, key, split.testIds).toFixed(4)) };
641
+ };
642
+ const gateInfo = () => ({ enabled: significanceGate, alpha: significanceAlpha, ...(gateUnderpowered ? { underpowered: true } : {}) });
643
+ // Rejected-edit memory (most recent K). Fed back into the next round's improvement
644
+ // prompt so the improver doesn't re-propose changes that already failed to help.
645
+ const rejectedEdits = [];
646
+ const REJECT_MEMORY_K = 3;
647
+ const rememberRejected = (round, summary, reason) => {
648
+ if (!rejectMemory)
649
+ return;
650
+ rejectedEdits.push(`【第 ${round} 轮被拒(${reason})】\n${summary}`);
651
+ if (rejectedEdits.length > REJECT_MEMORY_K)
652
+ rejectedEdits.shift();
653
+ };
424
654
  // Save original as r0
425
655
  let currentBest = readFileSync(absSkillPath, 'utf-8').trim();
426
656
  const r0Path = join(evolveDir, `${skillName}.r0.md`);
@@ -461,13 +691,13 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
461
691
  });
462
692
  }
463
693
  const baselineVariantKey = Object.keys(baselineReport.summary)[0];
464
- bestScore = baselineReport.summary[baselineVariantKey]?.avgCompositeScore ?? 0;
694
+ bestScore = decisionScore(baselineReport, baselineVariantKey);
465
695
  const baselineCost = baselineReused ? 0 : baselineReport.meta.totalCostUSD;
466
696
  totalCostUSD += baselineCost;
467
697
  const baselineCostReported = baselineReused || !reportHasUnreportedCost(baselineReport);
468
698
  if (!baselineCostReported)
469
699
  totalCostReported = false;
470
- trajectory.push({ round: 0, score: bestScore, delta: 0, accepted: true, costUSD: baselineCost });
700
+ trajectory.push({ round: 0, score: bestScore, delta: 0, accepted: true, costUSD: baselineCost, ...splitScores(baselineReport, baselineVariantKey, bestScore) });
471
701
  roundReports.push({ round: 0, accepted: true, report: baselineReport });
472
702
  if (onRoundProgress)
473
703
  onRoundProgress({ round: 0, totalRounds: rounds, phase: 'baseline', score: bestScore, costUSD: baselineCost, costReported: baselineCostReported, reused: baselineReused });
@@ -487,6 +717,9 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
487
717
  ...(sampleFixes.length > 0 && { sampleFixes }),
488
718
  ...(reusedBaselineReportId && { reusedBaselineReportId }),
489
719
  ...(totalCostReported ? {} : { costReported: false }),
720
+ ...(holdoutInfo ? { holdout: holdoutInfo } : {}),
721
+ ...buildGeneralization(),
722
+ gate: gateInfo(),
490
723
  trajectory,
491
724
  bestSkillPath: allVersions[bestRound],
492
725
  allVersions,
@@ -513,10 +746,10 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
513
746
  totalCostReported = false;
514
747
  }
515
748
  const lastVariantKey = Object.keys(lastReport.summary)[0];
516
- const weakSamples = extractWeakSamples(lastReport, lastVariantKey);
749
+ const weakSamples = extractWeakSamples(lastReport, lastVariantKey, 5, split?.trainIds);
517
750
  // Generate improvement
518
751
  const candidatePath = join(evolveDir, `${skillName}.r${round}.md`);
519
- const basePrompt = buildImprovementPrompt(currentBest, bestScore, weakSamples);
752
+ const basePrompt = buildImprovementPrompt(currentBest, bestScore, weakSamples, rejectMemory ? rejectedEdits : undefined);
520
753
  const executor = createExecutor(executorName);
521
754
  let candidateContent;
522
755
  let improveCostUSD;
@@ -570,12 +803,32 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
570
803
  if (!improveCostReported)
571
804
  totalCostReported = false;
572
805
  allVersions.push(candidatePath);
806
+ // Edit budget: reject oversized rewrites BEFORE paying for evaluation. The
807
+ // improvement prompt already asks for ≤ editBudget of lines changed — this
808
+ // enforces it. A small floor still lets tiny skills change a handful of lines.
809
+ const editDelta = computeEditDelta(currentBest, candidateContent);
810
+ if (editBudget > 0 && editDelta.ratio > editBudget && editDelta.changedLines > EDIT_BUDGET_FLOOR_LINES) {
811
+ totalCostUSD += improveCostUSD;
812
+ consecutiveRejects++;
813
+ const reason = `改动过大 ${(editDelta.ratio * 100).toFixed(0)}%(预算 ${(editBudget * 100).toFixed(0)}%),评测前判拒`;
814
+ rememberRejected(round, editDelta.summary, reason);
815
+ trajectory.push({ round, score: bestScore, delta: 0, accepted: false, costUSD: improveCostUSD, editRatio: Number(editDelta.ratio.toFixed(4)), rejectedPreEval: true });
816
+ if (onRoundProgress)
817
+ onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: bestScore, delta: 0, accepted: false, costUSD: improveCostUSD, costReported: improveCostReported });
818
+ if (consecutiveRejects >= 2) {
819
+ stopReason = 'consecutive-rejects';
820
+ break;
821
+ }
822
+ continue;
823
+ }
573
824
  let preEvalSampleFixCost = 0;
574
825
  if (autoFixSamples) {
575
826
  const sampleFix = await autoFixSamplesAfterSkillRound({
576
827
  samplesPath: absSamplesPath,
577
828
  skillContent: candidateContent,
578
- report: lastReport,
829
+ // Under an active holdout, the sample-fixer may only see training-split samples —
830
+ // never the holdout samples that drive the accept decision (leak guard).
831
+ report: split ? restrictReportToSamples(lastReport, split.trainIds) : lastReport,
579
832
  treatmentKey: lastVariantKey,
580
833
  executorName,
581
834
  model: improveModel,
@@ -593,13 +846,28 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
593
846
  samplesPath: absSamplesPath, skillDir, model, judgeModels: effectiveJudgeModels, executorName, concurrency, timeoutMs, skipConnectivity, effort, noDiagnostic, skipDoctor, onProgress,
594
847
  });
595
848
  const candidateVariantKey = Object.keys(candidateReport.summary)[0];
596
- const candidateScore = candidateReport.summary[candidateVariantKey]?.avgCompositeScore ?? 0;
849
+ const candidateScore = decisionScore(candidateReport, candidateVariantKey);
597
850
  const roundCost = improveCostUSD + preEvalSampleFixCost + candidateReport.meta.totalCostUSD;
598
851
  const roundCostReported = improveCostReported && !reportHasUnreportedCost(candidateReport);
599
852
  if (!roundCostReported)
600
853
  totalCostReported = false;
601
854
  totalCostUSD += improveCostUSD + candidateReport.meta.totalCostUSD;
602
- const accepted = candidateScore > bestScore;
855
+ // Significance accept gate: accept only when the candidate is *significantly*
856
+ // above the current best on the decision (val) set, not merely numerically higher
857
+ // — rejecting gains indistinguishable from judge noise. `lastReport` is the current
858
+ // best's fresh eval and `candidateReport` the candidate's, over the same samples;
859
+ // bootstrapDiffCI resamples the two arrays independently (conservative — not a paired
860
+ // bootstrap). Under-powered decision sets degrade to the legacy point-estimate accept
861
+ // (note: that path compares the prior-round best scalar, not this fresh re-eval) and
862
+ // flag `gate.underpowered`.
863
+ const valIds = split ? split.valIds : new Set(allSampleIds);
864
+ const bestScores = perSampleComposite(lastReport, lastVariantKey, valIds);
865
+ const candScores = perSampleComposite(candidateReport, candidateVariantKey, valIds);
866
+ const decision = decideAccept(bestScores, candScores, bestScore, candidateScore, { significanceGate, alpha: significanceAlpha, seed: gateSeed });
867
+ const accepted = decision.accepted;
868
+ const diffCI = decision.diffCI;
869
+ if (decision.underpowered)
870
+ gateUnderpowered = true;
603
871
  if (accepted) {
604
872
  currentBest = candidateContent;
605
873
  bestScore = candidateScore;
@@ -608,13 +876,15 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
608
876
  }
609
877
  else {
610
878
  consecutiveRejects++;
879
+ const reason = diffCI ? (diffCI.estimate > 0 ? '提升不显著' : '方向为负') : '未超过当前最优';
880
+ rememberRejected(round, editDelta.summary, reason);
611
881
  }
612
882
  if (accepted)
613
883
  roundReports.push({ round, accepted, report: candidateReport });
614
884
  const roundDelta = candidateScore - trajectory[trajectory.length - 1].score;
615
- trajectory.push({ round, score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost });
885
+ trajectory.push({ round, score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, ...splitScores(candidateReport, candidateVariantKey, candidateScore), ...(diffCI ? { diffCI } : {}), editRatio: Number(editDelta.ratio.toFixed(4)) });
616
886
  if (onRoundProgress)
617
- onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, costReported: roundCostReported });
887
+ onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, costReported: roundCostReported, ...(diffCI ? { significant: diffCI.significant } : {}) });
618
888
  // Early stop
619
889
  if (stopOnAssertionsPass && accepted && allNonTripwireAssertionsPass(candidateReport, candidateVariantKey)) {
620
890
  stopReason = 'assertions-pass';
@@ -652,6 +922,9 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
652
922
  ...(sampleFixes.length > 0 && { sampleFixes }),
653
923
  ...(reusedBaselineReportId && { reusedBaselineReportId }),
654
924
  ...(totalCostReported ? {} : { costReported: false }),
925
+ ...(holdoutInfo ? { holdout: holdoutInfo } : {}),
926
+ ...buildGeneralization(),
927
+ gate: gateInfo(),
655
928
  trajectory,
656
929
  bestSkillPath: allVersions[bestRound],
657
930
  allVersions,
@@ -678,4 +951,3 @@ async function evaluate(skillFilePath, { samplesPath, skillDir, model, judgeMode
678
951
  });
679
952
  return report;
680
953
  }
681
- //# sourceMappingURL=evolver.js.map