oh-my-knowledge 0.33.0 → 0.34.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (733) hide show
  1. package/README.md +18 -5
  2. package/README.zh.md +22 -9
  3. package/dist/analysis/coverage-analyzer.d.ts +0 -1
  4. package/dist/analysis/coverage-analyzer.js +0 -1
  5. package/dist/analysis/failure-clusterer.d.ts +0 -1
  6. package/dist/analysis/failure-clusterer.js +0 -1
  7. package/dist/analysis/gap-analyzer.d.ts +0 -1
  8. package/dist/analysis/gap-analyzer.js +0 -1
  9. package/dist/analysis/hedging-classifier.d.ts +0 -1
  10. package/dist/analysis/hedging-classifier.js +1 -2
  11. package/dist/analysis/report-diagnostics.d.ts +0 -1
  12. package/dist/analysis/report-diagnostics.js +0 -1
  13. package/dist/analysis/sample-diagnostics.d.ts +1 -2
  14. package/dist/analysis/sample-diagnostics.js +18 -19
  15. package/dist/analysis/saturation.d.ts +0 -1
  16. package/dist/analysis/saturation.js +0 -1
  17. package/dist/assets/agent-skills/omk/SKILL.md +197 -0
  18. package/dist/assets/agent-skills/omk/references/commands.md +547 -0
  19. package/dist/authoring/evolver.d.ts +130 -6
  20. package/dist/authoring/evolver.js +221 -32
  21. package/dist/authoring/generator.d.ts +0 -1
  22. package/dist/authoring/generator.js +0 -1
  23. package/dist/authoring/sample-fixer.d.ts +0 -1
  24. package/dist/authoring/sample-fixer.js +0 -1
  25. package/dist/cli/commands/doctor.d.ts +0 -1
  26. package/dist/cli/commands/doctor.js +2 -3
  27. package/dist/cli/commands/eval/gold/compare.d.ts +0 -1
  28. package/dist/cli/commands/eval/gold/compare.js +0 -1
  29. package/dist/cli/commands/eval/gold/index.d.ts +0 -1
  30. package/dist/cli/commands/eval/gold/index.js +0 -1
  31. package/dist/cli/commands/eval/gold/init.d.ts +0 -1
  32. package/dist/cli/commands/eval/gold/init.js +0 -1
  33. package/dist/cli/commands/eval/gold/validate.d.ts +0 -1
  34. package/dist/cli/commands/eval/gold/validate.js +0 -1
  35. package/dist/cli/commands/eval/index.d.ts +2 -1
  36. package/dist/cli/commands/eval/index.js +20 -9
  37. package/dist/cli/commands/evolve.d.ts +6 -1
  38. package/dist/cli/commands/evolve.js +77 -6
  39. package/dist/cli/commands/init.d.ts +0 -1
  40. package/dist/cli/commands/init.js +0 -1
  41. package/dist/cli/commands/install.d.ts +22 -0
  42. package/dist/cli/commands/install.js +411 -0
  43. package/dist/cli/commands/observe/inbox.d.ts +0 -1
  44. package/dist/cli/commands/observe/inbox.js +0 -1
  45. package/dist/cli/commands/observe/index.d.ts +0 -1
  46. package/dist/cli/commands/observe/index.js +0 -1
  47. package/dist/cli/commands/observe/ingest.d.ts +0 -1
  48. package/dist/cli/commands/observe/ingest.js +0 -1
  49. package/dist/cli/commands/observe/show.d.ts +0 -1
  50. package/dist/cli/commands/observe/show.js +0 -1
  51. package/dist/cli/commands/sample.d.ts +0 -1
  52. package/dist/cli/commands/sample.js +2 -3
  53. package/dist/cli/commands/studio.d.ts +0 -1
  54. package/dist/cli/commands/studio.js +0 -1
  55. package/dist/cli/index.d.ts +0 -1
  56. package/dist/cli/index.js +1 -2
  57. package/dist/cli/lib/cli-exit.d.ts +0 -1
  58. package/dist/cli/lib/cli-exit.js +0 -1
  59. package/dist/cli/lib/cmd-flags.d.ts +6 -1
  60. package/dist/cli/lib/cmd-flags.js +0 -1
  61. package/dist/cli/lib/i18n-dict/common.d.ts +0 -1
  62. package/dist/cli/lib/i18n-dict/common.js +0 -1
  63. package/dist/cli/lib/i18n-dict/evolve.d.ts +1 -2
  64. package/dist/cli/lib/i18n-dict/evolve.js +16 -1
  65. package/dist/cli/lib/i18n-dict/gen.d.ts +0 -1
  66. package/dist/cli/lib/i18n-dict/gen.js +0 -1
  67. package/dist/cli/lib/i18n-dict/help.d.ts +0 -1
  68. package/dist/cli/lib/i18n-dict/help.js +0 -1
  69. package/dist/cli/lib/i18n-dict/init.d.ts +0 -1
  70. package/dist/cli/lib/i18n-dict/init.js +0 -1
  71. package/dist/cli/lib/i18n-dict/install.d.ts +3 -0
  72. package/dist/cli/lib/i18n-dict/install.js +90 -0
  73. package/dist/cli/lib/i18n-dict/run.d.ts +0 -1
  74. package/dist/cli/lib/i18n-dict/run.js +0 -1
  75. package/dist/cli/lib/i18n-dict/types.d.ts +0 -1
  76. package/dist/cli/lib/i18n-dict/types.js +0 -1
  77. package/dist/cli/lib/i18n-dict.d.ts +2 -2
  78. package/dist/cli/lib/i18n-dict.js +2 -1
  79. package/dist/cli/lib/i18n.d.ts +0 -1
  80. package/dist/cli/lib/i18n.js +0 -1
  81. package/dist/cli/lib/parse-run-config/judge-models.d.ts +0 -1
  82. package/dist/cli/lib/parse-run-config/judge-models.js +0 -1
  83. package/dist/cli/lib/parse-run-config/samples-discovery.d.ts +0 -1
  84. package/dist/cli/lib/parse-run-config/samples-discovery.js +1 -4
  85. package/dist/cli/lib/parse-run-config/variant-resolution.d.ts +0 -1
  86. package/dist/cli/lib/parse-run-config/variant-resolution.js +34 -8
  87. package/dist/cli/lib/parse-run-config.d.ts +0 -1
  88. package/dist/cli/lib/parse-run-config.js +1 -2
  89. package/dist/cli/lib/progress.d.ts +0 -1
  90. package/dist/cli/lib/progress.js +0 -1
  91. package/dist/cli/lib/resolve-skill-input.d.ts +0 -1
  92. package/dist/cli/lib/resolve-skill-input.js +0 -1
  93. package/dist/cli/lib/run-tally.d.ts +0 -1
  94. package/dist/cli/lib/run-tally.js +1 -2
  95. package/dist/cli/lib/shared.d.ts +0 -1
  96. package/dist/cli/lib/shared.js +1 -2
  97. package/dist/cli/lib/update-check.d.ts +0 -1
  98. package/dist/cli/lib/update-check.js +0 -1
  99. package/dist/cli/lib/update-fetch-worker.d.ts +0 -1
  100. package/dist/cli/lib/update-fetch-worker.js +0 -1
  101. package/dist/cli/oclif/base-command.d.ts +0 -1
  102. package/dist/cli/oclif/base-command.js +0 -1
  103. package/dist/cli/oclif/help.d.ts +0 -1
  104. package/dist/cli/oclif/help.js +0 -1
  105. package/dist/cli/oclif/i18n.d.ts +0 -1
  106. package/dist/cli/oclif/i18n.js +0 -1
  107. package/dist/cli/oclif/parsers.d.ts +0 -1
  108. package/dist/cli/oclif/parsers.js +0 -1
  109. package/dist/cli/oclif/projection.d.ts +0 -1
  110. package/dist/cli/oclif/projection.js +0 -1
  111. package/dist/cli/oclif/run.d.ts +0 -1
  112. package/dist/cli/oclif/run.js +0 -1
  113. package/dist/diagnosis/observe-mapper.d.ts +2 -3
  114. package/dist/diagnosis/observe-mapper.js +6 -7
  115. package/dist/diagnosis/observe-producer.d.ts +0 -1
  116. package/dist/diagnosis/observe-producer.js +3 -4
  117. package/dist/diagnosis/studio-projection.d.ts +0 -1
  118. package/dist/diagnosis/studio-projection.js +0 -1
  119. package/dist/diagnosis/types.d.ts +0 -1
  120. package/dist/diagnosis/types.js +0 -1
  121. package/dist/doctor/fixer.d.ts +0 -1
  122. package/dist/doctor/fixer.js +0 -1
  123. package/dist/doctor/health/builtin-dimensions.d.ts +0 -1
  124. package/dist/doctor/health/builtin-dimensions.js +0 -1
  125. package/dist/doctor/health/composer.d.ts +0 -1
  126. package/dist/doctor/health/composer.js +1 -2
  127. package/dist/doctor/health/dimension-registry.d.ts +0 -1
  128. package/dist/doctor/health/dimension-registry.js +0 -1
  129. package/dist/doctor/health/dimension-spec.d.ts +0 -1
  130. package/dist/doctor/health/dimension-spec.js +0 -1
  131. package/dist/doctor/health/load-custom-dimensions.d.ts +0 -1
  132. package/dist/doctor/health/load-custom-dimensions.js +0 -1
  133. package/dist/doctor/health/parser.d.ts +0 -1
  134. package/dist/doctor/health/parser.js +0 -1
  135. package/dist/doctor/health/prompt-builder.d.ts +0 -1
  136. package/dist/doctor/health/prompt-builder.js +0 -1
  137. package/dist/doctor/health/register.d.ts +0 -1
  138. package/dist/doctor/health/register.js +0 -1
  139. package/dist/doctor/index.d.ts +0 -1
  140. package/dist/doctor/index.js +1 -2
  141. package/dist/doctor/messages.d.ts +0 -1
  142. package/dist/doctor/messages.js +1 -2
  143. package/dist/doctor/preflight.d.ts +0 -1
  144. package/dist/doctor/preflight.js +0 -1
  145. package/dist/doctor/renderer.d.ts +0 -1
  146. package/dist/doctor/renderer.js +0 -1
  147. package/dist/doctor/rules.d.ts +0 -1
  148. package/dist/doctor/rules.js +0 -1
  149. package/dist/eval-core/bootstrap.d.ts +0 -1
  150. package/dist/eval-core/bootstrap.js +0 -1
  151. package/dist/eval-core/cache.d.ts +0 -1
  152. package/dist/eval-core/cache.js +0 -1
  153. package/dist/eval-core/comparability.d.ts +0 -1
  154. package/dist/eval-core/comparability.js +3 -4
  155. package/dist/eval-core/dependency-checker.d.ts +0 -1
  156. package/dist/eval-core/dependency-checker.js +2 -2
  157. package/dist/eval-core/evaluation-execution.d.ts +0 -1
  158. package/dist/eval-core/evaluation-execution.js +0 -1
  159. package/dist/eval-core/evaluation-job.d.ts +0 -1
  160. package/dist/eval-core/evaluation-job.js +0 -1
  161. package/dist/eval-core/evaluation-reporting.d.ts +0 -1
  162. package/dist/eval-core/evaluation-reporting.js +5 -4
  163. package/dist/eval-core/execution-strategy.d.ts +0 -1
  164. package/dist/eval-core/execution-strategy.js +6 -3
  165. package/dist/eval-core/fact-checker.d.ts +0 -1
  166. package/dist/eval-core/fact-checker.js +0 -1
  167. package/dist/eval-core/layer-gates.d.ts +0 -1
  168. package/dist/eval-core/layer-gates.js +0 -1
  169. package/dist/eval-core/mocks-runtime.d.ts +0 -1
  170. package/dist/eval-core/mocks-runtime.js +0 -1
  171. package/dist/eval-core/schema.d.ts +0 -1
  172. package/dist/eval-core/schema.js +0 -1
  173. package/dist/eval-core/statistics.d.ts +0 -1
  174. package/dist/eval-core/statistics.js +0 -1
  175. package/dist/eval-core/task-planner.d.ts +0 -1
  176. package/dist/eval-core/task-planner.js +0 -1
  177. package/dist/eval-core/verdict.d.ts +0 -1
  178. package/dist/eval-core/verdict.js +0 -1
  179. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +18 -3
  180. package/dist/eval-workflows/batch-evaluation-workflow.js +31 -15
  181. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +0 -1
  182. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +0 -1
  183. package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +0 -1
  184. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +0 -1
  185. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +0 -1
  186. package/dist/eval-workflows/evaluation-pipeline/run-state.js +0 -1
  187. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +0 -1
  188. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +0 -1
  189. package/dist/eval-workflows/evaluation-pipeline.d.ts +0 -1
  190. package/dist/eval-workflows/evaluation-pipeline.js +0 -1
  191. package/dist/eval-workflows/evaluation-preparation.d.ts +1 -5
  192. package/dist/eval-workflows/evaluation-preparation.js +20 -44
  193. package/dist/eval-workflows/messages.d.ts +0 -1
  194. package/dist/eval-workflows/messages.js +0 -1
  195. package/dist/eval-workflows/run-evaluation.d.ts +3 -8
  196. package/dist/eval-workflows/run-evaluation.js +9 -19
  197. package/dist/executors/anthropic-api.d.ts +0 -1
  198. package/dist/executors/anthropic-api.js +1 -2
  199. package/dist/executors/claude-cli.d.ts +0 -1
  200. package/dist/executors/claude-cli.js +0 -1
  201. package/dist/executors/claude-sdk-trace.d.ts +0 -1
  202. package/dist/executors/claude-sdk-trace.js +0 -1
  203. package/dist/executors/claude-sdk.d.ts +0 -1
  204. package/dist/executors/claude-sdk.js +0 -1
  205. package/dist/executors/codex-cli-trace.d.ts +0 -1
  206. package/dist/executors/codex-cli-trace.js +0 -1
  207. package/dist/executors/codex-cli.d.ts +0 -1
  208. package/dist/executors/codex-cli.js +0 -1
  209. package/dist/executors/codex-sdk.d.ts +0 -1
  210. package/dist/executors/codex-sdk.js +0 -1
  211. package/dist/executors/gemini.d.ts +0 -1
  212. package/dist/executors/gemini.js +0 -1
  213. package/dist/executors/index.d.ts +0 -1
  214. package/dist/executors/index.js +0 -1
  215. package/dist/executors/openai-api.d.ts +0 -1
  216. package/dist/executors/openai-api.js +0 -1
  217. package/dist/executors/runtime-fingerprint.d.ts +0 -1
  218. package/dist/executors/runtime-fingerprint.js +2 -3
  219. package/dist/executors/script.d.ts +0 -1
  220. package/dist/executors/script.js +0 -1
  221. package/dist/executors/shared.d.ts +0 -2
  222. package/dist/executors/shared.js +0 -1
  223. package/dist/grading/assertions.d.ts +0 -1
  224. package/dist/grading/assertions.js +0 -1
  225. package/dist/grading/debias-validate.d.ts +0 -1
  226. package/dist/grading/debias-validate.js +0 -1
  227. package/dist/grading/diagnostic.d.ts +0 -1
  228. package/dist/grading/diagnostic.js +0 -1
  229. package/dist/grading/gold-cli.d.ts +0 -1
  230. package/dist/grading/gold-cli.js +0 -1
  231. package/dist/grading/gold-dataset.d.ts +0 -1
  232. package/dist/grading/gold-dataset.js +0 -1
  233. package/dist/grading/human-gold.d.ts +0 -1
  234. package/dist/grading/human-gold.js +0 -1
  235. package/dist/grading/index.d.ts +0 -1
  236. package/dist/grading/index.js +0 -1
  237. package/dist/grading/judge.d.ts +0 -1
  238. package/dist/grading/judge.js +0 -1
  239. package/dist/grading/layered-scores.d.ts +0 -1
  240. package/dist/grading/layered-scores.js +0 -1
  241. package/dist/inputs/eval-config.d.ts +0 -1
  242. package/dist/inputs/eval-config.js +2 -2
  243. package/dist/inputs/load-samples.d.ts +0 -1
  244. package/dist/inputs/load-samples.js +0 -1
  245. package/dist/inputs/mcp-resolver.d.ts +0 -1
  246. package/dist/inputs/mcp-resolver.js +0 -1
  247. package/dist/inputs/skill-loader.d.ts +81 -11
  248. package/dist/inputs/skill-loader.js +173 -52
  249. package/dist/inputs/source-resolver.d.ts +28 -0
  250. package/dist/inputs/source-resolver.js +125 -0
  251. package/dist/inputs/url-fetcher.d.ts +0 -1
  252. package/dist/inputs/url-fetcher.js +0 -1
  253. package/dist/managed/index.d.ts +5 -0
  254. package/dist/managed/index.js +5 -0
  255. package/dist/managed/store.d.ts +76 -0
  256. package/dist/managed/store.js +260 -0
  257. package/dist/observability/experience-frontmatter.d.ts +0 -1
  258. package/dist/observability/experience-frontmatter.js +0 -1
  259. package/dist/observability/experience.d.ts +1 -2
  260. package/dist/observability/experience.js +1 -2
  261. package/dist/observability/feedback-matchers.d.ts +0 -1
  262. package/dist/observability/feedback-matchers.js +0 -1
  263. package/dist/observability/feedback-projection.d.ts +0 -1
  264. package/dist/observability/feedback-projection.js +0 -1
  265. package/dist/observability/inbox-view-model.d.ts +0 -1
  266. package/dist/observability/inbox-view-model.js +0 -1
  267. package/dist/observability/inbox.d.ts +0 -1
  268. package/dist/observability/inbox.js +2 -3
  269. package/dist/observability/problem-patterns.d.ts +0 -1
  270. package/dist/observability/problem-patterns.js +0 -1
  271. package/dist/observability/resolved-review.d.ts +0 -1
  272. package/dist/observability/resolved-review.js +4 -5
  273. package/dist/observability/review-state.d.ts +0 -1
  274. package/dist/observability/review-state.js +3 -4
  275. package/dist/observability/skill-chain-advisories.d.ts +0 -1
  276. package/dist/observability/skill-chain-advisories.js +0 -1
  277. package/dist/observability/skill-chain.d.ts +0 -1
  278. package/dist/observability/skill-chain.js +5 -6
  279. package/dist/observability/skill-health-analyzer.d.ts +0 -1
  280. package/dist/observability/skill-health-analyzer.js +0 -1
  281. package/dist/observability/soft-standards/constants.d.ts +0 -1
  282. package/dist/observability/soft-standards/constants.js +0 -1
  283. package/dist/observability/soft-standards/index.d.ts +0 -1
  284. package/dist/observability/soft-standards/index.js +0 -1
  285. package/dist/observability/soft-standards/llm-extractor.d.ts +0 -1
  286. package/dist/observability/soft-standards/llm-extractor.js +7 -8
  287. package/dist/observability/soft-standards/runtime-evaluator.d.ts +0 -1
  288. package/dist/observability/soft-standards/runtime-evaluator.js +2 -3
  289. package/dist/observability/soft-standards/skill-standards-store.d.ts +0 -1
  290. package/dist/observability/soft-standards/skill-standards-store.js +5 -6
  291. package/dist/observability/soft-standards/types.d.ts +10 -11
  292. package/dist/observability/soft-standards/types.js +0 -1
  293. package/dist/observability/text-signals.d.ts +0 -1
  294. package/dist/observability/text-signals.js +0 -1
  295. package/dist/observability/trace-adapter.d.ts +0 -1
  296. package/dist/observability/trace-adapter.js +0 -1
  297. package/dist/observability/trace-attribution.d.ts +0 -1
  298. package/dist/observability/trace-attribution.js +0 -1
  299. package/dist/observability/trace-segmenter.d.ts +0 -1
  300. package/dist/observability/trace-segmenter.js +0 -1
  301. package/dist/observability/trace-source.d.ts +0 -1
  302. package/dist/observability/trace-source.js +0 -1
  303. package/dist/renderer/html-renderer.d.ts +0 -1
  304. package/dist/renderer/html-renderer.js +4 -5
  305. package/dist/renderer/layout.d.ts +0 -1
  306. package/dist/renderer/layout.js +0 -1
  307. package/dist/renderer/observation-inbox/helpers.d.ts +0 -1
  308. package/dist/renderer/observation-inbox/helpers.js +0 -1
  309. package/dist/renderer/observation-inbox/styles.d.ts +0 -1
  310. package/dist/renderer/observation-inbox/styles.js +0 -1
  311. package/dist/renderer/observation-inbox-renderer.d.ts +0 -1
  312. package/dist/renderer/observation-inbox-renderer.js +13 -14
  313. package/dist/renderer/skill-detail-renderer.d.ts +0 -1
  314. package/dist/renderer/skill-detail-renderer.js +3 -4
  315. package/dist/renderer/skill-health-renderer.d.ts +0 -1
  316. package/dist/renderer/skill-health-renderer.js +0 -1
  317. package/dist/renderer/skill-list-renderer.d.ts +0 -1
  318. package/dist/renderer/skill-list-renderer.js +0 -1
  319. package/dist/renderer/summary.d.ts +0 -1
  320. package/dist/renderer/summary.js +1 -2
  321. package/dist/renderer/table.d.ts +0 -1
  322. package/dist/renderer/table.js +0 -1
  323. package/dist/renderer/test-view.d.ts +0 -1
  324. package/dist/renderer/test-view.js +0 -1
  325. package/dist/renderer/trends.d.ts +0 -1
  326. package/dist/renderer/trends.js +0 -1
  327. package/dist/server/job-store.d.ts +0 -1
  328. package/dist/server/job-store.js +0 -1
  329. package/dist/server/report-server.d.ts +0 -1
  330. package/dist/server/report-server.js +1 -2
  331. package/dist/server/report-store.d.ts +1 -2
  332. package/dist/server/report-store.js +6 -7
  333. package/dist/server/skill-index.d.ts +0 -1
  334. package/dist/server/skill-index.js +4 -5
  335. package/dist/server/skill-insights.d.ts +0 -1
  336. package/dist/server/skill-insights.js +8 -9
  337. package/dist/shared/hard-rules.d.ts +0 -1
  338. package/dist/shared/hard-rules.js +0 -1
  339. package/dist/shared/llm-prompts/index.d.ts +0 -1
  340. package/dist/shared/llm-prompts/index.js +0 -1
  341. package/dist/shared/llm-prompts/skill-health.d.ts +0 -1
  342. package/dist/shared/llm-prompts/skill-health.js +0 -1
  343. package/dist/shared/time.d.ts +0 -1
  344. package/dist/shared/time.js +0 -1
  345. package/dist/shared/tool-search.d.ts +0 -1
  346. package/dist/shared/tool-search.js +0 -1
  347. package/dist/types/dependencies.d.ts +0 -1
  348. package/dist/types/dependencies.js +0 -1
  349. package/dist/types/diagnosis.d.ts +0 -1
  350. package/dist/types/diagnosis.js +0 -1
  351. package/dist/types/doctor.d.ts +4 -5
  352. package/dist/types/doctor.js +2 -3
  353. package/dist/types/eval.d.ts +1 -1
  354. package/dist/types/eval.js +0 -1
  355. package/dist/types/executor.d.ts +1 -2
  356. package/dist/types/executor.js +0 -1
  357. package/dist/types/index.d.ts +1 -1
  358. package/dist/types/index.js +1 -1
  359. package/dist/types/judge.d.ts +0 -1
  360. package/dist/types/judge.js +0 -1
  361. package/dist/types/managed.d.ts +85 -0
  362. package/dist/types/managed.js +1 -0
  363. package/dist/types/observability.d.ts +5 -6
  364. package/dist/types/observability.js +0 -1
  365. package/dist/types/report.d.ts +2 -3
  366. package/dist/types/report.js +0 -1
  367. package/dist/types/shared.d.ts +0 -1
  368. package/dist/types/shared.js +0 -1
  369. package/dist/types/skill-index.d.ts +0 -1
  370. package/dist/types/skill-index.js +0 -1
  371. package/dist/types/storage.d.ts +0 -1
  372. package/dist/types/storage.js +0 -1
  373. package/dist/util/safe-slice.d.ts +0 -1
  374. package/dist/util/safe-slice.js +0 -1
  375. package/package.json +7 -2
  376. package/dist/analysis/coverage-analyzer.d.ts.map +0 -1
  377. package/dist/analysis/coverage-analyzer.js.map +0 -1
  378. package/dist/analysis/failure-clusterer.d.ts.map +0 -1
  379. package/dist/analysis/failure-clusterer.js.map +0 -1
  380. package/dist/analysis/gap-analyzer.d.ts.map +0 -1
  381. package/dist/analysis/gap-analyzer.js.map +0 -1
  382. package/dist/analysis/hedging-classifier.d.ts.map +0 -1
  383. package/dist/analysis/hedging-classifier.js.map +0 -1
  384. package/dist/analysis/report-diagnostics.d.ts.map +0 -1
  385. package/dist/analysis/report-diagnostics.js.map +0 -1
  386. package/dist/analysis/sample-diagnostics.d.ts.map +0 -1
  387. package/dist/analysis/sample-diagnostics.js.map +0 -1
  388. package/dist/analysis/saturation.d.ts.map +0 -1
  389. package/dist/analysis/saturation.js.map +0 -1
  390. package/dist/authoring/evolver.d.ts.map +0 -1
  391. package/dist/authoring/evolver.js.map +0 -1
  392. package/dist/authoring/generator.d.ts.map +0 -1
  393. package/dist/authoring/generator.js.map +0 -1
  394. package/dist/authoring/sample-fixer.d.ts.map +0 -1
  395. package/dist/authoring/sample-fixer.js.map +0 -1
  396. package/dist/cli/commands/doctor.d.ts.map +0 -1
  397. package/dist/cli/commands/doctor.js.map +0 -1
  398. package/dist/cli/commands/eval/gold/compare.d.ts.map +0 -1
  399. package/dist/cli/commands/eval/gold/compare.js.map +0 -1
  400. package/dist/cli/commands/eval/gold/index.d.ts.map +0 -1
  401. package/dist/cli/commands/eval/gold/index.js.map +0 -1
  402. package/dist/cli/commands/eval/gold/init.d.ts.map +0 -1
  403. package/dist/cli/commands/eval/gold/init.js.map +0 -1
  404. package/dist/cli/commands/eval/gold/validate.d.ts.map +0 -1
  405. package/dist/cli/commands/eval/gold/validate.js.map +0 -1
  406. package/dist/cli/commands/eval/index.d.ts.map +0 -1
  407. package/dist/cli/commands/eval/index.js.map +0 -1
  408. package/dist/cli/commands/evolve.d.ts.map +0 -1
  409. package/dist/cli/commands/evolve.js.map +0 -1
  410. package/dist/cli/commands/init.d.ts.map +0 -1
  411. package/dist/cli/commands/init.js.map +0 -1
  412. package/dist/cli/commands/observe/inbox.d.ts.map +0 -1
  413. package/dist/cli/commands/observe/inbox.js.map +0 -1
  414. package/dist/cli/commands/observe/index.d.ts.map +0 -1
  415. package/dist/cli/commands/observe/index.js.map +0 -1
  416. package/dist/cli/commands/observe/ingest.d.ts.map +0 -1
  417. package/dist/cli/commands/observe/ingest.js.map +0 -1
  418. package/dist/cli/commands/observe/show.d.ts.map +0 -1
  419. package/dist/cli/commands/observe/show.js.map +0 -1
  420. package/dist/cli/commands/sample.d.ts.map +0 -1
  421. package/dist/cli/commands/sample.js.map +0 -1
  422. package/dist/cli/commands/studio.d.ts.map +0 -1
  423. package/dist/cli/commands/studio.js.map +0 -1
  424. package/dist/cli/index.d.ts.map +0 -1
  425. package/dist/cli/index.js.map +0 -1
  426. package/dist/cli/lib/cli-exit.d.ts.map +0 -1
  427. package/dist/cli/lib/cli-exit.js.map +0 -1
  428. package/dist/cli/lib/cmd-flags.d.ts.map +0 -1
  429. package/dist/cli/lib/cmd-flags.js.map +0 -1
  430. package/dist/cli/lib/i18n-dict/common.d.ts.map +0 -1
  431. package/dist/cli/lib/i18n-dict/common.js.map +0 -1
  432. package/dist/cli/lib/i18n-dict/evolve.d.ts.map +0 -1
  433. package/dist/cli/lib/i18n-dict/evolve.js.map +0 -1
  434. package/dist/cli/lib/i18n-dict/gen.d.ts.map +0 -1
  435. package/dist/cli/lib/i18n-dict/gen.js.map +0 -1
  436. package/dist/cli/lib/i18n-dict/help.d.ts.map +0 -1
  437. package/dist/cli/lib/i18n-dict/help.js.map +0 -1
  438. package/dist/cli/lib/i18n-dict/init.d.ts.map +0 -1
  439. package/dist/cli/lib/i18n-dict/init.js.map +0 -1
  440. package/dist/cli/lib/i18n-dict/run.d.ts.map +0 -1
  441. package/dist/cli/lib/i18n-dict/run.js.map +0 -1
  442. package/dist/cli/lib/i18n-dict/types.d.ts.map +0 -1
  443. package/dist/cli/lib/i18n-dict/types.js.map +0 -1
  444. package/dist/cli/lib/i18n-dict.d.ts.map +0 -1
  445. package/dist/cli/lib/i18n-dict.js.map +0 -1
  446. package/dist/cli/lib/i18n.d.ts.map +0 -1
  447. package/dist/cli/lib/i18n.js.map +0 -1
  448. package/dist/cli/lib/parse-run-config/judge-models.d.ts.map +0 -1
  449. package/dist/cli/lib/parse-run-config/judge-models.js.map +0 -1
  450. package/dist/cli/lib/parse-run-config/samples-discovery.d.ts.map +0 -1
  451. package/dist/cli/lib/parse-run-config/samples-discovery.js.map +0 -1
  452. package/dist/cli/lib/parse-run-config/variant-resolution.d.ts.map +0 -1
  453. package/dist/cli/lib/parse-run-config/variant-resolution.js.map +0 -1
  454. package/dist/cli/lib/parse-run-config.d.ts.map +0 -1
  455. package/dist/cli/lib/parse-run-config.js.map +0 -1
  456. package/dist/cli/lib/progress.d.ts.map +0 -1
  457. package/dist/cli/lib/progress.js.map +0 -1
  458. package/dist/cli/lib/resolve-skill-input.d.ts.map +0 -1
  459. package/dist/cli/lib/resolve-skill-input.js.map +0 -1
  460. package/dist/cli/lib/run-tally.d.ts.map +0 -1
  461. package/dist/cli/lib/run-tally.js.map +0 -1
  462. package/dist/cli/lib/shared.d.ts.map +0 -1
  463. package/dist/cli/lib/shared.js.map +0 -1
  464. package/dist/cli/lib/update-check.d.ts.map +0 -1
  465. package/dist/cli/lib/update-check.js.map +0 -1
  466. package/dist/cli/lib/update-fetch-worker.d.ts.map +0 -1
  467. package/dist/cli/lib/update-fetch-worker.js.map +0 -1
  468. package/dist/cli/oclif/base-command.d.ts.map +0 -1
  469. package/dist/cli/oclif/base-command.js.map +0 -1
  470. package/dist/cli/oclif/help.d.ts.map +0 -1
  471. package/dist/cli/oclif/help.js.map +0 -1
  472. package/dist/cli/oclif/i18n.d.ts.map +0 -1
  473. package/dist/cli/oclif/i18n.js.map +0 -1
  474. package/dist/cli/oclif/parsers.d.ts.map +0 -1
  475. package/dist/cli/oclif/parsers.js.map +0 -1
  476. package/dist/cli/oclif/projection.d.ts.map +0 -1
  477. package/dist/cli/oclif/projection.js.map +0 -1
  478. package/dist/cli/oclif/run.d.ts.map +0 -1
  479. package/dist/cli/oclif/run.js.map +0 -1
  480. package/dist/diagnosis/observe-mapper.d.ts.map +0 -1
  481. package/dist/diagnosis/observe-mapper.js.map +0 -1
  482. package/dist/diagnosis/observe-producer.d.ts.map +0 -1
  483. package/dist/diagnosis/observe-producer.js.map +0 -1
  484. package/dist/diagnosis/studio-projection.d.ts.map +0 -1
  485. package/dist/diagnosis/studio-projection.js.map +0 -1
  486. package/dist/diagnosis/types.d.ts.map +0 -1
  487. package/dist/diagnosis/types.js.map +0 -1
  488. package/dist/doctor/fixer.d.ts.map +0 -1
  489. package/dist/doctor/fixer.js.map +0 -1
  490. package/dist/doctor/health/builtin-dimensions.d.ts.map +0 -1
  491. package/dist/doctor/health/builtin-dimensions.js.map +0 -1
  492. package/dist/doctor/health/composer.d.ts.map +0 -1
  493. package/dist/doctor/health/composer.js.map +0 -1
  494. package/dist/doctor/health/dimension-registry.d.ts.map +0 -1
  495. package/dist/doctor/health/dimension-registry.js.map +0 -1
  496. package/dist/doctor/health/dimension-spec.d.ts.map +0 -1
  497. package/dist/doctor/health/dimension-spec.js.map +0 -1
  498. package/dist/doctor/health/load-custom-dimensions.d.ts.map +0 -1
  499. package/dist/doctor/health/load-custom-dimensions.js.map +0 -1
  500. package/dist/doctor/health/parser.d.ts.map +0 -1
  501. package/dist/doctor/health/parser.js.map +0 -1
  502. package/dist/doctor/health/prompt-builder.d.ts.map +0 -1
  503. package/dist/doctor/health/prompt-builder.js.map +0 -1
  504. package/dist/doctor/health/register.d.ts.map +0 -1
  505. package/dist/doctor/health/register.js.map +0 -1
  506. package/dist/doctor/index.d.ts.map +0 -1
  507. package/dist/doctor/index.js.map +0 -1
  508. package/dist/doctor/messages.d.ts.map +0 -1
  509. package/dist/doctor/messages.js.map +0 -1
  510. package/dist/doctor/preflight.d.ts.map +0 -1
  511. package/dist/doctor/preflight.js.map +0 -1
  512. package/dist/doctor/renderer.d.ts.map +0 -1
  513. package/dist/doctor/renderer.js.map +0 -1
  514. package/dist/doctor/rules.d.ts.map +0 -1
  515. package/dist/doctor/rules.js.map +0 -1
  516. package/dist/eval-core/bootstrap.d.ts.map +0 -1
  517. package/dist/eval-core/bootstrap.js.map +0 -1
  518. package/dist/eval-core/cache.d.ts.map +0 -1
  519. package/dist/eval-core/cache.js.map +0 -1
  520. package/dist/eval-core/comparability.d.ts.map +0 -1
  521. package/dist/eval-core/comparability.js.map +0 -1
  522. package/dist/eval-core/dependency-checker.d.ts.map +0 -1
  523. package/dist/eval-core/dependency-checker.js.map +0 -1
  524. package/dist/eval-core/evaluation-execution.d.ts.map +0 -1
  525. package/dist/eval-core/evaluation-execution.js.map +0 -1
  526. package/dist/eval-core/evaluation-job.d.ts.map +0 -1
  527. package/dist/eval-core/evaluation-job.js.map +0 -1
  528. package/dist/eval-core/evaluation-reporting.d.ts.map +0 -1
  529. package/dist/eval-core/evaluation-reporting.js.map +0 -1
  530. package/dist/eval-core/execution-strategy.d.ts.map +0 -1
  531. package/dist/eval-core/execution-strategy.js.map +0 -1
  532. package/dist/eval-core/fact-checker.d.ts.map +0 -1
  533. package/dist/eval-core/fact-checker.js.map +0 -1
  534. package/dist/eval-core/layer-gates.d.ts.map +0 -1
  535. package/dist/eval-core/layer-gates.js.map +0 -1
  536. package/dist/eval-core/mocks-runtime.d.ts.map +0 -1
  537. package/dist/eval-core/mocks-runtime.js.map +0 -1
  538. package/dist/eval-core/schema.d.ts.map +0 -1
  539. package/dist/eval-core/schema.js.map +0 -1
  540. package/dist/eval-core/statistics.d.ts.map +0 -1
  541. package/dist/eval-core/statistics.js.map +0 -1
  542. package/dist/eval-core/task-planner.d.ts.map +0 -1
  543. package/dist/eval-core/task-planner.js.map +0 -1
  544. package/dist/eval-core/verdict.d.ts.map +0 -1
  545. package/dist/eval-core/verdict.js.map +0 -1
  546. package/dist/eval-workflows/batch-evaluation-workflow.d.ts.map +0 -1
  547. package/dist/eval-workflows/batch-evaluation-workflow.js.map +0 -1
  548. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts.map +0 -1
  549. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js.map +0 -1
  550. package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts.map +0 -1
  551. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js.map +0 -1
  552. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts.map +0 -1
  553. package/dist/eval-workflows/evaluation-pipeline/run-state.js.map +0 -1
  554. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts.map +0 -1
  555. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js.map +0 -1
  556. package/dist/eval-workflows/evaluation-pipeline.d.ts.map +0 -1
  557. package/dist/eval-workflows/evaluation-pipeline.js.map +0 -1
  558. package/dist/eval-workflows/evaluation-preparation.d.ts.map +0 -1
  559. package/dist/eval-workflows/evaluation-preparation.js.map +0 -1
  560. package/dist/eval-workflows/messages.d.ts.map +0 -1
  561. package/dist/eval-workflows/messages.js.map +0 -1
  562. package/dist/eval-workflows/run-evaluation.d.ts.map +0 -1
  563. package/dist/eval-workflows/run-evaluation.js.map +0 -1
  564. package/dist/executors/anthropic-api.d.ts.map +0 -1
  565. package/dist/executors/anthropic-api.js.map +0 -1
  566. package/dist/executors/claude-cli.d.ts.map +0 -1
  567. package/dist/executors/claude-cli.js.map +0 -1
  568. package/dist/executors/claude-sdk-trace.d.ts.map +0 -1
  569. package/dist/executors/claude-sdk-trace.js.map +0 -1
  570. package/dist/executors/claude-sdk.d.ts.map +0 -1
  571. package/dist/executors/claude-sdk.js.map +0 -1
  572. package/dist/executors/codex-cli-trace.d.ts.map +0 -1
  573. package/dist/executors/codex-cli-trace.js.map +0 -1
  574. package/dist/executors/codex-cli.d.ts.map +0 -1
  575. package/dist/executors/codex-cli.js.map +0 -1
  576. package/dist/executors/codex-sdk.d.ts.map +0 -1
  577. package/dist/executors/codex-sdk.js.map +0 -1
  578. package/dist/executors/gemini.d.ts.map +0 -1
  579. package/dist/executors/gemini.js.map +0 -1
  580. package/dist/executors/index.d.ts.map +0 -1
  581. package/dist/executors/index.js.map +0 -1
  582. package/dist/executors/openai-api.d.ts.map +0 -1
  583. package/dist/executors/openai-api.js.map +0 -1
  584. package/dist/executors/runtime-fingerprint.d.ts.map +0 -1
  585. package/dist/executors/runtime-fingerprint.js.map +0 -1
  586. package/dist/executors/script.d.ts.map +0 -1
  587. package/dist/executors/script.js.map +0 -1
  588. package/dist/executors/shared.d.ts.map +0 -1
  589. package/dist/executors/shared.js.map +0 -1
  590. package/dist/grading/assertions.d.ts.map +0 -1
  591. package/dist/grading/assertions.js.map +0 -1
  592. package/dist/grading/debias-validate.d.ts.map +0 -1
  593. package/dist/grading/debias-validate.js.map +0 -1
  594. package/dist/grading/diagnostic.d.ts.map +0 -1
  595. package/dist/grading/diagnostic.js.map +0 -1
  596. package/dist/grading/gold-cli.d.ts.map +0 -1
  597. package/dist/grading/gold-cli.js.map +0 -1
  598. package/dist/grading/gold-dataset.d.ts.map +0 -1
  599. package/dist/grading/gold-dataset.js.map +0 -1
  600. package/dist/grading/human-gold.d.ts.map +0 -1
  601. package/dist/grading/human-gold.js.map +0 -1
  602. package/dist/grading/index.d.ts.map +0 -1
  603. package/dist/grading/index.js.map +0 -1
  604. package/dist/grading/judge.d.ts.map +0 -1
  605. package/dist/grading/judge.js.map +0 -1
  606. package/dist/grading/layered-scores.d.ts.map +0 -1
  607. package/dist/grading/layered-scores.js.map +0 -1
  608. package/dist/inputs/eval-config.d.ts.map +0 -1
  609. package/dist/inputs/eval-config.js.map +0 -1
  610. package/dist/inputs/load-samples.d.ts.map +0 -1
  611. package/dist/inputs/load-samples.js.map +0 -1
  612. package/dist/inputs/mcp-resolver.d.ts.map +0 -1
  613. package/dist/inputs/mcp-resolver.js.map +0 -1
  614. package/dist/inputs/skill-loader.d.ts.map +0 -1
  615. package/dist/inputs/skill-loader.js.map +0 -1
  616. package/dist/inputs/url-fetcher.d.ts.map +0 -1
  617. package/dist/inputs/url-fetcher.js.map +0 -1
  618. package/dist/observability/experience-frontmatter.d.ts.map +0 -1
  619. package/dist/observability/experience-frontmatter.js.map +0 -1
  620. package/dist/observability/experience.d.ts.map +0 -1
  621. package/dist/observability/experience.js.map +0 -1
  622. package/dist/observability/feedback-matchers.d.ts.map +0 -1
  623. package/dist/observability/feedback-matchers.js.map +0 -1
  624. package/dist/observability/feedback-projection.d.ts.map +0 -1
  625. package/dist/observability/feedback-projection.js.map +0 -1
  626. package/dist/observability/inbox-view-model.d.ts.map +0 -1
  627. package/dist/observability/inbox-view-model.js.map +0 -1
  628. package/dist/observability/inbox.d.ts.map +0 -1
  629. package/dist/observability/inbox.js.map +0 -1
  630. package/dist/observability/problem-patterns.d.ts.map +0 -1
  631. package/dist/observability/problem-patterns.js.map +0 -1
  632. package/dist/observability/resolved-review.d.ts.map +0 -1
  633. package/dist/observability/resolved-review.js.map +0 -1
  634. package/dist/observability/review-state.d.ts.map +0 -1
  635. package/dist/observability/review-state.js.map +0 -1
  636. package/dist/observability/skill-chain-advisories.d.ts.map +0 -1
  637. package/dist/observability/skill-chain-advisories.js.map +0 -1
  638. package/dist/observability/skill-chain.d.ts.map +0 -1
  639. package/dist/observability/skill-chain.js.map +0 -1
  640. package/dist/observability/skill-health-analyzer.d.ts.map +0 -1
  641. package/dist/observability/skill-health-analyzer.js.map +0 -1
  642. package/dist/observability/soft-standards/constants.d.ts.map +0 -1
  643. package/dist/observability/soft-standards/constants.js.map +0 -1
  644. package/dist/observability/soft-standards/index.d.ts.map +0 -1
  645. package/dist/observability/soft-standards/index.js.map +0 -1
  646. package/dist/observability/soft-standards/llm-extractor.d.ts.map +0 -1
  647. package/dist/observability/soft-standards/llm-extractor.js.map +0 -1
  648. package/dist/observability/soft-standards/runtime-evaluator.d.ts.map +0 -1
  649. package/dist/observability/soft-standards/runtime-evaluator.js.map +0 -1
  650. package/dist/observability/soft-standards/skill-standards-store.d.ts.map +0 -1
  651. package/dist/observability/soft-standards/skill-standards-store.js.map +0 -1
  652. package/dist/observability/soft-standards/types.d.ts.map +0 -1
  653. package/dist/observability/soft-standards/types.js.map +0 -1
  654. package/dist/observability/text-signals.d.ts.map +0 -1
  655. package/dist/observability/text-signals.js.map +0 -1
  656. package/dist/observability/trace-adapter.d.ts.map +0 -1
  657. package/dist/observability/trace-adapter.js.map +0 -1
  658. package/dist/observability/trace-attribution.d.ts.map +0 -1
  659. package/dist/observability/trace-attribution.js.map +0 -1
  660. package/dist/observability/trace-segmenter.d.ts.map +0 -1
  661. package/dist/observability/trace-segmenter.js.map +0 -1
  662. package/dist/observability/trace-source.d.ts.map +0 -1
  663. package/dist/observability/trace-source.js.map +0 -1
  664. package/dist/renderer/html-renderer.d.ts.map +0 -1
  665. package/dist/renderer/html-renderer.js.map +0 -1
  666. package/dist/renderer/layout.d.ts.map +0 -1
  667. package/dist/renderer/layout.js.map +0 -1
  668. package/dist/renderer/observation-inbox/helpers.d.ts.map +0 -1
  669. package/dist/renderer/observation-inbox/helpers.js.map +0 -1
  670. package/dist/renderer/observation-inbox/styles.d.ts.map +0 -1
  671. package/dist/renderer/observation-inbox/styles.js.map +0 -1
  672. package/dist/renderer/observation-inbox-renderer.d.ts.map +0 -1
  673. package/dist/renderer/observation-inbox-renderer.js.map +0 -1
  674. package/dist/renderer/skill-detail-renderer.d.ts.map +0 -1
  675. package/dist/renderer/skill-detail-renderer.js.map +0 -1
  676. package/dist/renderer/skill-health-renderer.d.ts.map +0 -1
  677. package/dist/renderer/skill-health-renderer.js.map +0 -1
  678. package/dist/renderer/skill-list-renderer.d.ts.map +0 -1
  679. package/dist/renderer/skill-list-renderer.js.map +0 -1
  680. package/dist/renderer/summary.d.ts.map +0 -1
  681. package/dist/renderer/summary.js.map +0 -1
  682. package/dist/renderer/table.d.ts.map +0 -1
  683. package/dist/renderer/table.js.map +0 -1
  684. package/dist/renderer/test-view.d.ts.map +0 -1
  685. package/dist/renderer/test-view.js.map +0 -1
  686. package/dist/renderer/trends.d.ts.map +0 -1
  687. package/dist/renderer/trends.js.map +0 -1
  688. package/dist/server/job-store.d.ts.map +0 -1
  689. package/dist/server/job-store.js.map +0 -1
  690. package/dist/server/report-server.d.ts.map +0 -1
  691. package/dist/server/report-server.js.map +0 -1
  692. package/dist/server/report-store.d.ts.map +0 -1
  693. package/dist/server/report-store.js.map +0 -1
  694. package/dist/server/skill-index.d.ts.map +0 -1
  695. package/dist/server/skill-index.js.map +0 -1
  696. package/dist/server/skill-insights.d.ts.map +0 -1
  697. package/dist/server/skill-insights.js.map +0 -1
  698. package/dist/shared/hard-rules.d.ts.map +0 -1
  699. package/dist/shared/hard-rules.js.map +0 -1
  700. package/dist/shared/llm-prompts/index.d.ts.map +0 -1
  701. package/dist/shared/llm-prompts/index.js.map +0 -1
  702. package/dist/shared/llm-prompts/skill-health.d.ts.map +0 -1
  703. package/dist/shared/llm-prompts/skill-health.js.map +0 -1
  704. package/dist/shared/time.d.ts.map +0 -1
  705. package/dist/shared/time.js.map +0 -1
  706. package/dist/shared/tool-search.d.ts.map +0 -1
  707. package/dist/shared/tool-search.js.map +0 -1
  708. package/dist/types/dependencies.d.ts.map +0 -1
  709. package/dist/types/dependencies.js.map +0 -1
  710. package/dist/types/diagnosis.d.ts.map +0 -1
  711. package/dist/types/diagnosis.js.map +0 -1
  712. package/dist/types/doctor.d.ts.map +0 -1
  713. package/dist/types/doctor.js.map +0 -1
  714. package/dist/types/eval.d.ts.map +0 -1
  715. package/dist/types/eval.js.map +0 -1
  716. package/dist/types/executor.d.ts.map +0 -1
  717. package/dist/types/executor.js.map +0 -1
  718. package/dist/types/index.d.ts.map +0 -1
  719. package/dist/types/index.js.map +0 -1
  720. package/dist/types/judge.d.ts.map +0 -1
  721. package/dist/types/judge.js.map +0 -1
  722. package/dist/types/observability.d.ts.map +0 -1
  723. package/dist/types/observability.js.map +0 -1
  724. package/dist/types/report.d.ts.map +0 -1
  725. package/dist/types/report.js.map +0 -1
  726. package/dist/types/shared.d.ts.map +0 -1
  727. package/dist/types/shared.js.map +0 -1
  728. package/dist/types/skill-index.d.ts.map +0 -1
  729. package/dist/types/skill-index.js.map +0 -1
  730. package/dist/types/storage.d.ts.map +0 -1
  731. package/dist/types/storage.js.map +0 -1
  732. package/dist/util/safe-slice.d.ts.map +0 -1
  733. package/dist/util/safe-slice.js.map +0 -1
@@ -21,6 +21,19 @@ interface HoldoutSplit {
21
21
  trainIds: Set<string>;
22
22
  holdoutIds: Set<string>;
23
23
  }
24
+ /** A train / val / test partition. `val` drives the accept decision; `test` is
25
+ * locked — never seen during the loop, read once at the end for an unbiased
26
+ * generalization score. */
27
+ interface TrainValTestSplit {
28
+ trainIds: Set<string>;
29
+ valIds: Set<string>;
30
+ testIds: Set<string>;
31
+ }
32
+ /** Below this many decision (val) samples the bootstrap diff CI almost never
33
+ * excludes 0 for realistic effect sizes, so the significance gate would reject
34
+ * every candidate. Under that floor evolve degrades to the point-estimate accept
35
+ * and flags `gate.underpowered`. */
36
+ export declare const MIN_GATE_SAMPLES = 8;
24
37
  /**
25
38
  * Deterministically split sample ids into train / holdout by `ratio` (fraction
26
39
  * held out). Holdout members are picked at an even stride so the partition is
@@ -29,6 +42,44 @@ interface HoldoutSplit {
29
42
  * MIN_HOLDOUT_SUBSET — the caller then scores on the full set.
30
43
  */
31
44
  export declare function splitHoldout(sampleIds: string[], ratio: number): HoldoutSplit | null;
45
+ /**
46
+ * Deterministically split sample ids into train / val / test. `val` is carved
47
+ * first at an even stride; `test` is carved at an even stride over what remains,
48
+ * so the three sets are disjoint and stable across rounds/runs (no RNG). Returns
49
+ * null when either ratio ≤ 0 or any of the three sides would drop below
50
+ * MIN_HOLDOUT_SUBSET — the caller then degrades to a 2-way (or full-set) split.
51
+ */
52
+ export declare function splitTrainValTest(sampleIds: string[], valRatio: number, testRatio: number): TrainValTestSplit | null;
53
+ export interface AcceptDecision {
54
+ accepted: boolean;
55
+ /** Diff CI (candidate − best) when the gate ran; absent when it degraded. */
56
+ diffCI?: {
57
+ low: number;
58
+ high: number;
59
+ estimate: number;
60
+ significant: boolean;
61
+ };
62
+ /** True when the gate was requested but the decision set was below MIN_GATE_SAMPLES,
63
+ * so the decision degraded to the point-estimate comparison. */
64
+ underpowered: boolean;
65
+ }
66
+ /**
67
+ * The accept decision for one round. With the significance gate on and enough
68
+ * decision samples, a candidate is accepted only when it is **significantly** above
69
+ * the current best's fresh re-eval (`bootstrapDiffCI(...).significant && estimate > 0`)
70
+ * AND its decision score actually beats the recorded best (`pointCand > pointBest`).
71
+ * The second clause preserves evolve's monotonic invariant: `bestScore` never
72
+ * decreases. Without it, an unlucky (noise-low) re-eval of the current best could let
73
+ * a candidate that is significantly above that re-eval — yet still below the recorded
74
+ * best — win and overwrite the best downward. Off, or under-powered, the gate degrades
75
+ * to the legacy point-estimate comparison alone. Pure (modulo the seeded bootstrap) so
76
+ * the core behavior is unit-testable.
77
+ */
78
+ export declare function decideAccept(bestScores: number[], candScores: number[], pointBest: number, pointCand: number, opts: {
79
+ significanceGate: boolean;
80
+ alpha: number;
81
+ seed: number;
82
+ }): AcceptDecision;
32
83
  /**
33
84
  * A view of `report` whose results are restricted to `sampleIds`. Used to keep the
34
85
  * holdout split out of the sample-fixer: under an active holdout, only training-split
@@ -38,7 +89,22 @@ export declare function splitHoldout(sampleIds: string[], ratio: number): Holdou
38
89
  */
39
90
  export declare function restrictReportToSamples(report: Report, sampleIds: Set<string>): Report;
40
91
  export declare function allNonTripwireAssertionsPass(report: Report, variantKey: string): boolean;
41
- export declare function buildImprovementPrompt(skillContent: string, score: number, weakSamples: WeakSample[]): string;
92
+ interface EditDelta {
93
+ /** Symmetric line difference (added + removed unique lines) over original line count. */
94
+ ratio: number;
95
+ /** Absolute count of added + removed unique lines. */
96
+ changedLines: number;
97
+ /** Compact `+`/`-` summary of the changed lines, truncated. */
98
+ summary: string;
99
+ }
100
+ /**
101
+ * How a candidate differs from the current best, by trimmed non-empty line sets.
102
+ * `ratio` drives the edit budget; `summary` feeds the rejected-edit memory so the
103
+ * improver doesn't re-propose changes that already failed. Order-insensitive and
104
+ * O(n) over small skill files.
105
+ */
106
+ export declare function computeEditDelta(before: string, after: string, maxSummaryLines?: number): EditDelta;
107
+ export declare function buildImprovementPrompt(skillContent: string, score: number, weakSamples: WeakSample[], rejectedEdits?: string[]): string;
42
108
  /** @deprecated Use ProgressCallback from evaluation-core.ts */
43
109
  export type EvolveProgressInfo = Parameters<ProgressCallback>[0];
44
110
  export interface EvolveRoundProgressInfo {
@@ -54,6 +120,10 @@ export interface EvolveRoundProgressInfo {
54
120
  costReported?: boolean;
55
121
  error?: string;
56
122
  reused?: boolean;
123
+ /** When the significance gate ran: whether the candidate's gain was significant.
124
+ * False on a rejected round means "score rose but within noise" — lets the CLI
125
+ * explain an otherwise-confusing `(+0.0x) ✗ REJECT`. Undefined = gate didn't run. */
126
+ significant?: boolean;
57
127
  }
58
128
  interface EvolveOptions {
59
129
  skillPath: string;
@@ -88,20 +158,56 @@ interface EvolveOptions {
88
158
  * so the skill is never tuned to the samples that judge it. Too small a split
89
159
  * (either side < MIN_HOLDOUT_SUBSET) falls back to full-set scoring + a warning. */
90
160
  holdoutRatio?: number;
161
+ /** Statistically gate acceptance: a candidate is accepted only when its
162
+ * per-sample composite is **significantly** above the current best on the
163
+ * decision (val) set — `bootstrapDiffCI(...).significant && estimate > 0` —
164
+ * not merely numerically higher. Default true (rejecting improvements
165
+ * indistinguishable from judge noise is the point). Below MIN_GATE_SAMPLES
166
+ * decision samples the gate is underpowered and degrades to the point-estimate
167
+ * comparison + a warning. Set false to force the legacy point-estimate accept. */
168
+ significanceGate?: boolean;
169
+ /** Significance level for the accept gate's diff CI. Default 0.05 (95% CI). */
170
+ significanceAlpha?: number;
171
+ /** Fraction of samples locked away as a **test** set (0..1). Default 0 = off.
172
+ * Only honored alongside `holdoutRatio` > 0 (test needs a separate val set to
173
+ * decide on). The test split is never seen during the loop — not by weak-sample
174
+ * extraction, not by the accept gate — and is read exactly once at the end for an
175
+ * unbiased `generalizationScore`. Too small a 3-way split degrades to 2-way. */
176
+ testRatio?: number;
177
+ /** Max fraction of skill lines a single round may change before the candidate is
178
+ * rejected **without paying for evaluation**. Default 0.2 (matches the "≤20%"
179
+ * the improvement prompt already asks for — this enforces it). A small floor
180
+ * always permits a handful of lines so tiny skills aren't frozen. Set 0 to disable. */
181
+ editBudget?: number;
182
+ /** Feed rejected candidate edits back into the next round's improvement prompt
183
+ * ("these were tried and did not help — don't repeat them"). Default true. */
184
+ rejectMemory?: boolean;
91
185
  onProgress?: ProgressCallback | null;
92
186
  onRoundProgress?: ((progress: EvolveRoundProgressInfo) => void) | null;
93
187
  }
94
188
  interface TrajectoryEntry {
95
189
  round: number;
96
- /** Accept-decision score: holdout composite when holdout is active, else full-set. */
190
+ /** Accept-decision score: val composite when a holdout split is active, else full-set. */
97
191
  score: number;
98
192
  delta: number;
99
193
  accepted: boolean;
100
194
  costUSD: number;
101
- /** Present when holdout is active: the training-split composite (improvement signal). */
195
+ /** Present when a holdout split is active: the training-split composite (improvement signal). */
102
196
  trainScore?: number;
103
- /** Present when holdout is active: the holdout-split composite (== score). */
197
+ /** Present when a holdout split is active: the val-split composite (== score). */
104
198
  holdoutScore?: number;
199
+ /** Significance-gate diff CI (candidate − current best) on the decision set, when
200
+ * the gate was powered enough to run. `significant` 决定接受。 */
201
+ diffCI?: {
202
+ low: number;
203
+ high: number;
204
+ estimate: number;
205
+ significant: boolean;
206
+ };
207
+ /** Fraction of skill lines this candidate changed vs the current best. */
208
+ editRatio?: number;
209
+ /** True when the candidate was rejected by the edit budget before evaluation. */
210
+ rejectedPreEval?: boolean;
105
211
  }
106
212
  export interface EvolveResult {
107
213
  startScore: number;
@@ -127,6 +233,25 @@ export interface EvolveResult {
127
233
  holdoutCount: number;
128
234
  disabled?: boolean;
129
235
  };
236
+ /** Locked-test split summary. `disabled` is true when `--test-ratio` was requested
237
+ * but the 3-way split was too small, so evolve fell back to a 2-way holdout and
238
+ * produced no generalization score. */
239
+ test?: {
240
+ ratio: number;
241
+ count: number;
242
+ disabled?: boolean;
243
+ };
244
+ /** Unbiased composite of the best skill on the locked test set — the headline honest
245
+ * number. Present only when a 3-way split was active. The test set never influenced
246
+ * selection or weak-sample extraction, so this is an out-of-sample estimate. */
247
+ generalizationScore?: number;
248
+ /** Accept-gate summary. `underpowered` = the decision set was below MIN_GATE_SAMPLES
249
+ * at least once, so the gate degraded to the point-estimate comparison + warned. */
250
+ gate?: {
251
+ enabled: boolean;
252
+ alpha: number;
253
+ underpowered?: boolean;
254
+ };
130
255
  trajectory: TrajectoryEntry[];
131
256
  bestSkillPath: string;
132
257
  allVersions: string[];
@@ -138,6 +263,5 @@ export interface RoundReport {
138
263
  report: Report;
139
264
  }
140
265
  export declare function mergeEvolveReports(roundReports: RoundReport[], skillName: string, totalCostUSD: number, samples?: Sample[], skillPath?: string): Report;
141
- export declare function evolveSkill({ skillPath, samplesPath, rounds, target, stopOnAssertionsPass, autoFixSamples, sampleFixMaxAttempts, reuseLatestEval, model, judgeModels, improveModel, improveMode, executorName, concurrency, timeoutMs, skipConnectivity, effort, noDiagnostic, skipDoctor, holdoutRatio, onProgress, onRoundProgress, }: EvolveOptions): Promise<EvolveResult>;
266
+ export declare function evolveSkill({ skillPath, samplesPath, rounds, target, stopOnAssertionsPass, autoFixSamples, sampleFixMaxAttempts, reuseLatestEval, model, judgeModels, improveModel, improveMode, executorName, concurrency, timeoutMs, skipConnectivity, effort, noDiagnostic, skipDoctor, holdoutRatio, significanceGate, significanceAlpha, testRatio, editBudget, rejectMemory, onProgress, onRoundProgress, }: EvolveOptions): Promise<EvolveResult>;
142
267
  export {};
143
- //# sourceMappingURL=evolver.d.ts.map
@@ -7,6 +7,7 @@ import { createFileStore } from '../server/report-store.js';
7
7
  import { analyzeResults } from '../analysis/report-diagnostics.js';
8
8
  import { loadSamples } from '../inputs/load-samples.js';
9
9
  import { buildVariantSummary } from '../eval-core/schema.js';
10
+ import { bootstrapDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
10
11
  import { fixSamples } from './sample-fixer.js';
11
12
  const IMPROVE_SYSTEM_PROMPT = `你是一个 AI 提示词改进专家。你的任务是分析评测结果中的薄弱环节,针对性地改进 skill(系统提示词),使其在评测中获得更高的分数。
12
13
 
@@ -156,9 +157,25 @@ export function extractWeakSamples(report, variantKey, count = 5, sampleIdFilter
156
157
  .sort((a, b) => a.compositeScore - b.compositeScore)
157
158
  .slice(0, count);
158
159
  }
159
- /** Below this many samples on either side, a holdout split is too small to be
160
- * meaningful — evolve falls back to full-set scoring and warns. */
160
+ /** Below this many samples on any side, a split is too small to be meaningful —
161
+ * evolve falls back to full-set scoring and warns. */
161
162
  const MIN_HOLDOUT_SUBSET = 3;
163
+ /** Below this many decision (val) samples the bootstrap diff CI almost never
164
+ * excludes 0 for realistic effect sizes, so the significance gate would reject
165
+ * every candidate. Under that floor evolve degrades to the point-estimate accept
166
+ * and flags `gate.underpowered`. */
167
+ export const MIN_GATE_SAMPLES = 8;
168
+ /** Pick `count` ids at an even stride across `ids` (deterministic, no RNG) so the
169
+ * picked subset is representative of the ordering and stable across rounds/runs. */
170
+ function pickByStride(ids, count) {
171
+ const picked = new Set();
172
+ if (count <= 0)
173
+ return picked;
174
+ const stride = ids.length / count;
175
+ for (let k = 0; k < count; k++)
176
+ picked.add(ids[Math.floor(k * stride)]);
177
+ return picked;
178
+ }
162
179
  /**
163
180
  * Deterministically split sample ids into train / holdout by `ratio` (fraction
164
181
  * held out). Holdout members are picked at an even stride so the partition is
@@ -173,14 +190,31 @@ export function splitHoldout(sampleIds, ratio) {
173
190
  const trainCount = sampleIds.length - holdoutCount;
174
191
  if (holdoutCount < MIN_HOLDOUT_SUBSET || trainCount < MIN_HOLDOUT_SUBSET)
175
192
  return null;
176
- const stride = sampleIds.length / holdoutCount;
177
- const holdoutIds = new Set();
178
- for (let k = 0; k < holdoutCount; k++) {
179
- holdoutIds.add(sampleIds[Math.floor(k * stride)]);
180
- }
193
+ const holdoutIds = pickByStride(sampleIds, holdoutCount);
181
194
  const trainIds = new Set(sampleIds.filter((id) => !holdoutIds.has(id)));
182
195
  return { trainIds, holdoutIds };
183
196
  }
197
+ /**
198
+ * Deterministically split sample ids into train / val / test. `val` is carved
199
+ * first at an even stride; `test` is carved at an even stride over what remains,
200
+ * so the three sets are disjoint and stable across rounds/runs (no RNG). Returns
201
+ * null when either ratio ≤ 0 or any of the three sides would drop below
202
+ * MIN_HOLDOUT_SUBSET — the caller then degrades to a 2-way (or full-set) split.
203
+ */
204
+ export function splitTrainValTest(sampleIds, valRatio, testRatio) {
205
+ if (!(valRatio > 0) || !(testRatio > 0) || sampleIds.length === 0)
206
+ return null;
207
+ const valCount = Math.round(sampleIds.length * valRatio);
208
+ const testCount = Math.round(sampleIds.length * testRatio);
209
+ const trainCount = sampleIds.length - valCount - testCount;
210
+ if (valCount < MIN_HOLDOUT_SUBSET || testCount < MIN_HOLDOUT_SUBSET || trainCount < MIN_HOLDOUT_SUBSET)
211
+ return null;
212
+ const valIds = pickByStride(sampleIds, valCount);
213
+ const remaining = sampleIds.filter((id) => !valIds.has(id));
214
+ const testIds = pickByStride(remaining, testCount);
215
+ const trainIds = new Set(sampleIds.filter((id) => !valIds.has(id) && !testIds.has(id)));
216
+ return { trainIds, valIds, testIds };
217
+ }
184
218
  /**
185
219
  * Mean composite over the subset of a report's results whose sample_id is in
186
220
  * `ids`, using the same aggregation as the full-run summary
@@ -200,6 +234,47 @@ function subsetCompositeScore(report, variantKey, ids) {
200
234
  return 0;
201
235
  return buildVariantSummary(entries).avgCompositeScore ?? 0;
202
236
  }
237
+ /**
238
+ * Per-sample composite scores over the subset of a report's results whose
239
+ * sample_id is in `ids`, in result order. Feeds `bootstrapDiffCI` for the
240
+ * significance accept gate — the array (not the mean) is what the bootstrap
241
+ * resamples. Entries without a numeric compositeScore are skipped.
242
+ */
243
+ function perSampleComposite(report, variantKey, ids) {
244
+ const scores = [];
245
+ for (const r of report.results) {
246
+ if (!ids.has(r.sample_id))
247
+ continue;
248
+ const v = r.variants[variantKey];
249
+ if (v && typeof v.compositeScore === 'number')
250
+ scores.push(v.compositeScore);
251
+ }
252
+ return scores;
253
+ }
254
+ /**
255
+ * The accept decision for one round. With the significance gate on and enough
256
+ * decision samples, a candidate is accepted only when it is **significantly** above
257
+ * the current best's fresh re-eval (`bootstrapDiffCI(...).significant && estimate > 0`)
258
+ * AND its decision score actually beats the recorded best (`pointCand > pointBest`).
259
+ * The second clause preserves evolve's monotonic invariant: `bestScore` never
260
+ * decreases. Without it, an unlucky (noise-low) re-eval of the current best could let
261
+ * a candidate that is significantly above that re-eval — yet still below the recorded
262
+ * best — win and overwrite the best downward. Off, or under-powered, the gate degrades
263
+ * to the legacy point-estimate comparison alone. Pure (modulo the seeded bootstrap) so
264
+ * the core behavior is unit-testable.
265
+ */
266
+ export function decideAccept(bestScores, candScores, pointBest, pointCand, opts) {
267
+ const powered = bestScores.length >= MIN_GATE_SAMPLES && candScores.length >= MIN_GATE_SAMPLES;
268
+ if (opts.significanceGate && powered) {
269
+ const diff = bootstrapDiffCI(bestScores, candScores, opts.alpha, DEFAULT_BOOTSTRAP_SAMPLES, opts.seed);
270
+ return {
271
+ accepted: diff.significant && diff.estimate > 0 && pointCand > pointBest,
272
+ diffCI: { low: diff.low, high: diff.high, estimate: diff.estimate, significant: diff.significant },
273
+ underpowered: false,
274
+ };
275
+ }
276
+ return { accepted: pointCand > pointBest, underpowered: opts.significanceGate && !powered };
277
+ }
203
278
  /**
204
279
  * A view of `report` whose results are restricted to `sampleIds`. Used to keep the
205
280
  * holdout split out of the sample-fixer: under an active holdout, only training-split
@@ -336,7 +411,36 @@ async function autoFixSamplesAfterSkillRound(opts) {
336
411
  }
337
412
  return { fixedCount: result.fixedCount, costUSD: result.costUSD };
338
413
  }
339
- export function buildImprovementPrompt(skillContent, score, weakSamples) {
414
+ /** Number of skill lines below which the edit budget never trips — so a tiny skill
415
+ * isn't frozen by a percentage threshold that a few lines already blow past. */
416
+ const EDIT_BUDGET_FLOOR_LINES = 10;
417
+ /**
418
+ * How a candidate differs from the current best, by trimmed non-empty line sets.
419
+ * `ratio` drives the edit budget; `summary` feeds the rejected-edit memory so the
420
+ * improver doesn't re-propose changes that already failed. Order-insensitive and
421
+ * O(n) over small skill files.
422
+ */
423
+ export function computeEditDelta(before, after, maxSummaryLines = 12) {
424
+ const beforeArr = before.split('\n').map((l) => l.trim()).filter(Boolean);
425
+ const afterArr = after.split('\n').map((l) => l.trim()).filter(Boolean);
426
+ const beforeSet = new Set(beforeArr);
427
+ const afterSet = new Set(afterArr);
428
+ const added = [...afterSet].filter((l) => !beforeSet.has(l));
429
+ const removed = [...beforeSet].filter((l) => !afterSet.has(l));
430
+ const changedLines = added.length + removed.length;
431
+ const ratio = changedLines / Math.max(beforeArr.length, 1);
432
+ const parts = [];
433
+ for (const l of added.slice(0, maxSummaryLines))
434
+ parts.push(`+ ${l}`);
435
+ if (added.length > maxSummaryLines)
436
+ parts.push(`+ …(其余 +${added.length - maxSummaryLines} 行)`);
437
+ for (const l of removed.slice(0, maxSummaryLines))
438
+ parts.push(`- ${l}`);
439
+ if (removed.length > maxSummaryLines)
440
+ parts.push(`- …(其余 -${removed.length - maxSummaryLines} 行)`);
441
+ return { ratio, changedLines, summary: parts.join('\n') || '(无文本差异)' };
442
+ }
443
+ export function buildImprovementPrompt(skillContent, score, weakSamples, rejectedEdits) {
340
444
  const weakDetails = weakSamples.map((s) => {
341
445
  const parts = [`### ${s.sample_id}(${s.compositeScore}/5.0)`];
342
446
  if (s.llmReason)
@@ -365,13 +469,16 @@ export function buildImprovementPrompt(skillContent, score, weakSamples) {
365
469
  }
366
470
  return parts.join('\n');
367
471
  }).join('\n\n');
472
+ const rejectedSection = rejectedEdits && rejectedEdits.length > 0
473
+ ? `\n\n## 已试过且未带来显著提升的改法(不要重复)\n\n${rejectedEdits.join('\n\n')}`
474
+ : '';
368
475
  return `## 当前 Skill(平均分: ${score.toFixed(2)}/5.0)
369
476
 
370
477
  ${skillContent}
371
478
 
372
479
  ## 低分用例分析
373
480
 
374
- ${weakDetails || '(无低分用例)'}`;
481
+ ${weakDetails || '(无低分用例)'}${rejectedSection}`;
375
482
  }
376
483
  function buildImprovementSuffix(mode, candidatePath) {
377
484
  if (mode === 'agent') {
@@ -434,7 +541,7 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
434
541
  }
435
542
  const runId = `evolve-${skillName}-${generateRunId([skillName]).split('-').slice(-2).join('-')}`;
436
543
  const report = {
437
- kind: 'evaluation',
544
+ reportKind: 'evaluation',
438
545
  id: runId,
439
546
  meta: {
440
547
  ...firstReport.meta,
@@ -458,7 +565,7 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
458
565
  report.analysis = analyzeResults(report, { samples });
459
566
  return report;
460
567
  }
461
- export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model = DEFAULT_MODEL, judgeModels, improveModel = DEFAULT_MODEL, improveMode = 'agent', executorName = 'claude', concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, holdoutRatio = 0, onProgress = null, onRoundProgress = null, }) {
568
+ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model = DEFAULT_MODEL, judgeModels, improveModel = DEFAULT_MODEL, improveMode = 'agent', executorName = 'claude', concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, holdoutRatio = 0, significanceGate = true, significanceAlpha = DEFAULT_BOOTSTRAP_ALPHA, testRatio = 0, editBudget = 0.2, rejectMemory = true, onProgress = null, onRoundProgress = null, }) {
462
569
  if (judgeModels && judgeModels.length > 1) {
463
570
  throw new Error('evolveSkill does not support multi-judge ensemble (received '
464
571
  + `${judgeModels.length} judges). Pass a single-judge array, e.g. `
@@ -478,28 +585,72 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
478
585
  if (!existsSync(absSamplesPath))
479
586
  throw new Error(`samples file not found: ${absSamplesPath}`);
480
587
  mkdirSync(evolveDir, { recursive: true });
481
- // Holdout split (opt-in). Computed once over the canonical sample order so it's
482
- // stable across rounds. When active, accept decisions use the holdout composite
483
- // and weak-sample extraction only sees the training split — the skill is never
484
- // tuned to the samples that decide whether it's accepted.
588
+ // Split (opt-in). Computed once over the canonical sample order so it's stable
589
+ // across rounds. `val` drives the accept decision; weak-sample extraction and the
590
+ // sample-fixer only ever see `train`; `test` is locked away — never seen during the
591
+ // loop — and read once at the end for an unbiased generalization score. With no
592
+ // holdout the decision runs on the full set (legacy). A too-small 3-way split
593
+ // degrades to 2-way, then to full-set.
485
594
  const allSampleIds = loadSamples(absSamplesPath).samples.map((s) => s.sample_id);
486
- const holdoutSplit = splitHoldout(allSampleIds, holdoutRatio);
595
+ const threeWay = (testRatio > 0 && holdoutRatio > 0) ? splitTrainValTest(allSampleIds, holdoutRatio, testRatio) : null;
596
+ const twoWay = (!threeWay && holdoutRatio > 0) ? splitHoldout(allSampleIds, holdoutRatio) : null;
597
+ const split = threeWay
598
+ ? { trainIds: threeWay.trainIds, valIds: threeWay.valIds, testIds: threeWay.testIds }
599
+ : twoWay
600
+ ? { trainIds: twoWay.trainIds, valIds: twoWay.holdoutIds, testIds: null }
601
+ : null;
487
602
  const holdoutInfo = holdoutRatio > 0
488
603
  ? {
489
604
  ratio: holdoutRatio,
490
- trainCount: holdoutSplit?.trainIds.size ?? allSampleIds.length,
491
- holdoutCount: holdoutSplit?.holdoutIds.size ?? 0,
492
- ...(holdoutSplit ? {} : { disabled: true }),
605
+ trainCount: split?.trainIds.size ?? allSampleIds.length,
606
+ holdoutCount: split?.valIds.size ?? 0,
607
+ ...(split ? {} : { disabled: true }),
493
608
  }
494
609
  : undefined;
495
- // Accept-decision score for a report's variant: holdout composite when the split
496
- // is active, otherwise the full-set composite (legacy behavior).
497
- const decisionScore = (report, key) => holdoutSplit
498
- ? subsetCompositeScore(report, key, holdoutSplit.holdoutIds)
610
+ // test 被请求(配了 --holdout-ratio)但 3-way 太小回退 → 标 disabled,别让用户
611
+ // 以为拿到了 locked-test 泛化分。
612
+ const testRequested = testRatio > 0 && holdoutRatio > 0;
613
+ const testInfo = threeWay
614
+ ? { ratio: testRatio, count: threeWay.testIds.size }
615
+ : testRequested ? { ratio: testRatio, count: 0, disabled: true } : undefined;
616
+ // Deterministic seed so the gate's CIs are reproducible across reruns (and
617
+ // assertable in tests). Derived from skill identity + sample count, parsed to a uint32.
618
+ const gateSeed = parseInt(hashString(`${skillName}:${allSampleIds.length}`).slice(0, 8), 16) >>> 0;
619
+ let gateUnderpowered = false;
620
+ // Accept-decision score for a report's variant: val composite when a split is
621
+ // active, otherwise the full-set composite (legacy behavior).
622
+ const decisionScore = (report, key) => split
623
+ ? subsetCompositeScore(report, key, split.valIds)
499
624
  : (report.summary[key]?.avgCompositeScore ?? 0);
500
- const trainScoreOf = (report, key) => holdoutSplit ? subsetCompositeScore(report, key, holdoutSplit.trainIds) : undefined;
625
+ const trainScoreOf = (report, key) => split ? subsetCompositeScore(report, key, split.trainIds) : undefined;
501
626
  // Per-round trajectory tail: train / holdout breakdown, only when split active.
502
- const splitScores = (report, key, decision) => holdoutSplit ? { trainScore: trainScoreOf(report, key), holdoutScore: decision } : {};
627
+ const splitScores = (report, key, decision) => split ? { trainScore: trainScoreOf(report, key), holdoutScore: decision } : {};
628
+ // Unbiased generalization: the best skill's composite on the locked test set,
629
+ // read once at the very end. Present only under a valid 3-way split; the test
630
+ // set never influenced selection or weak-sample extraction.
631
+ const buildGeneralization = () => {
632
+ if (!testInfo)
633
+ return {};
634
+ if (!threeWay || !split?.testIds)
635
+ return { test: testInfo }; // requested but degraded → disabled, no score
636
+ const best = roundReports.find((r) => r.round === bestRound)?.report;
637
+ if (!best)
638
+ return { test: testInfo };
639
+ const key = Object.keys(best.summary)[0];
640
+ return { test: testInfo, generalizationScore: Number(subsetCompositeScore(best, key, split.testIds).toFixed(4)) };
641
+ };
642
+ const gateInfo = () => ({ enabled: significanceGate, alpha: significanceAlpha, ...(gateUnderpowered ? { underpowered: true } : {}) });
643
+ // Rejected-edit memory (most recent K). Fed back into the next round's improvement
644
+ // prompt so the improver doesn't re-propose changes that already failed to help.
645
+ const rejectedEdits = [];
646
+ const REJECT_MEMORY_K = 3;
647
+ const rememberRejected = (round, summary, reason) => {
648
+ if (!rejectMemory)
649
+ return;
650
+ rejectedEdits.push(`【第 ${round} 轮被拒(${reason})】\n${summary}`);
651
+ if (rejectedEdits.length > REJECT_MEMORY_K)
652
+ rejectedEdits.shift();
653
+ };
503
654
  // Save original as r0
504
655
  let currentBest = readFileSync(absSkillPath, 'utf-8').trim();
505
656
  const r0Path = join(evolveDir, `${skillName}.r0.md`);
@@ -567,6 +718,8 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
567
718
  ...(reusedBaselineReportId && { reusedBaselineReportId }),
568
719
  ...(totalCostReported ? {} : { costReported: false }),
569
720
  ...(holdoutInfo ? { holdout: holdoutInfo } : {}),
721
+ ...buildGeneralization(),
722
+ gate: gateInfo(),
570
723
  trajectory,
571
724
  bestSkillPath: allVersions[bestRound],
572
725
  allVersions,
@@ -593,10 +746,10 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
593
746
  totalCostReported = false;
594
747
  }
595
748
  const lastVariantKey = Object.keys(lastReport.summary)[0];
596
- const weakSamples = extractWeakSamples(lastReport, lastVariantKey, 5, holdoutSplit?.trainIds);
749
+ const weakSamples = extractWeakSamples(lastReport, lastVariantKey, 5, split?.trainIds);
597
750
  // Generate improvement
598
751
  const candidatePath = join(evolveDir, `${skillName}.r${round}.md`);
599
- const basePrompt = buildImprovementPrompt(currentBest, bestScore, weakSamples);
752
+ const basePrompt = buildImprovementPrompt(currentBest, bestScore, weakSamples, rejectMemory ? rejectedEdits : undefined);
600
753
  const executor = createExecutor(executorName);
601
754
  let candidateContent;
602
755
  let improveCostUSD;
@@ -650,6 +803,24 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
650
803
  if (!improveCostReported)
651
804
  totalCostReported = false;
652
805
  allVersions.push(candidatePath);
806
+ // Edit budget: reject oversized rewrites BEFORE paying for evaluation. The
807
+ // improvement prompt already asks for ≤ editBudget of lines changed — this
808
+ // enforces it. A small floor still lets tiny skills change a handful of lines.
809
+ const editDelta = computeEditDelta(currentBest, candidateContent);
810
+ if (editBudget > 0 && editDelta.ratio > editBudget && editDelta.changedLines > EDIT_BUDGET_FLOOR_LINES) {
811
+ totalCostUSD += improveCostUSD;
812
+ consecutiveRejects++;
813
+ const reason = `改动过大 ${(editDelta.ratio * 100).toFixed(0)}%(预算 ${(editBudget * 100).toFixed(0)}%),评测前判拒`;
814
+ rememberRejected(round, editDelta.summary, reason);
815
+ trajectory.push({ round, score: bestScore, delta: 0, accepted: false, costUSD: improveCostUSD, editRatio: Number(editDelta.ratio.toFixed(4)), rejectedPreEval: true });
816
+ if (onRoundProgress)
817
+ onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: bestScore, delta: 0, accepted: false, costUSD: improveCostUSD, costReported: improveCostReported });
818
+ if (consecutiveRejects >= 2) {
819
+ stopReason = 'consecutive-rejects';
820
+ break;
821
+ }
822
+ continue;
823
+ }
653
824
  let preEvalSampleFixCost = 0;
654
825
  if (autoFixSamples) {
655
826
  const sampleFix = await autoFixSamplesAfterSkillRound({
@@ -657,7 +828,7 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
657
828
  skillContent: candidateContent,
658
829
  // Under an active holdout, the sample-fixer may only see training-split samples —
659
830
  // never the holdout samples that drive the accept decision (leak guard).
660
- report: holdoutSplit ? restrictReportToSamples(lastReport, holdoutSplit.trainIds) : lastReport,
831
+ report: split ? restrictReportToSamples(lastReport, split.trainIds) : lastReport,
661
832
  treatmentKey: lastVariantKey,
662
833
  executorName,
663
834
  model: improveModel,
@@ -681,7 +852,22 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
681
852
  if (!roundCostReported)
682
853
  totalCostReported = false;
683
854
  totalCostUSD += improveCostUSD + candidateReport.meta.totalCostUSD;
684
- const accepted = candidateScore > bestScore;
855
+ // Significance accept gate: accept only when the candidate is *significantly*
856
+ // above the current best on the decision (val) set, not merely numerically higher
857
+ // — rejecting gains indistinguishable from judge noise. `lastReport` is the current
858
+ // best's fresh eval and `candidateReport` the candidate's, over the same samples;
859
+ // bootstrapDiffCI resamples the two arrays independently (conservative — not a paired
860
+ // bootstrap). Under-powered decision sets degrade to the legacy point-estimate accept
861
+ // (note: that path compares the prior-round best scalar, not this fresh re-eval) and
862
+ // flag `gate.underpowered`.
863
+ const valIds = split ? split.valIds : new Set(allSampleIds);
864
+ const bestScores = perSampleComposite(lastReport, lastVariantKey, valIds);
865
+ const candScores = perSampleComposite(candidateReport, candidateVariantKey, valIds);
866
+ const decision = decideAccept(bestScores, candScores, bestScore, candidateScore, { significanceGate, alpha: significanceAlpha, seed: gateSeed });
867
+ const accepted = decision.accepted;
868
+ const diffCI = decision.diffCI;
869
+ if (decision.underpowered)
870
+ gateUnderpowered = true;
685
871
  if (accepted) {
686
872
  currentBest = candidateContent;
687
873
  bestScore = candidateScore;
@@ -690,13 +876,15 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
690
876
  }
691
877
  else {
692
878
  consecutiveRejects++;
879
+ const reason = diffCI ? (diffCI.estimate > 0 ? '提升不显著' : '方向为负') : '未超过当前最优';
880
+ rememberRejected(round, editDelta.summary, reason);
693
881
  }
694
882
  if (accepted)
695
883
  roundReports.push({ round, accepted, report: candidateReport });
696
884
  const roundDelta = candidateScore - trajectory[trajectory.length - 1].score;
697
- trajectory.push({ round, score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, ...splitScores(candidateReport, candidateVariantKey, candidateScore) });
885
+ trajectory.push({ round, score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, ...splitScores(candidateReport, candidateVariantKey, candidateScore), ...(diffCI ? { diffCI } : {}), editRatio: Number(editDelta.ratio.toFixed(4)) });
698
886
  if (onRoundProgress)
699
- onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, costReported: roundCostReported });
887
+ onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, costReported: roundCostReported, ...(diffCI ? { significant: diffCI.significant } : {}) });
700
888
  // Early stop
701
889
  if (stopOnAssertionsPass && accepted && allNonTripwireAssertionsPass(candidateReport, candidateVariantKey)) {
702
890
  stopReason = 'assertions-pass';
@@ -735,6 +923,8 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
735
923
  ...(reusedBaselineReportId && { reusedBaselineReportId }),
736
924
  ...(totalCostReported ? {} : { costReported: false }),
737
925
  ...(holdoutInfo ? { holdout: holdoutInfo } : {}),
926
+ ...buildGeneralization(),
927
+ gate: gateInfo(),
738
928
  trajectory,
739
929
  bestSkillPath: allVersions[bestRound],
740
930
  allVersions,
@@ -761,4 +951,3 @@ async function evaluate(skillFilePath, { samplesPath, skillDir, model, judgeMode
761
951
  });
762
952
  return report;
763
953
  }
764
- //# sourceMappingURL=evolver.js.map
@@ -63,4 +63,3 @@ export declare function sanitizeGeneratedSamples(samples: Sample[], opts?: {
63
63
  stripped: string[];
64
64
  };
65
65
  export {};
66
- //# sourceMappingURL=generator.d.ts.map
@@ -830,4 +830,3 @@ export function sanitizeGeneratedSamples(samples, opts = {}) {
830
830
  }
831
831
  return { stripped };
832
832
  }
833
- //# sourceMappingURL=generator.js.map
@@ -48,4 +48,3 @@ export interface FixSamplesResult {
48
48
  }>;
49
49
  }
50
50
  export declare function fixSamples(options: FixSamplesOptions): Promise<FixSamplesResult>;
51
- //# sourceMappingURL=sample-fixer.d.ts.map
@@ -210,4 +210,3 @@ ${sampleSections}
210
210
  };
211
211
  }
212
212
  }
213
- //# sourceMappingURL=sample-fixer.js.map
@@ -25,4 +25,3 @@ export default class Doctor extends BaseCommand {
25
25
  run(): Promise<void>;
26
26
  }
27
27
  export declare function pruneDoctorHistory(dir: string, skillName: string, maxKeep: number): void;
28
- //# sourceMappingURL=doctor.d.ts.map
@@ -117,7 +117,7 @@ export default class Doctor extends BaseCommand {
117
117
  }),
118
118
  samples: Flags.string({
119
119
  description: bilingual({
120
- zh: '样本文件路径(.json/.yaml)。不传则按 target / cwd 顺序自动发现。',
120
+ zh: '用例文件路径(.json/.yaml)。不传则按 target / cwd 顺序自动发现。',
121
121
  en: 'Samples file path (.json/.yaml). Auto-detects from target / cwd if omitted.',
122
122
  }),
123
123
  }),
@@ -300,7 +300,7 @@ export function pruneDoctorHistory(dir, skillName, maxKeep) {
300
300
  continue;
301
301
  try {
302
302
  const data = JSON.parse(readFileSync(join(dir, file), 'utf-8'));
303
- if (data?.kind !== 'doctor' || !Array.isArray(data.skills) || data.skills.length !== 1)
303
+ if (data?.reportKind !== 'doctor' || !Array.isArray(data.skills) || data.skills.length !== 1)
304
304
  continue;
305
305
  if (data.skills[0].skillName !== skillName)
306
306
  continue;
@@ -318,4 +318,3 @@ export function pruneDoctorHistory(dir, skillName, maxKeep) {
318
318
  catch { /* ignore */ }
319
319
  }
320
320
  }
321
- //# sourceMappingURL=doctor.js.map
@@ -14,4 +14,3 @@ export default class EvalGoldCompare extends BaseCommand {
14
14
  };
15
15
  run(): Promise<void>;
16
16
  }
17
- //# sourceMappingURL=compare.d.ts.map
@@ -88,4 +88,3 @@ export default class EvalGoldCompare extends BaseCommand {
88
88
  });
89
89
  }
90
90
  }
91
- //# sourceMappingURL=compare.js.map