oh-my-knowledge 0.28.0 → 0.30.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (281) hide show
  1. package/README.md +49 -3
  2. package/README.zh.md +49 -3
  3. package/dist/src/analysis/failure-clusterer.js +1 -1
  4. package/dist/src/analysis/failure-clusterer.js.map +1 -1
  5. package/dist/src/analysis/gap-analyzer.d.ts +8 -1
  6. package/dist/src/analysis/gap-analyzer.d.ts.map +1 -1
  7. package/dist/src/analysis/gap-analyzer.js +65 -33
  8. package/dist/src/analysis/gap-analyzer.js.map +1 -1
  9. package/dist/src/analysis/report-diagnostics.js +2 -0
  10. package/dist/src/analysis/report-diagnostics.js.map +1 -1
  11. package/dist/src/analysis/sample-diagnostics.js +14 -14
  12. package/dist/src/analysis/sample-diagnostics.js.map +1 -1
  13. package/dist/src/authoring/evolver.d.ts +7 -1
  14. package/dist/src/authoring/evolver.d.ts.map +1 -1
  15. package/dist/src/authoring/evolver.js +8 -5
  16. package/dist/src/authoring/evolver.js.map +1 -1
  17. package/dist/src/authoring/generator.d.ts +21 -23
  18. package/dist/src/authoring/generator.d.ts.map +1 -1
  19. package/dist/src/authoring/generator.js +621 -40
  20. package/dist/src/authoring/generator.js.map +1 -1
  21. package/dist/src/authoring/sample-fixer.d.ts +45 -0
  22. package/dist/src/authoring/sample-fixer.d.ts.map +1 -0
  23. package/dist/src/authoring/sample-fixer.js +186 -0
  24. package/dist/src/authoring/sample-fixer.js.map +1 -0
  25. package/dist/src/cli/commands/evolve.d.ts.map +1 -1
  26. package/dist/src/cli/commands/evolve.js +17 -0
  27. package/dist/src/cli/commands/evolve.js.map +1 -1
  28. package/dist/src/cli/commands/observe.d.ts.map +1 -1
  29. package/dist/src/cli/commands/observe.js +170 -0
  30. package/dist/src/cli/commands/observe.js.map +1 -1
  31. package/dist/src/cli/commands/registry.d.ts.map +1 -1
  32. package/dist/src/cli/commands/registry.js +9 -1
  33. package/dist/src/cli/commands/registry.js.map +1 -1
  34. package/dist/src/cli/commands/sample.d.ts +13 -0
  35. package/dist/src/cli/commands/sample.d.ts.map +1 -1
  36. package/dist/src/cli/commands/sample.js +307 -13
  37. package/dist/src/cli/commands/sample.js.map +1 -1
  38. package/dist/src/cli/commands/studio.d.ts.map +1 -1
  39. package/dist/src/cli/commands/studio.js +8 -0
  40. package/dist/src/cli/commands/studio.js.map +1 -1
  41. package/dist/src/cli/i18n-dict.d.ts +1 -1
  42. package/dist/src/cli/i18n-dict.d.ts.map +1 -1
  43. package/dist/src/cli/i18n-dict.js +189 -17
  44. package/dist/src/cli/i18n-dict.js.map +1 -1
  45. package/dist/src/cli/parse-run-config.d.ts +14 -2
  46. package/dist/src/cli/parse-run-config.d.ts.map +1 -1
  47. package/dist/src/cli/parse-run-config.js +62 -10
  48. package/dist/src/cli/parse-run-config.js.map +1 -1
  49. package/dist/src/doctor/health/builtin-dimensions.d.ts.map +1 -1
  50. package/dist/src/doctor/health/builtin-dimensions.js +20 -23
  51. package/dist/src/doctor/health/builtin-dimensions.js.map +1 -1
  52. package/dist/src/doctor/health/composer.d.ts.map +1 -1
  53. package/dist/src/doctor/health/composer.js +7 -3
  54. package/dist/src/doctor/health/composer.js.map +1 -1
  55. package/dist/src/doctor/health/dimension-spec.d.ts +1 -0
  56. package/dist/src/doctor/health/dimension-spec.d.ts.map +1 -1
  57. package/dist/src/doctor/health/parser.d.ts.map +1 -1
  58. package/dist/src/doctor/health/parser.js +1 -0
  59. package/dist/src/doctor/health/parser.js.map +1 -1
  60. package/dist/src/doctor/health/prompt-builder.js +25 -17
  61. package/dist/src/doctor/health/prompt-builder.js.map +1 -1
  62. package/dist/src/doctor/html-renderer.d.ts.map +1 -1
  63. package/dist/src/doctor/html-renderer.js +76 -66
  64. package/dist/src/doctor/html-renderer.js.map +1 -1
  65. package/dist/src/doctor/rules.d.ts.map +1 -1
  66. package/dist/src/doctor/rules.js +32 -0
  67. package/dist/src/doctor/rules.js.map +1 -1
  68. package/dist/src/eval-core/cache.d.ts +16 -2
  69. package/dist/src/eval-core/cache.d.ts.map +1 -1
  70. package/dist/src/eval-core/cache.js +86 -15
  71. package/dist/src/eval-core/cache.js.map +1 -1
  72. package/dist/src/eval-core/evaluation-execution.d.ts +11 -1
  73. package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
  74. package/dist/src/eval-core/evaluation-execution.js +78 -6
  75. package/dist/src/eval-core/evaluation-execution.js.map +1 -1
  76. package/dist/src/eval-core/evaluation-job.d.ts +2 -1
  77. package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
  78. package/dist/src/eval-core/evaluation-job.js +2 -1
  79. package/dist/src/eval-core/evaluation-job.js.map +1 -1
  80. package/dist/src/eval-core/evaluation-reporting.d.ts +7 -0
  81. package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
  82. package/dist/src/eval-core/evaluation-reporting.js +19 -2
  83. package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
  84. package/dist/src/eval-core/execution-strategy.d.ts +1 -1
  85. package/dist/src/eval-core/execution-strategy.d.ts.map +1 -1
  86. package/dist/src/eval-core/execution-strategy.js +12 -1
  87. package/dist/src/eval-core/execution-strategy.js.map +1 -1
  88. package/dist/src/eval-core/mock-hook.cjs +203 -0
  89. package/dist/src/eval-core/mocks-runtime.d.ts +96 -0
  90. package/dist/src/eval-core/mocks-runtime.d.ts.map +1 -0
  91. package/dist/src/eval-core/mocks-runtime.js +323 -0
  92. package/dist/src/eval-core/mocks-runtime.js.map +1 -0
  93. package/dist/src/eval-core/schema.d.ts.map +1 -1
  94. package/dist/src/eval-core/schema.js +18 -5
  95. package/dist/src/eval-core/schema.js.map +1 -1
  96. package/dist/src/eval-core/task-planner.d.ts +9 -1
  97. package/dist/src/eval-core/task-planner.d.ts.map +1 -1
  98. package/dist/src/eval-core/task-planner.js +40 -1
  99. package/dist/src/eval-core/task-planner.js.map +1 -1
  100. package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts +5 -1
  101. package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts.map +1 -1
  102. package/dist/src/eval-workflows/batch-evaluation-workflow.js +2 -1
  103. package/dist/src/eval-workflows/batch-evaluation-workflow.js.map +1 -1
  104. package/dist/src/eval-workflows/evaluation-pipeline.d.ts +22 -1
  105. package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
  106. package/dist/src/eval-workflows/evaluation-pipeline.js +39 -12
  107. package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
  108. package/dist/src/eval-workflows/evaluation-preparation.d.ts +5 -0
  109. package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
  110. package/dist/src/eval-workflows/evaluation-preparation.js +3 -1
  111. package/dist/src/eval-workflows/evaluation-preparation.js.map +1 -1
  112. package/dist/src/eval-workflows/run-evaluation.d.ts +12 -2
  113. package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
  114. package/dist/src/eval-workflows/run-evaluation.js +17 -5
  115. package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
  116. package/dist/src/executors/claude-cli.d.ts +1 -1
  117. package/dist/src/executors/claude-cli.d.ts.map +1 -1
  118. package/dist/src/executors/claude-cli.js +57 -6
  119. package/dist/src/executors/claude-cli.js.map +1 -1
  120. package/dist/src/executors/claude-sdk-trace.d.ts.map +1 -1
  121. package/dist/src/executors/claude-sdk-trace.js +10 -4
  122. package/dist/src/executors/claude-sdk-trace.js.map +1 -1
  123. package/dist/src/executors/claude-sdk.d.ts +1 -1
  124. package/dist/src/executors/claude-sdk.d.ts.map +1 -1
  125. package/dist/src/executors/claude-sdk.js +25 -2
  126. package/dist/src/executors/claude-sdk.js.map +1 -1
  127. package/dist/src/executors/shared.d.ts +1 -1
  128. package/dist/src/executors/shared.d.ts.map +1 -1
  129. package/dist/src/executors/shared.js +6 -1
  130. package/dist/src/executors/shared.js.map +1 -1
  131. package/dist/src/grading/assertions.d.ts +5 -0
  132. package/dist/src/grading/assertions.d.ts.map +1 -1
  133. package/dist/src/grading/assertions.js +27 -1
  134. package/dist/src/grading/assertions.js.map +1 -1
  135. package/dist/src/grading/debias-validate.js +4 -4
  136. package/dist/src/grading/debias-validate.js.map +1 -1
  137. package/dist/src/grading/diagnostic.d.ts +40 -0
  138. package/dist/src/grading/diagnostic.d.ts.map +1 -0
  139. package/dist/src/grading/diagnostic.js +450 -0
  140. package/dist/src/grading/diagnostic.js.map +1 -0
  141. package/dist/src/grading/gold-cli.js +2 -2
  142. package/dist/src/grading/gold-cli.js.map +1 -1
  143. package/dist/src/grading/index.d.ts +5 -0
  144. package/dist/src/grading/index.d.ts.map +1 -1
  145. package/dist/src/grading/judge.d.ts.map +1 -1
  146. package/dist/src/grading/judge.js +54 -2
  147. package/dist/src/grading/judge.js.map +1 -1
  148. package/dist/src/grading/layered-scores.d.ts.map +1 -1
  149. package/dist/src/grading/layered-scores.js +1 -0
  150. package/dist/src/grading/layered-scores.js.map +1 -1
  151. package/dist/src/inputs/eval-config.js +19 -0
  152. package/dist/src/inputs/eval-config.js.map +1 -1
  153. package/dist/src/inputs/load-samples.d.ts +22 -2
  154. package/dist/src/inputs/load-samples.d.ts.map +1 -1
  155. package/dist/src/inputs/load-samples.js +193 -22
  156. package/dist/src/inputs/load-samples.js.map +1 -1
  157. package/dist/src/inputs/skill-loader.d.ts.map +1 -1
  158. package/dist/src/inputs/skill-loader.js +11 -3
  159. package/dist/src/inputs/skill-loader.js.map +1 -1
  160. package/dist/src/observability/experience.d.ts +260 -0
  161. package/dist/src/observability/experience.d.ts.map +1 -0
  162. package/dist/src/observability/experience.js +1233 -0
  163. package/dist/src/observability/experience.js.map +1 -0
  164. package/dist/src/observability/inbox-view-model.d.ts +27 -0
  165. package/dist/src/observability/inbox-view-model.d.ts.map +1 -0
  166. package/dist/src/observability/inbox-view-model.js +135 -0
  167. package/dist/src/observability/inbox-view-model.js.map +1 -0
  168. package/dist/src/observability/inbox.d.ts +125 -0
  169. package/dist/src/observability/inbox.d.ts.map +1 -0
  170. package/dist/src/observability/inbox.js +741 -0
  171. package/dist/src/observability/inbox.js.map +1 -0
  172. package/dist/src/observability/problem-patterns.d.ts +51 -0
  173. package/dist/src/observability/problem-patterns.d.ts.map +1 -0
  174. package/dist/src/observability/problem-patterns.js +205 -0
  175. package/dist/src/observability/problem-patterns.js.map +1 -0
  176. package/dist/src/observability/review-state.d.ts +54 -0
  177. package/dist/src/observability/review-state.d.ts.map +1 -0
  178. package/dist/src/observability/review-state.js +131 -0
  179. package/dist/src/observability/review-state.js.map +1 -0
  180. package/dist/src/observability/skill-chain-advisories.d.ts +25 -0
  181. package/dist/src/observability/skill-chain-advisories.d.ts.map +1 -0
  182. package/dist/src/observability/skill-chain-advisories.js +69 -0
  183. package/dist/src/observability/skill-chain-advisories.js.map +1 -0
  184. package/dist/src/observability/skill-chain.d.ts +61 -0
  185. package/dist/src/observability/skill-chain.d.ts.map +1 -0
  186. package/dist/src/observability/skill-chain.js +233 -0
  187. package/dist/src/observability/skill-chain.js.map +1 -0
  188. package/dist/src/observability/text-signals.d.ts +7 -0
  189. package/dist/src/observability/text-signals.d.ts.map +1 -0
  190. package/dist/src/observability/text-signals.js +22 -0
  191. package/dist/src/observability/text-signals.js.map +1 -0
  192. package/dist/src/observability/trace-adapter.d.ts +12 -64
  193. package/dist/src/observability/trace-adapter.d.ts.map +1 -1
  194. package/dist/src/observability/trace-adapter.js +9 -355
  195. package/dist/src/observability/trace-adapter.js.map +1 -1
  196. package/dist/src/observability/trace-attribution.d.ts +56 -0
  197. package/dist/src/observability/trace-attribution.d.ts.map +1 -0
  198. package/dist/src/observability/trace-attribution.js +200 -0
  199. package/dist/src/observability/trace-attribution.js.map +1 -0
  200. package/dist/src/observability/trace-segmenter.d.ts +52 -0
  201. package/dist/src/observability/trace-segmenter.d.ts.map +1 -0
  202. package/dist/src/observability/trace-segmenter.js +320 -0
  203. package/dist/src/observability/trace-segmenter.js.map +1 -0
  204. package/dist/src/observability/trace-source.d.ts +99 -0
  205. package/dist/src/observability/trace-source.d.ts.map +1 -0
  206. package/dist/src/observability/trace-source.js +492 -0
  207. package/dist/src/observability/trace-source.js.map +1 -0
  208. package/dist/src/renderer/html-renderer.d.ts.map +1 -1
  209. package/dist/src/renderer/html-renderer.js +324 -218
  210. package/dist/src/renderer/html-renderer.js.map +1 -1
  211. package/dist/src/renderer/layout.d.ts.map +1 -1
  212. package/dist/src/renderer/layout.js +155 -71
  213. package/dist/src/renderer/layout.js.map +1 -1
  214. package/dist/src/renderer/observation-inbox-renderer.d.ts +4 -0
  215. package/dist/src/renderer/observation-inbox-renderer.d.ts.map +1 -0
  216. package/dist/src/renderer/observation-inbox-renderer.js +5376 -0
  217. package/dist/src/renderer/observation-inbox-renderer.js.map +1 -0
  218. package/dist/src/renderer/skill-detail-renderer.d.ts +15 -0
  219. package/dist/src/renderer/skill-detail-renderer.d.ts.map +1 -0
  220. package/dist/src/renderer/skill-detail-renderer.js +1234 -0
  221. package/dist/src/renderer/skill-detail-renderer.js.map +1 -0
  222. package/dist/src/renderer/skill-list-renderer.d.ts +4 -0
  223. package/dist/src/renderer/skill-list-renderer.d.ts.map +1 -0
  224. package/dist/src/renderer/skill-list-renderer.js +327 -0
  225. package/dist/src/renderer/skill-list-renderer.js.map +1 -0
  226. package/dist/src/renderer/summary.d.ts +17 -0
  227. package/dist/src/renderer/summary.d.ts.map +1 -1
  228. package/dist/src/renderer/summary.js +346 -74
  229. package/dist/src/renderer/summary.js.map +1 -1
  230. package/dist/src/renderer/table.d.ts.map +1 -1
  231. package/dist/src/renderer/table.js +213 -167
  232. package/dist/src/renderer/table.js.map +1 -1
  233. package/dist/src/renderer/test-view.d.ts +66 -0
  234. package/dist/src/renderer/test-view.d.ts.map +1 -0
  235. package/dist/src/renderer/test-view.js +905 -0
  236. package/dist/src/renderer/test-view.js.map +1 -0
  237. package/dist/src/renderer/trends.d.ts.map +1 -1
  238. package/dist/src/renderer/trends.js +3 -1
  239. package/dist/src/renderer/trends.js.map +1 -1
  240. package/dist/src/server/report-server.d.ts +6 -1
  241. package/dist/src/server/report-server.d.ts.map +1 -1
  242. package/dist/src/server/report-server.js +182 -6
  243. package/dist/src/server/report-server.js.map +1 -1
  244. package/dist/src/server/report-store.d.ts.map +1 -1
  245. package/dist/src/server/report-store.js +33 -1
  246. package/dist/src/server/report-store.js.map +1 -1
  247. package/dist/src/server/skill-index.d.ts +77 -0
  248. package/dist/src/server/skill-index.d.ts.map +1 -0
  249. package/dist/src/server/skill-index.js +331 -0
  250. package/dist/src/server/skill-index.js.map +1 -0
  251. package/dist/src/server/skill-insights.d.ts +92 -0
  252. package/dist/src/server/skill-insights.d.ts.map +1 -0
  253. package/dist/src/server/skill-insights.js +676 -0
  254. package/dist/src/server/skill-insights.js.map +1 -0
  255. package/dist/src/shared/hard-rules.d.ts +40 -0
  256. package/dist/src/shared/hard-rules.d.ts.map +1 -0
  257. package/dist/src/shared/hard-rules.js +200 -0
  258. package/dist/src/shared/hard-rules.js.map +1 -0
  259. package/dist/src/shared/time.d.ts +2 -0
  260. package/dist/src/shared/time.d.ts.map +1 -0
  261. package/dist/src/shared/time.js +10 -0
  262. package/dist/src/shared/time.js.map +1 -0
  263. package/dist/src/shared/tool-search.d.ts +9 -0
  264. package/dist/src/shared/tool-search.d.ts.map +1 -0
  265. package/dist/src/shared/tool-search.js +104 -0
  266. package/dist/src/shared/tool-search.js.map +1 -0
  267. package/dist/src/types/eval.d.ts +83 -0
  268. package/dist/src/types/eval.d.ts.map +1 -1
  269. package/dist/src/types/executor.d.ts +47 -0
  270. package/dist/src/types/executor.d.ts.map +1 -1
  271. package/dist/src/types/judge.d.ts +76 -0
  272. package/dist/src/types/judge.d.ts.map +1 -1
  273. package/dist/src/types/judge.js +3 -1
  274. package/dist/src/types/judge.js.map +1 -1
  275. package/dist/src/types/report.d.ts +63 -3
  276. package/dist/src/types/report.d.ts.map +1 -1
  277. package/dist/src/util/safe-slice.d.ts +23 -0
  278. package/dist/src/util/safe-slice.d.ts.map +1 -0
  279. package/dist/src/util/safe-slice.js +33 -0
  280. package/dist/src/util/safe-slice.js.map +1 -0
  281. package/package.json +5 -4
@@ -1,15 +1,274 @@
1
- import { createExecutor, DEFAULT_MODEL } from '../executors/index.js';
1
+ import { createExecutor } from '../executors/index.js';
2
+ /**
3
+ * Generator 默认模型 'opus' (跟 eval 默认对齐)。
4
+ * lean=true 路径会自动追加 `--effort low`,关掉 opus 默认的扩展思考,
5
+ * 所以 opus + lean + effort-low 在 generator 场景下速度仍然可控(单 skill ~30-60s)。
6
+ * 成本约 sonnet 的 5x,但 opus 在结构化指令遵循 / 长 prompt 一致性上更稳。
7
+ * 用户想要省钱时显式 `--model sonnet` 即可。
8
+ */
9
+ const GENERATOR_DEFAULT_MODEL = 'opus';
2
10
  const SYSTEM_PROMPT = `你是一个评测用例生成器。你的任务是根据用户提供的 skill(系统提示词)内容,生成高质量的评测用例。
3
11
 
12
+ 样本结构决策(必须先做):先扫一遍 skill 内容判断它属于哪一类,按对应配比和数量生成。
13
+ - **工作流型** — skill 含"典型工作流"/"端到端"/明确多步流程章节,或描述"用户做一件事要按顺序调多个工具"的领域(端上自动化 / CI 部署 / API 编排 / 多步业务查询等)。
14
+ → **建议 6-8 条**:约 70% 工作流样本 + 约 30% 诱错样本(tripwire,测反模式)。
15
+ - **原子型** — skill 主要是知识点 / 查询规则 / 单步动作合集,各能力间无明显先后依赖(代码评审 / SQL 优化 / 术语解释等)。
16
+ → **建议 4-6 条**:全部用原子样本;若 skill 反模式 / 安全规则丰富(如强制白名单 / 红线检查),可拉到 6-8 条加诱错样本。
17
+ - **混合型** — 同时有多步流程章节 + 独立反模式 / 规则。
18
+ → **建议 5-7 条**:约 60% 工作流 + 约 40% 原子样本(含诱错)。
19
+
20
+ 数量决策的优先级:
21
+ 1. 用户在 prompt 末尾若明确给定数量("生成 N 个评测用例"),**优先按 N**,再按类型分配配比。
22
+ 2. 用户若让你"自行判断合适数量",按上述类型对应范围自定。
23
+ 3. 数量 <= 4 时优先保证覆盖度而非配比,哪怕全用原子。
24
+
25
+ **工作流样本要点**:
26
+ - prompt 必须含编号步骤("1. xxx 2. xxx 3. xxx"),不要写成"帮我做完整个流程"这种开放式任务
27
+ - 一条样本覆盖 4-7 个原子能力,assertion 以 **mock_hit**(每个关键步骤一条)为骨干,不要靠 contains 兜底——mock_hit 能在中间步骤失败时仍然精确告诉评测系统"挂在第几步"
28
+ - 工作流之间应彼此**正交**(各覆盖不同能力组合),不要重复测同一组
29
+
30
+ **诱错样本(tripwire)要点**:
31
+ - 短 prompt(1-2 句),prompt 故意藏与 skill 矛盾的诱导("直接用 X 就行 / 不用检查 / 我已经知道是 Y..."),测 skill 文档里写明的反模式 / 边界 / "不要做 X"
32
+ - 必须含 **tools_not_called**(测"baseline 会犯的错") + 1 条 tool_input_contains(测正确做法)
33
+ - 用于装不进工作流的反应式知识(如"不要用 CGEvent / 必须用 PTY 模式 / 架构兼容性提示"等)
34
+ - **必须在 sample 顶层加 \`"tripwire": true\`** — 让 omk 诊断知道"LLM fail 是预期",不要建议改 skill 文档
35
+
36
+ 判断完后在内部规划好配比再开始生成。**不要**在输出 JSON 里说明判断过程或配比,直接按规划生成样本即可。
37
+
38
+ ---
39
+
4
40
  每个用例需包含以下字段:
5
41
  - sample_id: 唯一标识,格式为 s001, s002, ...
6
42
  - prompt: 用户会向使用此 skill 的 AI 提出的典型问题或指令
7
43
  - context: 可选,附加上下文信息(如代码片段、文档段落等),仅在需要时提供
8
- - rubric: 评分标准,描述一个好的回答应该具备什么特征(1-2 句话)
9
- - assertions: 2-3 个断言检查,可选类型:
10
- - { "type": "contains", "value": "English keyword or code token", "weight": 1 }
11
- - { "type": "not_contains", "value": "English phrase that should not appear", "weight": 0.5 }
12
- - { "type": "regex", "pattern": "English|code|token", "weight": 1 }
44
+ - rubric: **judge 评分的输入**,要写 3-5 个**可分辨好坏的判分维度**,不要写一句话总结。
45
+ omk 的 judge pipeline 拿 rubric 让 judge LLM 看完整 trace(toolCalls + 最终输出 +
46
+ 关键中间产物)后按 rubric 每个维度逐项打 1-5 分,取均值作为该 sample 的 judge 综合分。
47
+ rubric 写得越具体 / 越多维度,judge 给的分数区分度越高;写得空泛(如"应当正确完成
48
+ 任务")则 judge 倾向给所有 sample 都 3-4 分中位,verdict 失去信号。
49
+
50
+ **rubric 应当涵盖的维度类型**(选 3-5 个相关的):
51
+ 1. **流程顺序**:"应先 X 再 Y"、"遇到 X 失败应当 fallback 到 Y"、"不应跳过 Z 步"
52
+ 2. **关键决策**:"识别请求属于 A 类还是 B 类"、"对边界情况(空输入 / 已存在文件)
53
+ 应当如何处理"、"用户诱导跳过 X 时应当坚持原流程"
54
+ 3. **输出结构**:"最终回答应当包含 [字段名 / 段落标题] 这几个组成部分"、"应当
55
+ 给用户明确的 next-step 指引而非含糊"
56
+ 4. **错误处理**:"工具失败时应当如实报告 + 给出降级方案,不应虚构成功"、"对
57
+ 不支持的请求应当拒绝并说明原因"
58
+ 5. **范围边界**:"应当严格遵守 skill 描述的职责边界,不主动越界做 Y 操作"
59
+
60
+ 示例:
61
+ 弱 rubric(❌): "应当正确生成评审报告"(judge 看不出"正确"是什么,只能给个中位分)
62
+ 强 rubric(✅): "应当:(1) 第一步识别当前评审属于需求阶段还是编码阶段并据此
63
+ 选 checks/ 下对应的检查清单文件,(2) 用户说'不用 git push'时仍按 SKILL.md
64
+ 默认规则把结果留档到知识库(因为'不 push'不在'temp 模式'触发词列表里),
65
+ (3) 报告里不向用户透出红线检查的逐项细节,只给最终风险等级 + 留档链接"
66
+
67
+ **禁忌(时间敏感数据)**: 不要在 rubric 里硬编码具体日期 / 时间戳 / 工号 / IP /
68
+ 临时 token 等会随评测时刻变化的具体值。
69
+ - 错(❌): "应写入 temp/2026-05-07/technical/ 目录" — 跨日跑就过期
70
+ - 对(✅): "应写入 temp/<today>/technical/ 目录(today=评测当天日期)" — 用占位
71
+ 描述,对应 assertion 用 regex 抓**模式**(如 \`temp/\\d{4}-\\d{2}-\\d{2}/technical/\`)
72
+ 而不是精确字符串。
73
+ - 占位符约定: \`<today>\` / \`<now>\` / \`<current_user>\` / \`<random_id>\` 等用尖括号包,
74
+ 跟 judge 说"这是占位,实际值看 trace 即可"。
75
+ - assertions: **fact 层硬验证清单**,**总数 2-4 条 hard cap**(不许靠堆"测每一步参数"
76
+ 来涨数量)。omk 的评分体系是 layered scoring: **fact 层**(deterministic 字面/工具断言)
77
+ + **behavior 层**(代价指标如 turn 数 / 工具失败率) + **judge 层**(主观语义评分,从
78
+ sample.rubric 派生维度,judge LLM 看 trace 评 1-5)三层独立计分,verdict 是三层
79
+ 独立过 threshold(默认 3.5)。**fact 层的本职是测 deterministic 端点,不是测轨迹**。
80
+
81
+ **断言哲学(关键):fact 测结果+里程碑,过程质量交 judge**
82
+ ─────────────────────────────────────────────────────────────
83
+ fact 层断言**只测两类东西**:
84
+ A. **结果断言**(最终产物)
85
+ - 最终写入的文件路径/内容 → \`tool_input_contains "Write:11-knowledge-base/X.md"\`
86
+ - 关键中间产物的字段 → \`tool_output_contains "Read:<expected-token-in-mock>"\`
87
+ - 最终回答应当包含的不可替换字面 token(错误码 / SDK 名 / 路径片段)
88
+ - JSON schema 命中(返回值结构正确) → \`json_schema\` 或 regex 抓固定模式
89
+ B. **里程碑断言**(流程必经瓶颈)
90
+ - **只有 SKILL.md 明文强约束**("必须 git push"、"必须先读 checks/X.md")
91
+ 的步骤算"里程碑",这种 sample 通常 0-2 条即够 → \`mock_hit "Tool:N"\`
92
+ 或 \`tools_called: ["Bash"]\`
93
+ - 判别标准:你能在 SKILL.md 里 grep 到原话说"必须做 X"或"流程第 N 步要
94
+ 调 X 工具",才算里程碑。**你"觉得应该重要"** 的步骤不算 — 那是过程,
95
+ 归 judge 评。
96
+
97
+ **fact 层不测的**(转给 sample.rubric → judge):
98
+ - 中间步骤的具体命令/参数字面("git diff 用的是 --name-only 还是 --stat") —
99
+ 命令变体等价,字面匹配是 false-negative 噪音源
100
+ - 工具调用顺序("应该先 stash 再 pull 还是先 pull 再 stash") — 顺序质量是
101
+ judge 看完整 trace 才能判的语义判断
102
+ - 错误处理路径("API 失败时应当 retry 几次" / "应当 fallback 到 X") — 同理,
103
+ 是行为质量,judge 拿 rubric 维度评分
104
+ - "应当礼貌拒绝用户的诱导改代码请求" — 这是语义意图,rubric 维度,不是字面 token
105
+
106
+ **典型分布**(单 sample):
107
+ - 结果断言 1-2 条(最终产物 / 关键字段 / 错误码)
108
+ - 里程碑断言 0-2 条(SKILL.md 明写的必经步)
109
+ - tools_not_called 反模式断言 0-1 条(禁止接触某禁忌工具,如 tripwire sample)
110
+ - rubric 3-5 个判分维度(细致写明 judge 该看什么),由 sample.rubric 字段承载
111
+
112
+ *测量学背景:* 当前 omk verdict 三层独立 threshold(默认 3.5),fact 条目少之后单条
113
+ 权重大、单次评测方差大,**强烈建议** 评测时带 \`--repeat 2\` 或更大测稳定性(coefficient
114
+ of variation),并参考 bootstrap CI 而非点估计。这是 fact 层稀疏化的代价,换来的是
115
+ fact 信号干净(不被 trajectory 字面噪音污染)。
116
+
117
+ 各 fact 类型详解(下面这些都属于"结果"或"里程碑"范畴,不是"过程"):
118
+
119
+ 工具/流程类(强信号,首选):
120
+ - { "type": "tool_input_contains", "value": "Bash:tag-list", "weight": 1 }
121
+ ↑ 检查某 toolCall 的 input(JSON.stringify 后)包含子串。格式: "Tool:期望子串"。
122
+ 用于断言"LLM 调对了命令/参数"——这才是 skill 知识的真凭据。
123
+ **子串选词原则**: 选**语义关键词**(命令名 / 关键工具名 / 关键参数 / SDK 函数名),
124
+ 不要选**完整命令字符串 / 精确路径 / flag 完整形态**。因为 LLM 写法常有等价变体,
125
+ 精确字符串会让正确行为也判挂。
126
+ - 错(❌): \`tool_input_contains "Bash:grep '^temp/$' .gitignore"\` —
127
+ LLM 用 \`grep -q\` 或加 \`~/\` 前缀就挂(全是等价写法)
128
+ - 对(✅): \`tool_input_contains "Bash:grep"\` + \`tool_input_contains "Bash:.gitignore"\` —
129
+ 抓"用了 grep" + "操作的是 .gitignore" 这两件语义事
130
+ - 错(❌): \`tool_input_contains "Bash:git push origin master"\` — 分支名 / remote 名都有变体
131
+ - 对(✅): \`tool_input_contains "Bash:git push"\` — 只抓核心动作 "git push"
132
+ 路径类同理:用 \`temp/\` 而不是 \`/abs/path/to/temp/\`,用 \`.json\` 而不是完整文件名。
133
+ - { "type": "tool_output_contains", "value": "Read:DevAPI", "weight": 0.5 }
134
+ ↑ 检查某工具返回(被 mock 的内容)的子串,格式同上。验证"LLM 看到了关键中间产物"。
135
+ - { "type": "tools_called", "values": ["Bash", "Read"], "weight": 0.5 }
136
+ ↑ 必须调过这些工具。
137
+ - { "type": "tools_not_called", "values": ["searchWorkItem"], "weight": 0.5 }
138
+ ↑ 不得调用某工具(典型场景:不要走错的 MCP / 旧接口)。values 必须非空,
139
+ 否则 loader 直接拒;如果想表达"不要写到某路径",用 tool_input_not_contains。
140
+ - { "type": "tool_input_not_contains", "value": "Write:/tmp/", "weight": 0.5 }
141
+ ↑ **反向**版 tool_input_contains:某工具的输入参数**不应**包含某子串。
142
+ 典型场景:工作流不应踩到某路径 / 不该传某 flag / 临时文件不应进永久目录。
143
+ 和 not_contains 的区别 — 这条只看工具调用参数,**不看 LLM 最终文本**,
144
+ 所以 LLM 在总结里说"我没写到 X" 不会假阳性触发。
145
+ **格式硬约束**: tool_input_contains / tool_input_not_contains / tool_output_contains /
146
+ mock_hit 的 value **必须**是 "Tool:needle" 格式(冒号分隔,工具名 + 子串两侧均非空)。
147
+ 不要写成 \`"--force"\` / \`"lastTaskPatrol"\` 这种裸 needle — 没有工具上下文,
148
+ loader 会直接拒。要表达"任何 Bash 调用都不该含 --force":写 \`"Bash:--force"\`。
149
+ - { "type": "mock_hit", "value": "Bash:2", "weight": 1 }
150
+ ↑ 校验"驱动流程": sample.mocks 数组里第 N 条(1-based)是否被命中至少一次。
151
+ 例: mocks=[A,B,C](A=PROJECT 空 / B=WORKSPACE 命中 / C=search),
152
+ 用 mock_hit "Bash:2" 强制 LLM 必须走到第 2 步(WORKSPACE 兜底),否则失分。
153
+ threshold 字段可选,默认 >=1。
154
+ 文本类(**严格限定:只测不可替换字面量,不要测语义/论点**):
155
+
156
+ ⛔ **绝对禁止**(产了这种就是错误,样本会被拒绝):
157
+ contains / not_contains / contains_any / contains_all / regex 的 value/values
158
+ **不允许**出现以下任一情况:
159
+ (1) **含 CJK 中文字符**(留档 / 已修复 / 不阻塞 / 死循环 / 系分方案 等)
160
+ 理由:中文同义改写最厉害,"留档"/"归档"/"存档"/"记录",LLM 每次发挥都换说法。
161
+ (2) **含空格的短语**("not safe" / "git push origin master" 等)
162
+ 理由:多 token 短语本质是自然语言片段,LLM 句式重排就挂。
163
+ 注意:测"LLM 是否调对命令" 用 tool_input_contains,**不**走 contains。
164
+ (3) **含中文标点**(,。!?「」【】等)
165
+ 理由:含标点必是句子片段,不是 token。
166
+ (4) **长度 > 30 字符**
167
+ 理由:超过 30 字符基本不是单 token,是句子片段了。
168
+
169
+ ✅ **允许的 contains value 形态**(只有这一类):
170
+ 全 ASCII / 只含字母数字 + 下划线 / 连字符 / 点 / 斜杠,长度 3–30,看起来像代码 token:
171
+ - 错误码:"ECONNREFUSED" / "EAI_AGAIN" / "E404"
172
+ - SDK 函数名 / 类名:"skylark_doc_create" / "AsyncOperation"
173
+ - HTTP header 名:"x-trace-id" / "Content-Type"
174
+ - 命令 flag:"--force" / "-ff-only" / "--dry-run"
175
+ - 路径片段:"tasks/" / "/api/v2/" / ".gitignore"
176
+
177
+ 📋 **每条 contains 系列断言自检清单**(产 sample 前必走):
178
+ 1. value 含任何中文字符? → 改用 sample.rubric 表达,**不要**写 contains
179
+ 2. value 含空格的短语? → 同上,或考虑 tool_input_contains
180
+ 3. 表达的是"LLM 应该提到 X 概念" 类语义判断? → **必须**走 rubric → judge,
181
+ 即使 value 看起来像 token 也不行
182
+ 4. 只有当 value 是机器可识别的 ASCII 代码 token / 错误码 / flag 时,
183
+ contains 才合法
184
+
185
+ 生成 sample 时遇到诱惑想用 contains 测语义概念(如"应该说明不阻塞"、
186
+ "应该提供建议"、"应该留档") → **强制改写**:把这点加到 sample.rubric,
187
+ 让 judge 多维度评分;不要试图用 contains_any 列同义词糊弄过去。
188
+
189
+ 以上禁令是**硬性规则**,违反的 sample 会被人工审查拒绝并要求重写。
190
+ - { "type": "contains", "value": "code-token", "weight": 1 }
191
+ ↑ **唯一**用法:抓代码 token / 错误码 / SDK 函数名 / 不可替换字面量(必须是
192
+ ASCII + 长度 3–30 + 像 token 形态)。LLM 在这些字面上没有同义改写空间。
193
+ - { "type": "contains_any", "values": ["x","y","z"], "weight": 0.5 }
194
+ ↑ 多候选字面任一命中即过。仅在**少数有限的字面变体**场景用 — 如错误码组
195
+ ["ECONNREFUSED","ETIMEDOUT","EHOSTUNREACH"]。**不要**用 contains_any 列同义词
196
+ 来"测概念覆盖" — 同义词永远列不全,LLM 第 N+1 次发挥总能想出第 N+1 个写法。
197
+ - { "type": "contains_all", "values": ["X","Y"], "weight": 1 }
198
+ ↑ 必须同时包含全部字面 token(如某 API 响应应同时含两个具体字段名)。
199
+ - { "type": "not_contains", "value": "...", "weight": 0.5 }
200
+ ↑ 只查 LLM **最终文本**不应出现某固定字面 token。**不要**用它表达"不应踩到 X" —
201
+ LLM 在总结里复述"已避开 X" 会自触发假阳性。"工具调用层面不该走"→
202
+ tool_input_not_contains 或 tools_not_called。
203
+ - { "type": "regex", "pattern": "...", "weight": 1 }
204
+ ↑ 同 contains 限制:只用在固定格式字面量(如 SHA / UUID / 路径模板)。
205
+ - environment: 可选,对象。**评测环境的"已就绪"声明**,LLM 看到后跳过环境探测直接进工作流。
206
+ 字段:
207
+ - cli_available: string[],已在 PATH 上的 CLI(如 ["node", "git", "code-host"])
208
+ - files_available: string[],已存在的文件/脚本(如 ["~/.req-tool-api.json", "$SKILL_DIR/scripts/x.js"])
209
+ - notes: string,自由文本兜底(如"DevAPI 凭证有效,工号 testuser001")
210
+ 原则:
211
+ 凡是 skill 跑起来需要的环境(凭证文件 / 业务 CLI / 自带脚本 / API token 等),
212
+ 都写到这里,而不是在 mock 里 mock 它们的探测命令。这让 mock 只关注业务调用本身。
213
+ - mocksStrict: **必填且必须设为 true**(只要 sample 配了 mocks)。
214
+ 原因:mocksStrict=false 时,LLM 调到没匹配 mock 的命令会**透传到真 shell**,
215
+ 既可能真调外部接口产生副作用,也可能因二进制不存在(如 mcporter)报噪声错误污染评测信号。
216
+ 评测目的是在隔离环境下测 LLM 行为,不是测真接口可用性 — 总是 strict。
217
+ - mocks: 可选,数组。该 sample 跑评测时拦截工具调用 + 返回 stub。**避免真调外部接口/CLI/MCP/写状态**。
218
+ 生成原则:
219
+ 1. **mocks 覆盖范围 = 业务调用 + 工作流前置 / 校验步骤** —
220
+ (a) 业务调用(submit / create / push / search ...) 必 mock
221
+ (b) **工作流前置 / 校验步骤**(skill 强制要求的检查动作,如 \`ls -la\` 检查目录是否存在、
222
+ \`grep -q\` 检查 .gitignore、\`git status\` 看是否干净等)**也必须 mock**,因为它们
223
+ 会被 mocks-strict 拦截 — 这是 obsidian / 知识库整理 / 部署类 skill 大量挂在
224
+ "环境拦截"的根因。
225
+ **关键**:这些前置步骤的 assertion 通常是 \`tools_called: ["Bash"]\` 或
226
+ \`tool_input_contains "ls -la"\`,意味 LLM 必须真调这些命令。如果 mock 没盖,
227
+ LLM 行为完全正确还是会因为 mocks-strict 拦截而挂。
228
+ - 错(❌):rubric 要求"先 ls -la 检查目录",但 mocks 数组里没 \`{tool:"Bash",
229
+ match:{command_glob:"ls *"},return:{...}}\`,LLM 调 ls 就被拦,工作流断在第 0 步
230
+ - 对(✅):写一条宽 mock:\`{tool:"Bash", match:{command_glob:"ls *"},
231
+ return:{stdout:"<模拟目录列表>", exit:0}}\` — \`command_glob\` 用 \`*\` 兜底各种
232
+ ls 参数变体(\`ls\` / \`ls -la\` / \`ls -d\` / \`ls /xx\` 全命中)
233
+ (c) 单纯"已就绪"声明(凭证文件 / 业务 CLI 是否安装)还是走 \`environment\` 字段,
234
+ 不需要 LLM 真调命令检查 — environment 字段就是告诉 LLM "这些不用检查"。
235
+ 2. **mock 数据要"驱动流程"而非"提前给答案"** — 这是关键:
236
+ - 如果 skill 描述的工作流是多步的(A→B→C),mock 数据要**让最终答案只在最后一步出现**,
237
+ 前面的 mock 只能给出"推进到下一步必需的中间产物",不能直接揭示完整答案。
238
+ - 反例(❌):第 1 步 mock 直接返回完整答案 → LLM 觉得"够了"跳过后续步骤,
239
+ 评测拿不到"是否走完合规流程"的信号。
240
+ - 正例(✅):第 1 步 mock 返回空 / 局部 / 索引 ID → LLM 必须用这个中间产物去调下一步,
241
+ 一直走到最后一步才能拿到最终答案。
242
+ - 例:req-tool 查标签工作流(PROJECT tag-list → WORKSPACE tag-list → tag-search):
243
+ * 错的设计:PROJECT 直接返回 \`[{tagName:"Daily",count:99}]\` → LLM 跳过 WORKSPACE
244
+ * 对的设计:PROJECT 返回 \`[]\`,WORKSPACE 返回 \`[{tagId:"W001",tagName:"Daily"}]\`,
245
+ tag-search 用 \`W001\` 才返回最终工作项列表
246
+ 4. write 类调用(submit / create / push)mock 返回成功响应即可
247
+ 5. 不要 mock LLM 内部 think/text 行为,只 mock 外部副作用工具
248
+ mock 项 schema:
249
+ {
250
+ "tool": "Bash" | "Read" | "Edit" | "Write" | "WebFetch" | "Grep" | "Glob",
251
+ "match": {
252
+ "file_path_endswith": "<相对路径后缀,如 tasks/foo/state.json>", // 推荐用这条 (Read/Edit/Write)
253
+ "file_path": "<完整路径,~ 或绝对>", // 仅当能预测完整 path 时用,否则首选 _endswith
254
+ "url": "<exact url>" or "url_glob": "<glob>", // WebFetch
255
+ "command_glob": "<glob>", // Bash 拦 mcporter / cli
256
+ "input": { "<key>": "<value>" } // generic deep-equal subset
257
+ },
258
+ "return": "<string>" or { "stdout": "...", "exit": 0 },
259
+ "return_seq": [<r1>, <r2>] // optional 状态机:同 mock 多次命中按序返回
260
+ }
261
+ **file_path 匹配的关键陷阱**:
262
+ - claude-cli / claude-sdk 的 PreToolUse hook 拿到的 file_path 是 LLM 原话 — LLM 经常把
263
+ 相对路径写成绝对(尤其当 environment.notes 给了 cwd 提示),mock 用 file_path 严格相等
264
+ 会 miss 整条 sample。**默认用 file_path_endswith 后缀匹配**(actual 等于 suffix
265
+ 或在路径分隔符后以 suffix 结尾即命中),无论 LLM 传相对、绝对、~ 起头都能命中。
266
+ - 仅当 sample 明确给了 absolute path 且要测 LLM 用对完整路径(如 ~/.config/x.json)时才
267
+ 用 file_path 严格相等。
268
+ command_glob 示例:
269
+ - "mcporter call * --tool find_drm_value*" (拦 MCP find_drm_value 调用)
270
+ - "code-host pr show *" (拦 code-host CLI)
271
+ - "git push *" (拦 git push)
13
272
 
14
273
  要求:
15
274
  1. 评测用例应覆盖 skill 的不同能力维度
@@ -19,48 +278,172 @@ const SYSTEM_PROMPT = `你是一个评测用例生成器。你的任务是根据
19
278
  5. assertions 的 value / pattern / values / reference 必须使用英文、数字或代码 token,不要使用中文关键词。
20
279
  6. 断言应检测 skill 文档中的具体细节(如特定参数名、配置值、工作流步骤),而非通用知识。
21
280
  避免使用 baseline 凭常识或搜索文件也能答对的断言(如 not_contains 通用错误写法)。
22
- 优先使用 contains 检测文档独有的术语、参数组合或特定值
281
+
282
+ ⛔ **不许凭空具体化** — 这是 generator 的最大反模式之一:
283
+ 如果 SKILL.md 描述了某个步骤但**没明文指定该步用什么工具 / 什么命令 / 什么 API**
284
+ (只说"留档到 X"、"通知 Y"、"调用第三方服务 Z"这种意图描述,不说具体 tool / endpoint),
285
+ **fact 层断言不许猜测具体工具名**:
286
+ - ❌ 错的做法:SKILL.md 说"留档到语雀",generator 自己脑补"语雀 = URL = WebFetch",
287
+ 产 \`tool_input_contains "WebFetch:语雀URL"\` + \`mock_hit "WebFetch:N"\` —
288
+ 这是在测 generator 自己的脑补,不是 SKILL 实际要求,LLM 一选别的工具就判挂
289
+ - ❌ 错的做法:SKILL.md 说"通知钉钉",generator 假设走 Bash + 某个钉钉机器人 URL —
290
+ SKILL.md 没说就别假设
291
+ - ✅ 对的做法:把这个"应当完成的任务"写进 sample.rubric,让 judge 按 rubric 评分,
292
+ 工具选择交给 LLM 自由发挥,judge 看意图(任务完成与否)而不是字面(用了哪个工具)
293
+ - ✅ 兜底做法:如果一定要测"必须调到某工具",也只在 SKILL.md 明文说过该工具时才用
294
+ tool_input_contains;否则用 tools_called 列一组"可接受工具"也比单写一个稳
295
+ 判断标准:**写 sample 时,如果你需要去 SKILL.md 外的知识(推断"语雀对应什么工具"、
296
+ "通知钉钉用什么 API")才能写出 fact 层断言,这条断言就不该存在,应该归到 rubric。**
297
+
298
+ 📌 **URL/路径出现在 SKILL.md 里 ≠ 知道用什么工具访问它**(高频陷阱):
299
+ SKILL.md 文档里出现 \`https://wiki.example.com/xxx/yyy\` 这种 wiki 形态 URL,**不代表**
300
+ 该步骤就走 WebFetch。WebFetch / WebSearch 是 readonly GET 类工具,**只用于
301
+ "读取 / 抓取 / 查询 / 搜索" 语义**。SKILL.md 描述是"留档 / 写入 / 创建 / 推送 /
302
+ 通知 / 上传"这类**写动作**,而又没明文说"用 X 工具调"时:
303
+ - ❌ 不要产 \`tool_input_contains "WebFetch:irk5ik/kg7h1z"\` —— WebFetch 不写入,
304
+ LLM 调它也是 GET,断言铁定挂
305
+ - ❌ 不要假设 "URL 出现 = 该用 WebFetch" 的联想链,SKILL.md 给 URL 经常只是
306
+ 说明性指向(告诉读者"我们的知识库地址"),不是规定 LLM 必须 fetch 它
307
+ - ✅ 把"应当留档到 X"写进 rubric,工具留给 LLM/judge 决定。如果作者真的知道
308
+ 写语雀用什么 CLI/MCP(比如 \`skylark-doc\`),要么 SKILL.md 明文写,要么
309
+ sample.environment.cli_available 加上,fact 层断言才有依据
310
+ - ✅ 自检:在产 \`tool_input_contains "T:needle"\` 之前,grep 一下 SKILL.md
311
+ 看有没有出现过工具名 T(WebFetch / Bash / Read / Edit / Write / Glob /
312
+ Grep / 某 MCP 名),没出现就别用这个工具名 — 不许猜
313
+
314
+
315
+ **断言类型选择口诀**(fact 层只测**结果 + 必经里程碑**,过程质量交给 judge 评 rubric):
316
+
317
+ ✅ fact 应该测的(结果 / 里程碑):
318
+ - 测"最终产物是否写对" → tool_input_contains 抓 Write 的目标路径片段 /
319
+ tool_output_contains 抓 Read 命中的关键字段
320
+ - 测"最终回答包含某不可替换字面"(错误码 / SDK 函数名 / 路径 token) → contains(单值,
321
+ value 必须是 ASCII token 形态,见上方"绝对禁止"清单)
322
+ - 测"必经的工具调用里程碑"(SKILL.md 明文强约束的步骤) → mock_hit "Tool:N" 或
323
+ tools_called: ["Bash", "Read"]
324
+ - 测"必须**没**调用某禁忌工具"(tripwire / 反模式) → tools_not_called(values 必须
325
+ 给具体工具名,不能空数组,见上方 loader 校验)
326
+ - 测"工作流不应踩到某路径或 flag" → tool_input_not_contains "Tool:needle"(注意
327
+ 不要用 not_contains — 那是扫文本输出的,LLM 在总结里复述就自触发)
328
+
329
+ ❌ fact **不应该**测的(都属于"过程/语义",归 rubric → judge 评):
330
+ - "中间步骤的具体命令字面"("git diff 用了 --name-only 没") — 命令变体太多
331
+ - "工具调用的顺序"("先 stash 再 pull" / "先识别阶段再读 checks") — 顺序质量是
332
+ judge 看完整 trace 的活,不是 fact 层一条 assertion 能表达的
333
+ - "错误处理路径"("API 失败时是否重试 / 是否 fallback") — 行为质量,rubric 维度
334
+ - "LLM 是否礼貌拒绝用户的诱导请求" — 语义意图,rubric 维度,judge 看意图不看字面
335
+ - "LLM 是否在解释中说明了 X 概念" — contains 字面挂"不阻塞"/"暂停"这种汉字 token
336
+ 在 7 道 prompt 演进 + hardcode sanitize 之后已经被 strip 干净了,不要再尝试
337
+
338
+ **数量配额**(hard cap):**每个 sample 总共 2-4 条 fact 断言** — 不许靠堆"测每一步
339
+ 工具参数"涨数量,多出来的都是 trajectory 噪音。如果你觉得 2-4 条覆盖不完作者意图
340
+ 的细节,把那些细节写进 sample.rubric 让 judge 按维度评分 — judge 信号本来就比"某
341
+ 汉字是否出现在 trace"更接近"任务做没做对"。
342
+
343
+ *跟测量学的关联:* fact 条目稀疏化后单条权重相对大、单次评测方差变大,跑评测时
344
+ 建议带 \`--repeat 2\`(或更大)测同 variant 内部 coefficient of variation,看 bootstrap
345
+ CI 下限而非点估计。这是 fact 干净换稳定性的等价交换,omk eval CLI 在 N<20 且
346
+ --repeat=1 时已有 stderr 警告提醒。
347
+ 7. 如果 skill 涉及外部调用(MCP/CLI/HTTP/文件读),**必须**为本 sample 生成 mocks 数组,
348
+ 保证评测时 0 真调底层。query 类返回贴近真实 schema 的示例数据,write 类返回 success。
23
349
 
24
350
  可选元数据字段如能判断顺便填,无法判断时省略整个字段即可):
25
- - capability: string[] — 该用例覆盖的能力维度,如 ["api-selection", "error-diagnosis"]
26
- - difficulty: "easy" | "medium" | "hard" — 难度等级
27
- - construct: string — 用例测的 construct 类型,建议值 "necessity"(测知识必要性)/ "quality"(测 skill 写得好不好)/ "capability"(测某具体能力)
351
+ - capability: string[] — 该用例覆盖的能力维度。**值必须是中文短语**,描述这条 sample 在测什么能力,如 ["接口选择", "错误诊断", "PR 编号解析", "多步工作流"]。**不要用英文 slug 形式**(如 ❌ "api-selection" / "pr-iid-resolution")。专有名词(API / PR / SQL / SDK)可以保留英文,但短语主体用中文。
352
+ - difficulty: "easy" | "medium" | "hard" — 难度等级。**值保持英文 enum**(系统识别符,UI 会自动展示成"容易/中等/困难")。
353
+ - construct: string — 用例测的 construct 类型。**值用中文**,三选一:"必要性"(测知识必要性,LLM 没 skill 时该 fail)/"质量"(测 skill 写得好不好)/"能力"(测某具体能力)。
354
+ - **tripwire: true** — **此 sample 是诱错样本时必填**。诱错样本(tripwire)= 故意诱导 LLM 走错的样本(用户用错前提 / 跳步骤 / 用错参数类型),目的是测 skill 是否能让 LLM 识破并纠正,**LLM 失败是预期结果**。
355
+ 影响:omk 评测时,diagnostic 看到 tripwire:true 不会建议改 skill(因为 LLM 该 fail),避免误导 skill 作者。
356
+ 典型识别:prompt 含"直接用 X 就行了"/"不用检查"/"我已经知道是 Y"等用户错误前提诱导 + assertions 含 tools_not_called 或反模式断言 + construct 通常是 "necessity"。
357
+ 规则:诱错样本必填 tripwire:true。普通 capability sample 不要写 tripwire 字段。
28
358
 
29
- 直接输出 JSON 数组,不要包含 markdown 代码块标记或其他内容。`;
30
- export async function generateSamples({ skillContent, count = 5, model = DEFAULT_MODEL, executorName = 'claude' }) {
31
- const executor = createExecutor(executorName);
32
- const prompt = `以下是需要评测的 skill 内容:
359
+ **JSON 输出规范(必须遵守)**:
360
+ - 直接输出 JSON 数组,不要包含 markdown 代码块标记或其他文字
361
+ - 字符串字段(prompt / rubric / capability 等)内部如需引号,**必须用全角「」**而不是半角 \`""\`,避免漏转义破坏 JSON 解析
362
+ - 例:错 → \`"prompt": "查询"Daily"标签..."\`(内部 \`"\` 未转义,JSON 解析失败)
363
+ 对 → \`"prompt": "查询「Daily」标签..."\`(全角引号,无转义压力)`;
364
+ /**
365
+ * 拼出送给 LLM 的 user prompt。抽出来便于单测验证 focus 是否真的注入了。
366
+ *
367
+ * count 语义:
368
+ * - number: 强制生成 N 条
369
+ * - undefined: 让 LLM 按系统提示里"样本结构决策"的类型对应范围自行判断数量
370
+ */
371
+ export function buildSamplesPrompt({ skillContent, count, focus }) {
372
+ const focusBlock = focus && focus.trim()
373
+ ? `\n\n额外要求(用户指定的场景重点):\n${focus.trim()}\n生成的用例必须优先覆盖以上场景,再在剩余配额内补充其它能力维度。`
374
+ : '';
375
+ const countLine = typeof count === 'number'
376
+ ? `请根据这个 skill 生成 ${count} 个评测用例。`
377
+ : `请根据这个 skill 自行判断合适的数量并生成评测用例(参考系统提示中"样本结构决策"对应类型的数量范围)。`;
378
+ return `以下是需要评测的 skill 内容:
33
379
 
34
380
  ${skillContent}
35
381
 
36
- 请根据这个 skill 生成 ${count} 个评测用例。直接输出 JSON 数组。`;
37
- const result = await executor({ model, system: SYSTEM_PROMPT, prompt });
38
- if (!result.ok) {
39
- throw new Error(`generation failed: ${result.error || 'unknown error'}`);
40
- }
41
- // Extract JSON from output (handle possible markdown code blocks)
42
- let jsonStr = result.output.trim();
43
- const jsonMatch = jsonStr.match(/```(?:json)?\s*([\s\S]*?)```/);
44
- if (jsonMatch) {
45
- jsonStr = jsonMatch[1].trim();
46
- }
47
- let samples;
48
- try {
49
- samples = JSON.parse(jsonStr);
50
- }
51
- catch {
52
- throw new Error('generated content is not valid JSON, please retry');
53
- }
54
- if (!Array.isArray(samples) || samples.length === 0) {
55
- throw new Error('generated result is empty, please retry');
382
+ ${countLine}直接输出 JSON 数组。${focusBlock}`;
383
+ }
384
+ export async function generateSamples({ skillContent, count, model = GENERATOR_DEFAULT_MODEL, executorName = 'claude', focus }) {
385
+ const executor = createExecutor(executorName);
386
+ const prompt = buildSamplesPrompt({ skillContent, count, focus });
387
+ // 生成场景比单次 eval 调用更重(LLM 要思考结构 + 输出大段 JSON),
388
+ // 默认 120s 对长 skill + count >= 8 经常不够,这里用 5 分钟兜底。
389
+ // lean=true 关掉 agent 工具循环 / skill 发现 — 生成只需要纯文本,不需要 Bash / Read 等工具。
390
+ // 重试机制:LLM 偶尔输出 JSON 内含未转义引号 / 截断 / 多余文字。最多 2 次额外尝试,
391
+ // 第二次起在 prompt 末尾追加上一次的错误反馈,引导模型自纠。
392
+ const MAX_ATTEMPTS = 3;
393
+ let lastErr = '';
394
+ let totalCost = 0;
395
+ for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) {
396
+ const attemptPrompt = attempt === 1
397
+ ? prompt
398
+ : `${prompt}\n\n上一次输出解析失败:${lastErr}\n请严格按 JSON 规范输出(字符串内部用「」全角引号),只输出数组,不要包含其他文字。`;
399
+ const result = await executor({ model, system: SYSTEM_PROMPT, prompt: attemptPrompt, timeoutMs: 300_000, lean: true });
400
+ totalCost += result.costUSD || 0;
401
+ if (!result.ok) {
402
+ lastErr = result.error || 'unknown error';
403
+ if (attempt === MAX_ATTEMPTS)
404
+ throw new Error(`generation failed after ${MAX_ATTEMPTS} attempts: ${lastErr}`);
405
+ continue;
406
+ }
407
+ let jsonStr = result.output.trim();
408
+ const jsonMatch = jsonStr.match(/```(?:json)?\s*([\s\S]*?)```/);
409
+ if (jsonMatch)
410
+ jsonStr = jsonMatch[1].trim();
411
+ let samples;
412
+ try {
413
+ samples = JSON.parse(jsonStr);
414
+ }
415
+ catch (e) {
416
+ lastErr = `JSON 解析失败: ${e.message}`;
417
+ if (attempt === MAX_ATTEMPTS)
418
+ throw new Error(`generation failed after ${MAX_ATTEMPTS} attempts (JSON invalid): ${lastErr}`);
419
+ process.stderr.write(`[omk improve samples] 第 ${attempt} 次输出 JSON 无效,重试中...\n`);
420
+ continue;
421
+ }
422
+ if (!Array.isArray(samples) || samples.length === 0) {
423
+ lastErr = '输出为空数组';
424
+ if (attempt === MAX_ATTEMPTS)
425
+ throw new Error(`generation failed after ${MAX_ATTEMPTS} attempts: ${lastErr}`);
426
+ continue;
427
+ }
428
+ // 通过校验,跳出循环继续后续 sanitize
429
+ return await finalizeSamples(samples, totalCost, skillContent);
56
430
  }
57
- // Validate required fields + sanitize metadata enums *at generator boundary*
58
- // (see sanitizeGeneratedSamples).
59
- const { stripped } = sanitizeGeneratedSamples(samples);
431
+ // 不可达 (循环里所有出口都 throw 或 return),保留是为了 TS 类型推断
432
+ throw new Error('unreachable');
433
+ }
434
+ async function finalizeSamples(samples, costUSD, skillContent) {
435
+ // Validate required fields + sanitize metadata enums *at generator boundary*
436
+ // (see sanitizeGeneratedSamples). skillContent is passed so the function can
437
+ // strip "脑补"-style fact assertions whose tool name has no literal mention
438
+ // in SKILL.md — closes the prompt-can't-fully-suppress-this gap exposed by
439
+ // the data-security-review v1-v5 regen series (generator kept producing
440
+ // tool_input_contains "WebFetch:语雀URL" even after 7 prompt iterations,
441
+ // because LLM's "URL → fetch" training prior overrides instructional text).
442
+ const { stripped } = sanitizeGeneratedSamples(samples, { skillContent });
60
443
  if (stripped.length > 0) {
61
- process.stderr.write(`[omk sample] LLM-output 含 ${stripped.length} 个非法元数据字段,已剥离避免污染:\n - ${stripped.join('\n - ')}\n`);
444
+ process.stderr.write(`[omk sample] LLM-output 含 ${stripped.length} 个非法元数据/断言字段,已剥离避免污染:\n - ${stripped.join('\n - ')}\n`);
62
445
  }
63
- return { samples, costUSD: result.costUSD };
446
+ return { samples, costUSD };
64
447
  }
65
448
  /**
66
449
  * Validate + sanitize LLM-generated samples at generator boundary.
@@ -72,6 +455,30 @@ ${skillContent}
72
455
  * with a stderr warn (don't throw — valid required fields should still
73
456
  * produce usable samples).
74
457
  *
458
+ * Assertion-level sanitize (hard rules complementing SYSTEM_PROMPT soft guidance —
459
+ * prompt-only path was proven insufficient: data-security-review v1-v5 regens kept
460
+ * producing the same WebFetch-on-URL hallucination across 7 prompt revisions):
461
+ *
462
+ * A. Text-class assertion value (contains / not_contains / contains_any /
463
+ * contains_all / regex.pattern) must not contain CJK characters,
464
+ * fullwidth punctuation, internal ASCII whitespace, or be out of
465
+ * length range [2, 40]. LLM's natural text matches are unstable under
466
+ * synonym rewriting, so a literal Chinese-phrase contains is guaranteed
467
+ * noise — it either misses on every alternative phrasing the LLM picks
468
+ * next run, or triggers on the LLM's own summary mentioning the
469
+ * forbidden word.
470
+ *
471
+ * B. Positive tool-bound assertion (tool_input_contains / tool_output_contains /
472
+ * mock_hit) must have a tool name (left half of "Tool:needle") that
473
+ * literally appears (case-insensitive, word-boundary) somewhere in the
474
+ * provided SKILL.md content. Rationale: if the SKILL.md author meant the
475
+ * step to involve a specific tool, the tool name appears in the doc.
476
+ * Generator inferring "URL → WebFetch" or "Slack notification → curl" is
477
+ * hallucination that turns into 100% false-negative pressure on fact
478
+ * score. Negative variants (tool_input_not_contains, tools_not_called) are
479
+ * exempt — they encode forbidden actions, which by definition aren't in
480
+ * the SKILL.md.
481
+ *
75
482
  * Behavior:
76
483
  * - `sample_id` defaulted if missing
77
484
  * - `prompt` missing → throw (required)
@@ -79,13 +486,50 @@ ${skillContent}
79
486
  * - `difficulty` not in enum → strip
80
487
  * - `construct` not non-empty string → strip
81
488
  * - `provenance` not in enum → strip,then auto-stamp 'llm-generated'
489
+ * - assertion violating rules A or B above → strip that one assertion
490
+ * (keep the sample, since the rest of its assertions / judge rubric
491
+ * are still valid signal sources)
492
+ *
493
+ * `opts.skillContent` is the raw SKILL.md text the generator fed to the
494
+ * authoring LLM. When omitted, rule B silently passes (loader-side tests
495
+ * and unit-level callers that don't have a SKILL.md handy still work).
82
496
  *
83
497
  * Mutates the samples array in-place (matches generator's existing style).
84
498
  * Returns `{ stripped: string[] }` for warning aggregation + tests.
85
499
  */
86
- export function sanitizeGeneratedSamples(samples) {
500
+ const CJK_OR_FULLWIDTH = /[ -〿一-鿿㐀-䶿＀-￯]/;
501
+ const TEXT_VALUE_TYPES = new Set([
502
+ 'contains', 'not_contains', 'contains_all', 'contains_any', 'equals', 'not_equals',
503
+ ]);
504
+ const TOOL_POSITIVE_TYPES = new Set([
505
+ 'tool_input_contains', 'tool_output_contains', 'mock_hit',
506
+ ]);
507
+ function isAsciiTokenLike(v) {
508
+ if (typeof v !== 'string')
509
+ return false;
510
+ const s = v.trim();
511
+ if (s.length < 2 || s.length > 40)
512
+ return false;
513
+ if (CJK_OR_FULLWIDTH.test(s))
514
+ return false;
515
+ // 含内部空白(多 token 短语)拒,但允许首尾空格被 trim 已忽略
516
+ if (/\s/.test(s))
517
+ return false;
518
+ return true;
519
+ }
520
+ function toolNameAppearsInSkill(tool, skillContent) {
521
+ if (!skillContent)
522
+ return true; // no skill context — let it through (loader-side)
523
+ // case-insensitive word-boundary match. tool 名是 ASCII 标识符 (Bash/Read/WebFetch/MCP 名),
524
+ // 不会含正则元字符,直接拼即可 — 但 hyphen 在某些 MCP 名里出现(如 skylark-doc),
525
+ // hyphen 不是正则特殊字符,RegExp 构造也无需转义。
526
+ const esc = tool.replace(/[\\^$.*+?()[\]{}|]/g, '\\$&');
527
+ return new RegExp(`(?:^|[^A-Za-z0-9_-])${esc}(?:$|[^A-Za-z0-9_-])`, 'i').test(skillContent);
528
+ }
529
+ export function sanitizeGeneratedSamples(samples, opts = {}) {
87
530
  const VALID_DIFFICULTY = new Set(['easy', 'medium', 'hard']);
88
531
  const VALID_PROVENANCE = new Set(['human', 'llm-generated', 'production-trace']);
532
+ const skillContent = opts.skillContent || '';
89
533
  const stripped = [];
90
534
  for (const [i, s] of samples.entries()) {
91
535
  // sample_id / prompt 必须是 non-empty string。LLM 偶尔返回 number / null,
@@ -117,6 +561,143 @@ export function sanitizeGeneratedSamples(samples) {
117
561
  // After stripping invalid provenance, auto-stamp the generator's authority value.
118
562
  if (!s.provenance)
119
563
  s.provenance = 'llm-generated';
564
+ // tripwire 校验:必须是 boolean(true / false 都允许,但 LLM 偶尔写 "true" 字符串)。
565
+ // 非 boolean 一律 strip。omk diagnostic 会查 sample.tripwire === true。
566
+ if (s.tripwire !== undefined && typeof s.tripwire !== 'boolean') {
567
+ stripped.push(`samples[${i}].tripwire (${typeof s.tripwire})`);
568
+ delete s.tripwire;
569
+ }
570
+ // assertions 校验:loader 会拒掉两类无效断言 — 在 generator boundary 提前 strip,
571
+ // 避免落盘的 sample 跑不动:
572
+ // 1. tools_called / tools_not_called 的 values 必须非空
573
+ // 2. tool_input_contains / tool_input_not_contains / tool_output_contains / mock_hit
574
+ // 的 value 必须是 "Tool:needle" 格式(冒号分隔,两侧非空)
575
+ if (Array.isArray(s.assertions)) {
576
+ const before = s.assertions.length;
577
+ const TOOL_COLON = new Set([
578
+ 'tool_input_contains', 'tool_input_not_contains', 'tool_output_contains', 'mock_hit',
579
+ ]);
580
+ s.assertions = s.assertions.filter((a, j) => {
581
+ if (a?.type === 'tools_called' || a?.type === 'tools_not_called') {
582
+ const vals = Array.isArray(a.values) ? a.values : [];
583
+ const ok = vals.length > 0 && vals.every((v) => typeof v === 'string' && v.length > 0);
584
+ if (!ok)
585
+ stripped.push(`samples[${i}].assertions[${j}].${a.type} (empty values)`);
586
+ return ok;
587
+ }
588
+ if (TOOL_COLON.has(a?.type)) {
589
+ const v = a?.value;
590
+ if (typeof v !== 'string' || v.length === 0) {
591
+ stripped.push(`samples[${i}].assertions[${j}].${a.type} (missing value)`);
592
+ return false;
593
+ }
594
+ const sep = v.indexOf(':');
595
+ if (sep <= 0 || sep === v.length - 1) {
596
+ stripped.push(`samples[${i}].assertions[${j}].${a.type} (value not "Tool:needle": ${JSON.stringify(v)})`);
597
+ return false;
598
+ }
599
+ // Rule B: positive tool-bound assertions — tool name must literally
600
+ // appear in SKILL.md. Negative variants (tool_input_not_contains) are
601
+ // exempt because forbidden tools won't be mentioned in the doc.
602
+ if (TOOL_POSITIVE_TYPES.has(a.type) && skillContent) {
603
+ const toolName = v.slice(0, sep);
604
+ if (!toolNameAppearsInSkill(toolName, skillContent)) {
605
+ stripped.push(`samples[${i}].assertions[${j}].${a.type} 工具名 "${toolName}" 未在 SKILL.md 字面出现 — generator 凭空联想,断言去归 rubric`);
606
+ return false;
607
+ }
608
+ }
609
+ }
610
+ // Rule A: text-class value content guard — reject CJK chars, fullwidth
611
+ // punctuation, internal whitespace, or out-of-range length [2, 40].
612
+ // LLM text output is unstable under synonym/句式 rewriting; literal
613
+ // matches on Chinese phrases are guaranteed noise.
614
+ if (TEXT_VALUE_TYPES.has(a?.type)) {
615
+ const items = Array.isArray(a.values) ? a.values
616
+ : a.value !== undefined ? [a.value]
617
+ : [];
618
+ if (items.length === 0) {
619
+ stripped.push(`samples[${i}].assertions[${j}].${a.type} (empty value/values)`);
620
+ return false;
621
+ }
622
+ for (const v of items) {
623
+ if (!isAsciiTokenLike(v)) {
624
+ stripped.push(`samples[${i}].assertions[${j}].${a.type} value 非 ASCII token (含中文/全角标点/空格/长度越界): ${JSON.stringify(v)}`);
625
+ return false;
626
+ }
627
+ }
628
+ }
629
+ if (a?.type === 'regex' && typeof a.pattern === 'string' && CJK_OR_FULLWIDTH.test(a.pattern)) {
630
+ stripped.push(`samples[${i}].assertions[${j}].regex pattern 含 CJK/全角字符: ${JSON.stringify(a.pattern)}`);
631
+ return false;
632
+ }
633
+ return true;
634
+ });
635
+ // 全部 assertions 被 strip 完留空数组也保留 — sample 仍可用纯 LLM judge 评。
636
+ if (s.assertions.length === 0 && before > 0) {
637
+ delete s.assertions;
638
+ }
639
+ }
640
+ // environment 校验:必须是对象,内部字段要么是 string[] 要么是 string。
641
+ // 非法字段 strip 掉,避免 runtime 注入时炸 prompt。
642
+ if (s.environment !== undefined) {
643
+ if (typeof s.environment !== 'object' || s.environment === null || Array.isArray(s.environment)) {
644
+ stripped.push(`samples[${i}].environment (${typeof s.environment})`);
645
+ delete s.environment;
646
+ }
647
+ else {
648
+ const env = s.environment;
649
+ if (env.cli_available !== undefined && (!Array.isArray(env.cli_available) || !env.cli_available.every((x) => typeof x === 'string' && x.length > 0))) {
650
+ stripped.push(`samples[${i}].environment.cli_available (${typeof env.cli_available})`);
651
+ delete env.cli_available;
652
+ }
653
+ if (env.files_available !== undefined && (!Array.isArray(env.files_available) || !env.files_available.every((x) => typeof x === 'string' && x.length > 0))) {
654
+ stripped.push(`samples[${i}].environment.files_available (${typeof env.files_available})`);
655
+ delete env.files_available;
656
+ }
657
+ if (env.notes !== undefined && (typeof env.notes !== 'string')) {
658
+ stripped.push(`samples[${i}].environment.notes (${typeof env.notes})`);
659
+ delete env.notes;
660
+ }
661
+ // 如果所有子字段都没了,整个 environment 也删掉
662
+ if (Object.keys(env).length === 0) {
663
+ delete s.environment;
664
+ }
665
+ }
666
+ }
667
+ // mocks 校验:必须是数组,每项必须有 tool(string)+ 至少一种 return。
668
+ // 非法的 strip 掉,避免 runtime 装 hook 时炸。
669
+ if (s.mocks !== undefined) {
670
+ if (!Array.isArray(s.mocks)) {
671
+ stripped.push(`samples[${i}].mocks (${typeof s.mocks})`);
672
+ delete s.mocks;
673
+ }
674
+ else {
675
+ const validMocks = [];
676
+ for (let j = 0; j < s.mocks.length; j++) {
677
+ const m = s.mocks[j];
678
+ if (typeof m?.tool !== 'string' || m.tool.length === 0) {
679
+ stripped.push(`samples[${i}].mocks[${j}].tool (missing/invalid)`);
680
+ continue;
681
+ }
682
+ if (m.return === undefined && m.return_file === undefined && m.return_seq === undefined) {
683
+ stripped.push(`samples[${i}].mocks[${j}] (no return/return_file/return_seq)`);
684
+ continue;
685
+ }
686
+ validMocks.push(m);
687
+ }
688
+ if (validMocks.length > 0)
689
+ s.mocks = validMocks;
690
+ else
691
+ delete s.mocks;
692
+ }
693
+ }
694
+ // mocksStrict 兜底:有 mocks 时强制 true。
695
+ // SYSTEM_PROMPT 已要求 LLM 必填,但偶尔 LLM 漏填 — 在 generator boundary 修掉,
696
+ // 避免运行时 mock 未命中透传到真 shell(报 mcporter not found 等噪声错误)。
697
+ // LLM 显式给 false 时尊重(罕见,但保留 escape hatch — 比如混合 mock + 真 fs 的特殊场景)。
698
+ if (s.mocks && s.mocks.length > 0 && s.mocksStrict === undefined) {
699
+ s.mocksStrict = true;
700
+ }
120
701
  }
121
702
  return { stripped };
122
703
  }