oh-my-knowledge 0.51.1 → 0.52.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/README.md +27 -6
  2. package/README.zh.md +27 -6
  3. package/dist/assets/agent-skills/omk/SKILL.md +10 -8
  4. package/dist/assets/agent-skills/omk/references/commands.md +2 -2
  5. package/dist/authoring/generator.d.ts +15 -5
  6. package/dist/authoring/generator.js +181 -18
  7. package/dist/authoring/sample-fixer.d.ts +2 -0
  8. package/dist/authoring/sample-fixer.js +19 -5
  9. package/dist/cli/commands/sample.js +11 -4
  10. package/dist/eval-core/evaluation-execution.js +11 -2
  11. package/dist/eval-core/fact-checker.d.ts +15 -2
  12. package/dist/eval-core/fact-checker.js +75 -20
  13. package/dist/eval-core/mock-hook.cjs +40 -1
  14. package/dist/eval-core/mocks-runtime.js +39 -2
  15. package/dist/eval-core/task-planner.d.ts +2 -2
  16. package/dist/eval-core/task-planner.js +7 -7
  17. package/dist/eval-workflows/evaluation-pipeline.js +2 -0
  18. package/dist/eval-workflows/run-evaluation.js +2 -1
  19. package/dist/executors/capabilities.d.ts +15 -0
  20. package/dist/executors/capabilities.js +64 -0
  21. package/dist/executors/index.d.ts +1 -0
  22. package/dist/executors/index.js +4 -1
  23. package/dist/observability/codex-conversation-index.d.ts +88 -0
  24. package/dist/observability/codex-conversation-index.js +566 -0
  25. package/dist/observability/codex-protocol.d.ts +12 -0
  26. package/dist/observability/codex-protocol.js +78 -0
  27. package/dist/observability/codex-tool-status.d.ts +16 -0
  28. package/dist/observability/codex-tool-status.js +114 -0
  29. package/dist/observability/codex-trace-adapter.js +342 -119
  30. package/dist/observability/conversation-catalog.d.ts +38 -0
  31. package/dist/observability/conversation-catalog.js +573 -0
  32. package/dist/observability/conversation-index-process.d.ts +1 -0
  33. package/dist/observability/conversation-index-process.js +31 -0
  34. package/dist/observability/conversation-view-model.d.ts +5 -0
  35. package/dist/observability/conversation-view-model.js +100 -0
  36. package/dist/observability/experience.d.ts +3 -1
  37. package/dist/observability/experience.js +248 -7
  38. package/dist/observability/inbox.js +40 -2
  39. package/dist/observability/knowledge-debugger.d.ts +4 -0
  40. package/dist/observability/knowledge-debugger.js +364 -0
  41. package/dist/observability/polling-subscription-hub.d.ts +27 -0
  42. package/dist/observability/polling-subscription-hub.js +149 -0
  43. package/dist/observability/source-record-archive.d.ts +13 -0
  44. package/dist/observability/source-record-archive.js +324 -0
  45. package/dist/observability/task-window.d.ts +8 -0
  46. package/dist/observability/task-window.js +36 -0
  47. package/dist/observability/trace-ir.d.ts +66 -1
  48. package/dist/observability/trace-source.d.ts +1 -0
  49. package/dist/observability/trace-source.js +12 -5
  50. package/dist/observability/turn-index.d.ts +3 -0
  51. package/dist/observability/turn-index.js +251 -0
  52. package/dist/renderer/conversation-renderer.d.ts +9 -0
  53. package/dist/renderer/conversation-renderer.js +410 -0
  54. package/dist/renderer/icons.js +1 -0
  55. package/dist/renderer/inline-markdown.d.ts +6 -0
  56. package/dist/renderer/inline-markdown.js +158 -0
  57. package/dist/renderer/knowledge-debugger-renderer.d.ts +9 -0
  58. package/dist/renderer/knowledge-debugger-renderer.js +2280 -0
  59. package/dist/renderer/observation-inbox-renderer.js +9 -0
  60. package/dist/renderer/skill-list-renderer.js +2 -1
  61. package/dist/renderer/trajectory-live.d.ts +42 -0
  62. package/dist/renderer/trajectory-live.js +258 -0
  63. package/dist/renderer/trajectory-routing.d.ts +60 -0
  64. package/dist/renderer/trajectory-routing.js +351 -0
  65. package/dist/server/report-server.d.ts +4 -1
  66. package/dist/server/report-server.js +274 -6
  67. package/dist/server/skill-insights.js +1 -1
  68. package/dist/shared/sample-contract.d.ts +1 -0
  69. package/dist/shared/sample-contract.js +35 -0
  70. package/dist/shared/tool-identity.d.ts +8 -0
  71. package/dist/shared/tool-identity.js +13 -0
  72. package/dist/types/eval.d.ts +7 -6
  73. package/dist/types/executor.d.ts +3 -2
  74. package/dist/types/observability.d.ts +225 -1
  75. package/package.json +10 -7
@@ -20,8 +20,17 @@ const FIX_SYSTEM_PROMPT = `你是一个评测用例修复专家。根据诊断
20
20
  输出一个 **JSON 数组**,包含所有待修复 sample(修改过的和原样保留的都要包含)。
21
21
  第一字符 \`[\`,最后 \`]\`,不要用 \`\`\`json\`\`\` 围栏,不要寒暄。
22
22
  如果判断某条是 LLM 行为问题不需要改,也原样放进数组。`;
23
- function sanitizeFixedSamples(samples, skillContent) {
24
- sanitizeGeneratedSamples(samples, { skillContent });
23
+ const MOCKLESS_FIX_OVERRIDE = `
24
+
25
+ 目标执行器不支持工具调用拦截。不得新增或保留 mocks、mocksStrict、
26
+ mock_hit、tools_called、tools_count_min、tool_input_contains、tool_output_contains。
27
+ environment 仅表示题设上下文,不得当作已物化 fixture;负向安全断言可以保留。`;
28
+ function sanitizeFixedSamples(samples, skillContent, mockless) {
29
+ sanitizeGeneratedSamples(samples, {
30
+ skillContent,
31
+ mockless,
32
+ preserveMocklessEnvironment: true,
33
+ });
25
34
  return samples;
26
35
  }
27
36
  const FIXABLE_SAMPLE_FIELDS = new Set([
@@ -97,7 +106,7 @@ function parseFixedSamples(text) {
97
106
  return null;
98
107
  }
99
108
  export async function fixSamples(options) {
100
- const { skillContent, samples, report, treatmentKey, executor, model, maxAttemptsPerSample = 2 } = options;
109
+ const { skillContent, samples, report, treatmentKey, executor, model, maxAttemptsPerSample = 2, mockless = false, } = options;
101
110
  const sampleMap = new Map(samples.map((s) => [s.sample_id, s]));
102
111
  // Collect all fixable samples into one batch
103
112
  const fixContexts = [];
@@ -185,7 +194,12 @@ ${sampleSections}
185
194
  let incurredCostUSD = 0;
186
195
  let incurredCostReported = false;
187
196
  try {
188
- const result = await executor({ model, system: FIX_SYSTEM_PROMPT, prompt, timeoutMs: 300_000 });
197
+ const result = await executor({
198
+ model,
199
+ system: mockless ? `${FIX_SYSTEM_PROMPT}${MOCKLESS_FIX_OVERRIDE}` : FIX_SYSTEM_PROMPT,
200
+ prompt,
201
+ timeoutMs: 300_000,
202
+ });
189
203
  incurredCostUSD = result.costUSD;
190
204
  incurredCostReported = result.costReported !== false;
191
205
  if (!result.ok) {
@@ -226,7 +240,7 @@ ${sampleSections}
226
240
  })),
227
241
  };
228
242
  }
229
- const sanitizedFixedArr = sanitizeFixedSamples(fixedArr, skillContent);
243
+ const sanitizedFixedArr = sanitizeFixedSamples(fixedArr, skillContent, mockless);
230
244
  const outOfScope = sanitizedFixedArr.find((fixed) => {
231
245
  const sid = fixed.sample_id;
232
246
  const original = sampleMap.get(sid);
@@ -274,7 +274,7 @@ async function runSampleFix(args, flags, lang) {
274
274
  throw new CliExit(1);
275
275
  }
276
276
  process.stderr.write(lang === 'zh' ? `🔧 发现 ${sampleDesignCount} 条 sample_design 失败,开始修复...\n` : `🔧 Found ${sampleDesignCount} sample_design failure(s), fixing...\n`);
277
- const { createExecutor } = await import('../../executors/index.js');
277
+ const { createExecutor, executorSupportsSampleMocks, } = await import('../../executors/index.js');
278
278
  const exec = createExecutor(executorName);
279
279
  const executorFn = async (opts) => {
280
280
  const result = await exec({
@@ -298,6 +298,7 @@ async function runSampleFix(args, flags, lang) {
298
298
  treatmentKey: treatmentName,
299
299
  executor: executorFn,
300
300
  model,
301
+ mockless: !executorSupportsSampleMocks(executorName),
301
302
  });
302
303
  let writtenFiles = [];
303
304
  if (result.fixedCount > 0) {
@@ -363,7 +364,13 @@ export async function runSampleFromTraces(flags, lang) {
363
364
  ? `🔭 发现 ${items.length} 个${flags.skill ? ` ${flags.skill} 的` : ''}失败信号,正在生成评测用例草稿...\n`
364
365
  : `🔭 Found ${items.length}${flags.skill ? ` ${flags.skill}` : ''} failure signal(s); generating regression-sample drafts...\n`);
365
366
  try {
366
- const { samples, costUSD } = await generateSamplesFromTraces({ items, count, model, executorName });
367
+ const { samples, costUSD } = await generateSamplesFromTraces({
368
+ items,
369
+ count,
370
+ model,
371
+ executorName,
372
+ noMock: flags['no-mock'],
373
+ });
367
374
  const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
368
375
  if (samples.length === 0) {
369
376
  // The model conservatively skipped every signal (noise / unreproducible). That's a
@@ -663,8 +670,8 @@ export default class Sample extends BaseCommand {
663
670
  }),
664
671
  'no-mock': Flags.boolean({
665
672
  description: bilingual({
666
- zh: '不生成 mocks,eval 时所有工具调用真实执行。',
667
- en: 'Skip mock generation; all tool calls execute for real during eval.',
673
+ zh: '不生成 mocks。执行器不支持工具拦截时会自动启用,避免产生必然失败的 mock_hit。',
674
+ en: 'Skip mocks. Automatically enabled when the executor cannot intercept tools, preventing impossible mock_hit assertions.',
668
675
  }),
669
676
  default: false,
670
677
  }),
@@ -258,8 +258,17 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
258
258
  }
259
259
  }
260
260
  let factCheck;
261
- if (execResult.ok && execResult.output && task.cwd) {
262
- factCheck = checkFacts(execResult.output, resolve(task.cwd));
261
+ if (execResult.ok && execResult.output) {
262
+ const sharedEvidence = {
263
+ ...(task._sample.context && { context: task._sample.context }),
264
+ ...(task._sample.environment?.files_available?.length && {
265
+ declaredFiles: task._sample.environment.files_available,
266
+ }),
267
+ // Resolve exactly like the executor's cwd. samplesBaseDir is for bundle
268
+ // assets such as mocks, not for changing Sample.cwd path semantics.
269
+ ...(task._sample.cwd && { cwd: resolve(task._sample.cwd) }),
270
+ };
271
+ factCheck = checkFacts(execResult.output, sharedEvidence);
263
272
  }
264
273
  const sampleResults = ownRecordValue(results, task.sample_id)
265
274
  ?? setOwnRecordValue(results, task.sample_id, {});
@@ -14,11 +14,24 @@ export interface FactCheckResult {
14
14
  totalCount: number;
15
15
  verifiedRate: number;
16
16
  }
17
+ export interface FactCheckEvidence {
18
+ /** Shared sample fixture root. Arm-specific execution directories must not be used. */
19
+ cwd?: string;
20
+ /** Facts supplied to both arms in the sample context. */
21
+ context?: string;
22
+ /** Declarative fixture paths from sample.environment.files_available. */
23
+ declaredFiles?: string[];
24
+ }
17
25
  /**
18
26
  * Extract file path claims from agent output text.
19
27
  */
20
28
  export declare function extractPathClaims(output: string): string[];
21
29
  /**
22
- * Check facts in agent output by verifying file paths exist in cwd.
30
+ * Check file-path facts against sample-level evidence shared by every arm.
31
+ *
32
+ * A string keeps the legacy direct-filesystem API for callers outside the
33
+ * evaluation pipeline. The pipeline passes structured evidence and deliberately
34
+ * excludes each artifact's execution cwd, because that directory is not a
35
+ * comparable source of truth across control and treatment arms.
23
36
  */
24
- export declare function checkFacts(output: string, cwd: string): FactCheckResult;
37
+ export declare function checkFacts(output: string, evidence: string | FactCheckEvidence): FactCheckResult;
@@ -22,6 +22,10 @@ const IGNORE_PATTERNS = [
22
22
  /^\.git\//,
23
23
  /^index\.\w+$/,
24
24
  ];
25
+ // Product/runtime names that happen to end in a supported source extension.
26
+ const NON_PATH_REFERENCES = new Set([
27
+ 'node.js',
28
+ ]);
25
29
  /**
26
30
  * Extract file path claims from agent output text.
27
31
  */
@@ -36,35 +40,86 @@ export function extractPathClaims(output) {
36
40
  path = path.replace(/[.,;:!?))]+$/, '');
37
41
  // Clean trailing backtick/quote
38
42
  path = path.replace(/[`'"]+$/, '');
39
- if (path.length > 3 && !IGNORE_PATTERNS.some((p) => p.test(path))) {
43
+ if (path.length > 3
44
+ && !NON_PATH_REFERENCES.has(path.toLowerCase())
45
+ && !IGNORE_PATTERNS.some((p) => p.test(path))) {
40
46
  paths.add(path);
41
47
  }
42
48
  }
43
49
  }
44
50
  return [...paths];
45
51
  }
52
+ function normalizeEvidencePath(path) {
53
+ return path
54
+ .trim()
55
+ .replace(/^['"`]+|['"`]+$/g, '')
56
+ .replace(/\\/g, '/')
57
+ .replace(/^(?:\$SKILL_DIR|~)\//, '')
58
+ .replace(/^\.\//, '')
59
+ .replace(/\/+$/, '');
60
+ }
61
+ function isRootedEvidencePath(path) {
62
+ return path.startsWith('/') || /^[a-zA-Z]:\//.test(path);
63
+ }
64
+ function refersToSamePath(left, right) {
65
+ const a = normalizeEvidencePath(left);
66
+ const b = normalizeEvidencePath(right);
67
+ return a === b
68
+ || (isRootedEvidencePath(a) && !isRootedEvidencePath(b) && a.endsWith(`/${b}`))
69
+ || (isRootedEvidencePath(b) && !isRootedEvidencePath(a) && b.endsWith(`/${a}`));
70
+ }
46
71
  /**
47
- * Check facts in agent output by verifying file paths exist in cwd.
72
+ * Check file-path facts against sample-level evidence shared by every arm.
73
+ *
74
+ * A string keeps the legacy direct-filesystem API for callers outside the
75
+ * evaluation pipeline. The pipeline passes structured evidence and deliberately
76
+ * excludes each artifact's execution cwd, because that directory is not a
77
+ * comparable source of truth across control and treatment arms.
48
78
  */
49
- export function checkFacts(output, cwd) {
79
+ export function checkFacts(output, evidence) {
50
80
  const pathClaims = extractPathClaims(output);
51
- const root = resolve(cwd);
52
- const claims = pathClaims.map((path) => {
53
- const fullPath = resolve(root, path);
54
- const relativePath = relative(root, fullPath);
55
- const insideCwd = relativePath === ''
56
- || (!relativePath.startsWith('..') && !isAbsolute(relativePath));
57
- const exists = insideCwd && existsSync(fullPath);
58
- return {
59
- type: 'file-path',
60
- value: path,
61
- verified: exists,
62
- ...(!exists && {
63
- evidence: insideCwd
64
- ? `${fullPath} not found`
65
- : `${path} is outside the evaluation cwd`,
66
- }),
67
- };
81
+ const sources = typeof evidence === 'string'
82
+ ? { cwd: evidence }
83
+ : evidence;
84
+ const root = sources.cwd ? resolve(sources.cwd) : null;
85
+ const contextClaims = sources.context
86
+ ? extractPathClaims(sources.context)
87
+ : [];
88
+ const declaredFiles = sources.declaredFiles ?? [];
89
+ const claims = pathClaims.flatMap((path) => {
90
+ if (declaredFiles.some((declared) => refersToSamePath(path, declared))) {
91
+ return [{
92
+ type: 'file-path',
93
+ value: path,
94
+ verified: true,
95
+ evidence: 'source=context(sample.environment.files_available)',
96
+ }];
97
+ }
98
+ if (root) {
99
+ const fullPath = resolve(root, path);
100
+ const relativePath = relative(root, fullPath);
101
+ const insideCwd = relativePath === ''
102
+ || (!relativePath.startsWith('..') && !isAbsolute(relativePath));
103
+ const exists = insideCwd && existsSync(fullPath);
104
+ return [{
105
+ type: 'file-path',
106
+ value: path,
107
+ verified: exists,
108
+ evidence: insideCwd
109
+ ? `source=runtime-filesystem; ${fullPath} ${exists ? 'exists' : 'not found'}`
110
+ : `source=runtime-filesystem; ${path} is outside the evaluation cwd`,
111
+ }];
112
+ }
113
+ if (contextClaims.some((contextPath) => refersToSamePath(path, contextPath))) {
114
+ return [{
115
+ type: 'file-path',
116
+ value: path,
117
+ verified: true,
118
+ evidence: 'source=context(sample.context)',
119
+ }];
120
+ }
121
+ // Without a shared fixture, absence from context is unknown rather than false.
122
+ return [];
68
123
  });
69
124
  const verifiedCount = claims.filter((c) => c.verified).length;
70
125
  const totalCount = claims.length;
@@ -92,8 +92,47 @@ function anyStringContains(obj, needle) {
92
92
  return false;
93
93
  }
94
94
 
95
+ const BUILTIN_TOOL_ALIASES = {
96
+ bash: 'Bash',
97
+ shell: 'Bash',
98
+ exec_command: 'Bash',
99
+ command_execution: 'Bash',
100
+ read: 'Read',
101
+ file_read: 'Read',
102
+ grep: 'Grep',
103
+ edit: 'Edit',
104
+ apply_patch: 'Edit',
105
+ file_change: 'Edit',
106
+ write: 'Write',
107
+ file_write: 'Write',
108
+ view_image: 'ViewImage',
109
+ viewimage: 'ViewImage',
110
+ write_stdin: 'WriteStdin',
111
+ writestdin: 'WriteStdin',
112
+ web_search: 'WebSearch',
113
+ websearch: 'WebSearch',
114
+ };
115
+
116
+ function canonicalToolName(name) {
117
+ const sourceName = String(name);
118
+ const builtin = BUILTIN_TOOL_ALIASES[sourceName.toLowerCase()];
119
+ if (builtin) return builtin;
120
+ const parts = sourceName.split('__').filter(Boolean);
121
+ if (parts[0] === 'mcp' && parts.length > 2) {
122
+ const providerParts = parts.slice(1, -1);
123
+ if (providerParts[0] === 'codex_apps' && providerParts.length > 1) providerParts.shift();
124
+ return providerParts.join('.') + '.' + parts[parts.length - 1];
125
+ }
126
+ return sourceName;
127
+ }
128
+
129
+ function toolIdentityMatches(expectedName, runtimeName) {
130
+ return expectedName === runtimeName
131
+ || canonicalToolName(expectedName) === canonicalToolName(runtimeName);
132
+ }
133
+
95
134
  function isMockHit(mock, toolName, toolInput) {
96
- if (mock.tool !== '*' && mock.tool !== toolName) return false;
135
+ if (mock.tool !== '*' && !toolIdentityMatches(mock.tool, toolName)) return false;
97
136
  const m = mock.match;
98
137
  if (!m) return true;
99
138
  const ti = toolInput || {};
@@ -18,6 +18,7 @@ import { homedir, tmpdir } from 'node:os';
18
18
  import { dirname, isAbsolute, join, resolve } from 'node:path';
19
19
  import { fileURLToPath } from 'node:url';
20
20
  import { incrementRecordCount, setOwnRecordValue } from '../shared/record-count.js';
21
+ import { toolIdentityMatches } from '../shared/tool-identity.js';
21
22
  // ─── Match logic ────────────────────────────────────────────────────────────
22
23
  function expandHome(p) {
23
24
  if (p.startsWith('~/'))
@@ -112,7 +113,7 @@ function anyStringContains(obj, needle) {
112
113
  }
113
114
  /** 单条 mock 是否命中给定 tool 调用。 */
114
115
  export function isMockHit(mock, toolName, toolInput) {
115
- if (mock.tool !== '*' && mock.tool !== toolName)
116
+ if (mock.tool !== '*' && !toolIdentityMatches(mock.tool, toolName))
116
117
  return false;
117
118
  const m = mock.match;
118
119
  if (!m)
@@ -414,8 +415,44 @@ function anyStringContains(obj, needle) {
414
415
  if (typeof obj === 'object' && obj !== null) return Object.values(obj).some((v) => anyStringContains(v, needle));
415
416
  return false;
416
417
  }
418
+ const BUILTIN_TOOL_ALIASES = {
419
+ bash: 'Bash',
420
+ shell: 'Bash',
421
+ exec_command: 'Bash',
422
+ command_execution: 'Bash',
423
+ read: 'Read',
424
+ file_read: 'Read',
425
+ grep: 'Grep',
426
+ edit: 'Edit',
427
+ apply_patch: 'Edit',
428
+ file_change: 'Edit',
429
+ write: 'Write',
430
+ file_write: 'Write',
431
+ view_image: 'ViewImage',
432
+ viewimage: 'ViewImage',
433
+ write_stdin: 'WriteStdin',
434
+ writestdin: 'WriteStdin',
435
+ web_search: 'WebSearch',
436
+ websearch: 'WebSearch',
437
+ };
438
+ function canonicalToolName(name) {
439
+ const sourceName = String(name);
440
+ const builtin = BUILTIN_TOOL_ALIASES[sourceName.toLowerCase()];
441
+ if (builtin) return builtin;
442
+ const parts = sourceName.split('__').filter(Boolean);
443
+ if (parts[0] === 'mcp' && parts.length > 2) {
444
+ const providerParts = parts.slice(1, -1);
445
+ if (providerParts[0] === 'codex_apps' && providerParts.length > 1) providerParts.shift();
446
+ return providerParts.join('.') + '.' + parts[parts.length - 1];
447
+ }
448
+ return sourceName;
449
+ }
450
+ function toolIdentityMatches(expectedName, runtimeName) {
451
+ return expectedName === runtimeName
452
+ || canonicalToolName(expectedName) === canonicalToolName(runtimeName);
453
+ }
417
454
  function isMockHit(mock, toolName, toolInput) {
418
- if (mock.tool !== '*' && mock.tool !== toolName) return false;
455
+ if (mock.tool !== '*' && !toolIdentityMatches(mock.tool, toolName)) return false;
419
456
  const m = mock.match;
420
457
  if (!m) return true;
421
458
  const ti = toolInput || {};
@@ -1,8 +1,8 @@
1
1
  import type { Artifact, Sample, SampleEnvironment, Task } from '../types/index.js';
2
2
  /**
3
3
  * 把 sample.environment 渲染成自然语言段落,放在用户 prompt 前。
4
- * 让 LLM 读到"环境已就绪",跳过 Glob / find / which / Read 这些环境探测,
5
- * 直接进入 skill 描述的工作流 — 评测信号纯,mock 设计也简化(不用 mock 探测命令)。
4
+ * 这是题设上下文,不会修改 PATH、物化文件或改变 runtime;LLM 可据此跳过
5
+ * which / test -f 等可用性探测,直接进入 skill 描述的工作流。
6
6
  *
7
7
  * 输出 null 表示 sample 没声明 environment,prompt 不变。
8
8
  */
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * 把 sample.environment 渲染成自然语言段落,放在用户 prompt 前。
3
- * 让 LLM 读到"环境已就绪",跳过 Glob / find / which / Read 这些环境探测,
4
- * 直接进入 skill 描述的工作流 — 评测信号纯,mock 设计也简化(不用 mock 探测命令)。
3
+ * 这是题设上下文,不会修改 PATH、物化文件或改变 runtime;LLM 可据此跳过
4
+ * which / test -f 等可用性探测,直接进入 skill 描述的工作流。
5
5
  *
6
6
  * 输出 null 表示 sample 没声明 environment,prompt 不变。
7
7
  */
@@ -10,26 +10,26 @@ export function renderEnvironmentSection(env) {
10
10
  return null;
11
11
  const lines = [];
12
12
  if (env.cli_available && env.cli_available.length > 0) {
13
- lines.push('- 已安装 CLI(已在 PATH,无需 `which` / `command -v` / `type` 探测):');
13
+ lines.push('- 题设声明可用的 CLI(仅作上下文,不修改 PATH):');
14
14
  for (const c of env.cli_available)
15
15
  lines.push(` - \`${c}\``);
16
16
  }
17
17
  if (env.files_available && env.files_available.length > 0) {
18
- lines.push('- 已存在文件(无需 Glob / `Read` / `test -f` 探测):');
18
+ lines.push('- 题设引用的文件路径(仅作上下文,不会在 cwd 物化):');
19
19
  for (const f of env.files_available)
20
20
  lines.push(` - \`${f}\``);
21
21
  }
22
22
  if (env.notes && env.notes.trim()) {
23
- lines.push(`- 备注:${env.notes.trim()}`);
23
+ lines.push(`- 备注:${env.notes.trim()}`);
24
24
  }
25
25
  if (lines.length === 0)
26
26
  return null;
27
27
  return [
28
- '## 评测环境前置(已就绪,无需探测)',
28
+ '## 题设环境声明(仅作上下文)',
29
29
  '',
30
30
  ...lines,
31
31
  '',
32
- '请直接进入 skill 描述的主流程,**不要做环境检查 / `Glob` / `find` / `which` / `test -f` 等探测**。',
32
+ '请按以上题设进入 skill 描述的主流程,**不要额外做 `find` / `which` / `test -f` 等可用性探测**。这些声明不会自动创建文件或修改 runtime 环境。',
33
33
  ].join('\n');
34
34
  }
35
35
  export function buildTasks(samples, variants, skills) {
@@ -28,6 +28,7 @@ import { finalizeSuccessfulRun, initializeEvaluationRunState, persistFailedJob,
28
28
  import { finalizeEvaluationReport } from './evaluation-pipeline/report-finalize.js';
29
29
  import { emitIsolationWarnings, emitPowerWarnings } from './evaluation-pipeline/preflight-warnings.js';
30
30
  import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
31
+ import { assertSamplesCompatibleWithExecutor } from '../executors/capabilities.js';
31
32
  // 兼容 re-export:测试与 run-evaluation.ts 动态 import 仍打 evaluation-pipeline.js
32
33
  export { buildPowerWarnings, buildIsolationWarnings } from './evaluation-pipeline/preflight-warnings.js';
33
34
  export { _computeTestSetHashForTest } from './evaluation-pipeline/test-set-hash.js';
@@ -35,6 +36,7 @@ export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, s
35
36
  // requires 现在由 runEvaluation 上游传给 doctor 处理; eval-pipeline 不再用
36
37
  // 但保留接口字段,避免破 programmatic API (类型层面接收, 内部忽略)
37
38
  requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
39
+ assertSamplesCompatibleWithExecutor(samples, executorName, lang);
38
40
  const variantNames = artifacts.map((artifact) => artifact.name);
39
41
  const runState = await initializeEvaluationRunState({
40
42
  samplesPath,
@@ -1,6 +1,6 @@
1
1
  import { resolve } from 'node:path';
2
2
  import { DEFAULT_OUTPUT_DIR, persistReport } from '../eval-core/evaluation-reporting.js';
3
- import { createExecutor } from '../executors/index.js';
3
+ import { assertSamplesCompatibleWithExecutor, createExecutor, } from '../executors/index.js';
4
4
  import { discoverBatchSkills } from '../inputs/skill-loader.js';
5
5
  import { confidenceInterval, tTest, effectSize } from '../eval-core/statistics.js';
6
6
  import { executeBatchEvaluationRuns, buildBatchVariantSpecs } from './batch-evaluation-workflow.js';
@@ -26,6 +26,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
26
26
  mcpConfig,
27
27
  strictBaseline,
28
28
  });
29
+ assertSamplesCompatibleWithExecutor(samples, executorName, lang);
29
30
  // doctor 强制门禁: skill 静态结构 + 元数据 + 依赖 + 用例契约。
30
31
  // 在 dryRun 分支之前跑, 让 dry-run 也得到 doctor 覆盖(保护 garbage-in 的 verdict)。
31
32
  // 默认强制启用; --skip-doctor 提供 escape hatch,典型场景是评测环境用 mock/stub
@@ -0,0 +1,15 @@
1
+ import type { ExecutorFn, ExecutorInput, Sample } from '../types/index.js';
2
+ export type SampleMockSupport = 'native-hooks' | 'delegated-script' | 'unsupported';
3
+ export interface ExecutorCapabilities {
4
+ sampleMocks: SampleMockSupport;
5
+ }
6
+ /**
7
+ * Custom script executors receive the OMK_MOCK_* protocol environment and own
8
+ * the final adapter. Built-ins are explicit so unsupported runtimes can never
9
+ * silently turn mock assertions into model failures.
10
+ */
11
+ export declare function getExecutorCapabilities(executorName: string): ExecutorCapabilities;
12
+ export declare function executorSupportsSampleMocks(executorName: string): boolean;
13
+ export declare function assertSamplesCompatibleWithExecutor(samples: Sample[], executorName: string, lang?: 'zh' | 'en'): void;
14
+ export declare function assertExecutorInputCapabilities(executorName: string, input: ExecutorInput): void;
15
+ export declare function enforceExecutorCapabilities(executorName: string, executor: ExecutorFn): ExecutorFn;
@@ -0,0 +1,64 @@
1
+ const BUILTIN_CAPABILITIES = {
2
+ claude: { sampleMocks: 'native-hooks' },
3
+ 'claude-sdk': { sampleMocks: 'native-hooks' },
4
+ codex: { sampleMocks: 'unsupported' },
5
+ 'codex-sdk': { sampleMocks: 'unsupported' },
6
+ gemini: { sampleMocks: 'unsupported' },
7
+ 'anthropic-api': { sampleMocks: 'unsupported' },
8
+ 'openai-api': { sampleMocks: 'unsupported' },
9
+ };
10
+ /**
11
+ * Custom script executors receive the OMK_MOCK_* protocol environment and own
12
+ * the final adapter. Built-ins are explicit so unsupported runtimes can never
13
+ * silently turn mock assertions into model failures.
14
+ */
15
+ export function getExecutorCapabilities(executorName) {
16
+ return BUILTIN_CAPABILITIES[executorName]
17
+ ?? { sampleMocks: 'delegated-script' };
18
+ }
19
+ export function executorSupportsSampleMocks(executorName) {
20
+ return getExecutorCapabilities(executorName).sampleMocks !== 'unsupported';
21
+ }
22
+ function unsupportedMocksMessage(executorName, sampleIds, lang) {
23
+ const ids = sampleIds.slice(0, 8).join(', ');
24
+ const overflow = sampleIds.length > 8
25
+ ? lang === 'zh'
26
+ ? ` 等 ${sampleIds.length} 条`
27
+ : ` and ${sampleIds.length - 8} more`
28
+ : '';
29
+ if (lang === 'en') {
30
+ return `Executor "${executorName}" does not support Sample.mocks tool interception. `
31
+ + `Continuing would make mock_hit assertions structurally impossible and create false evidence. `
32
+ + `Affected samples: ${ids}${overflow}. `
33
+ + 'Regenerate them with "omk sample --no-mock", remove mocks/mock_hit, '
34
+ + 'or evaluate with claude/claude-sdk.';
35
+ }
36
+ return `执行器「${executorName}」不支持 Sample.mocks 工具拦截。`
37
+ + `继续运行会让 mock_hit 在结构上必然失败并产生伪证据。`
38
+ + `受影响用例:${ids}${overflow}。`
39
+ + '请用「omk sample --no-mock」重新生成、删除 mocks/mock_hit,'
40
+ + '或改用 claude/claude-sdk 评测。';
41
+ }
42
+ export function assertSamplesCompatibleWithExecutor(samples, executorName, lang = 'zh') {
43
+ if (executorSupportsSampleMocks(executorName))
44
+ return;
45
+ const affected = samples
46
+ .filter((sample) => Array.isArray(sample.mocks) && sample.mocks.length > 0)
47
+ .map((sample) => sample.sample_id);
48
+ if (affected.length === 0)
49
+ return;
50
+ throw new Error(unsupportedMocksMessage(executorName, affected, lang));
51
+ }
52
+ export function assertExecutorInputCapabilities(executorName, input) {
53
+ if (executorSupportsSampleMocks(executorName)
54
+ || !Array.isArray(input.mocks)
55
+ || input.mocks.length === 0)
56
+ return;
57
+ throw new Error(unsupportedMocksMessage(executorName, ['<programmatic-input>'], 'zh'));
58
+ }
59
+ export function enforceExecutorCapabilities(executorName, executor) {
60
+ return async (input) => {
61
+ assertExecutorInputCapabilities(executorName, input);
62
+ return executor(input);
63
+ };
64
+ }
@@ -2,4 +2,5 @@ import type { ExecutorFn } from '../types/index.js';
2
2
  import { extractAgentTrace } from './claude-sdk-trace.js';
3
3
  import { createScriptExecutor } from './script.js';
4
4
  export { extractAgentTrace, createScriptExecutor };
5
+ export { assertExecutorInputCapabilities, assertSamplesCompatibleWithExecutor, executorSupportsSampleMocks, getExecutorCapabilities, } from './capabilities.js';
5
6
  export declare function createExecutor(name: string): ExecutorFn;
@@ -7,6 +7,7 @@ import { codexSdkExecutor } from './codex-sdk.js';
7
7
  import { geminiExecutor } from './gemini.js';
8
8
  import { openAiApiExecutor } from './openai-api.js';
9
9
  import { createScriptExecutor } from './script.js';
10
+ import { enforceExecutorCapabilities } from './capabilities.js';
10
11
  // 命名一致性:provider HTTP 路径统一用 `<vendor>-api`(`anthropic-api` / `openai-api`),
11
12
  // vendor coding agent CLI 用 vendor 名(`claude` / `codex`)。`openai` 这个不带 -api 后缀的
12
13
  // 旧 alias 历史上指 openai-cli 子进程实现,删除后不再设别名 — 用 `--executor openai-api`。
@@ -20,9 +21,11 @@ const EXECUTOR_REGISTRY = {
20
21
  'openai-api': openAiApiExecutor,
21
22
  };
22
23
  export { extractAgentTrace, createScriptExecutor };
24
+ export { assertExecutorInputCapabilities, assertSamplesCompatibleWithExecutor, executorSupportsSampleMocks, getExecutorCapabilities, } from './capabilities.js';
23
25
  export function createExecutor(name) {
24
26
  if (name.trim().length === 0) {
25
27
  throw new Error('executor name or script command is required');
26
28
  }
27
- return EXECUTOR_REGISTRY[name] || createScriptExecutor(name);
29
+ const executor = EXECUTOR_REGISTRY[name] || createScriptExecutor(name);
30
+ return enforceExecutorCapabilities(name, executor);
28
31
  }