oh-my-knowledge 0.51.0 → 0.51.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. package/dist/assets/agent-skills/omk/SKILL.md +2 -0
  2. package/dist/assets/agent-skills/omk/references/commands.md +2 -2
  3. package/dist/authoring/generator.d.ts +15 -5
  4. package/dist/authoring/generator.js +181 -18
  5. package/dist/authoring/sample-fixer.d.ts +2 -0
  6. package/dist/authoring/sample-fixer.js +19 -5
  7. package/dist/cli/commands/sample.js +11 -4
  8. package/dist/eval-core/dependency-checker.js +30 -18
  9. package/dist/eval-core/evaluation-execution.js +11 -2
  10. package/dist/eval-core/fact-checker.d.ts +15 -2
  11. package/dist/eval-core/fact-checker.js +75 -20
  12. package/dist/eval-core/mock-hook.cjs +40 -1
  13. package/dist/eval-core/mocks-runtime.js +39 -2
  14. package/dist/eval-core/task-planner.d.ts +2 -2
  15. package/dist/eval-core/task-planner.js +7 -7
  16. package/dist/eval-workflows/evaluation-pipeline.js +2 -0
  17. package/dist/eval-workflows/run-evaluation.js +2 -1
  18. package/dist/executors/capabilities.d.ts +15 -0
  19. package/dist/executors/capabilities.js +64 -0
  20. package/dist/executors/index.d.ts +1 -0
  21. package/dist/executors/index.js +4 -1
  22. package/dist/observability/codex-exec-command.d.ts +7 -0
  23. package/dist/observability/codex-exec-command.js +134 -0
  24. package/dist/observability/codex-trace-adapter.js +16 -6
  25. package/dist/observability/trace-attribution.js +11 -107
  26. package/dist/server/skill-insights.js +1 -1
  27. package/dist/shared/sample-contract.d.ts +1 -0
  28. package/dist/shared/sample-contract.js +35 -0
  29. package/dist/shared/tool-identity.d.ts +8 -0
  30. package/dist/shared/tool-identity.js +13 -0
  31. package/dist/types/eval.d.ts +7 -6
  32. package/dist/types/executor.d.ts +3 -2
  33. package/package.json +4 -4
@@ -22,6 +22,10 @@ const IGNORE_PATTERNS = [
22
22
  /^\.git\//,
23
23
  /^index\.\w+$/,
24
24
  ];
25
+ // Product/runtime names that happen to end in a supported source extension.
26
+ const NON_PATH_REFERENCES = new Set([
27
+ 'node.js',
28
+ ]);
25
29
  /**
26
30
  * Extract file path claims from agent output text.
27
31
  */
@@ -36,35 +40,86 @@ export function extractPathClaims(output) {
36
40
  path = path.replace(/[.,;:!?))]+$/, '');
37
41
  // Clean trailing backtick/quote
38
42
  path = path.replace(/[`'"]+$/, '');
39
- if (path.length > 3 && !IGNORE_PATTERNS.some((p) => p.test(path))) {
43
+ if (path.length > 3
44
+ && !NON_PATH_REFERENCES.has(path.toLowerCase())
45
+ && !IGNORE_PATTERNS.some((p) => p.test(path))) {
40
46
  paths.add(path);
41
47
  }
42
48
  }
43
49
  }
44
50
  return [...paths];
45
51
  }
52
+ function normalizeEvidencePath(path) {
53
+ return path
54
+ .trim()
55
+ .replace(/^['"`]+|['"`]+$/g, '')
56
+ .replace(/\\/g, '/')
57
+ .replace(/^(?:\$SKILL_DIR|~)\//, '')
58
+ .replace(/^\.\//, '')
59
+ .replace(/\/+$/, '');
60
+ }
61
+ function isRootedEvidencePath(path) {
62
+ return path.startsWith('/') || /^[a-zA-Z]:\//.test(path);
63
+ }
64
+ function refersToSamePath(left, right) {
65
+ const a = normalizeEvidencePath(left);
66
+ const b = normalizeEvidencePath(right);
67
+ return a === b
68
+ || (isRootedEvidencePath(a) && !isRootedEvidencePath(b) && a.endsWith(`/${b}`))
69
+ || (isRootedEvidencePath(b) && !isRootedEvidencePath(a) && b.endsWith(`/${a}`));
70
+ }
46
71
  /**
47
- * Check facts in agent output by verifying file paths exist in cwd.
72
+ * Check file-path facts against sample-level evidence shared by every arm.
73
+ *
74
+ * A string keeps the legacy direct-filesystem API for callers outside the
75
+ * evaluation pipeline. The pipeline passes structured evidence and deliberately
76
+ * excludes each artifact's execution cwd, because that directory is not a
77
+ * comparable source of truth across control and treatment arms.
48
78
  */
49
- export function checkFacts(output, cwd) {
79
+ export function checkFacts(output, evidence) {
50
80
  const pathClaims = extractPathClaims(output);
51
- const root = resolve(cwd);
52
- const claims = pathClaims.map((path) => {
53
- const fullPath = resolve(root, path);
54
- const relativePath = relative(root, fullPath);
55
- const insideCwd = relativePath === ''
56
- || (!relativePath.startsWith('..') && !isAbsolute(relativePath));
57
- const exists = insideCwd && existsSync(fullPath);
58
- return {
59
- type: 'file-path',
60
- value: path,
61
- verified: exists,
62
- ...(!exists && {
63
- evidence: insideCwd
64
- ? `${fullPath} not found`
65
- : `${path} is outside the evaluation cwd`,
66
- }),
67
- };
81
+ const sources = typeof evidence === 'string'
82
+ ? { cwd: evidence }
83
+ : evidence;
84
+ const root = sources.cwd ? resolve(sources.cwd) : null;
85
+ const contextClaims = sources.context
86
+ ? extractPathClaims(sources.context)
87
+ : [];
88
+ const declaredFiles = sources.declaredFiles ?? [];
89
+ const claims = pathClaims.flatMap((path) => {
90
+ if (declaredFiles.some((declared) => refersToSamePath(path, declared))) {
91
+ return [{
92
+ type: 'file-path',
93
+ value: path,
94
+ verified: true,
95
+ evidence: 'source=context(sample.environment.files_available)',
96
+ }];
97
+ }
98
+ if (root) {
99
+ const fullPath = resolve(root, path);
100
+ const relativePath = relative(root, fullPath);
101
+ const insideCwd = relativePath === ''
102
+ || (!relativePath.startsWith('..') && !isAbsolute(relativePath));
103
+ const exists = insideCwd && existsSync(fullPath);
104
+ return [{
105
+ type: 'file-path',
106
+ value: path,
107
+ verified: exists,
108
+ evidence: insideCwd
109
+ ? `source=runtime-filesystem; ${fullPath} ${exists ? 'exists' : 'not found'}`
110
+ : `source=runtime-filesystem; ${path} is outside the evaluation cwd`,
111
+ }];
112
+ }
113
+ if (contextClaims.some((contextPath) => refersToSamePath(path, contextPath))) {
114
+ return [{
115
+ type: 'file-path',
116
+ value: path,
117
+ verified: true,
118
+ evidence: 'source=context(sample.context)',
119
+ }];
120
+ }
121
+ // Without a shared fixture, absence from context is unknown rather than false.
122
+ return [];
68
123
  });
69
124
  const verifiedCount = claims.filter((c) => c.verified).length;
70
125
  const totalCount = claims.length;
@@ -92,8 +92,47 @@ function anyStringContains(obj, needle) {
92
92
  return false;
93
93
  }
94
94
 
95
+ const BUILTIN_TOOL_ALIASES = {
96
+ bash: 'Bash',
97
+ shell: 'Bash',
98
+ exec_command: 'Bash',
99
+ command_execution: 'Bash',
100
+ read: 'Read',
101
+ file_read: 'Read',
102
+ grep: 'Grep',
103
+ edit: 'Edit',
104
+ apply_patch: 'Edit',
105
+ file_change: 'Edit',
106
+ write: 'Write',
107
+ file_write: 'Write',
108
+ view_image: 'ViewImage',
109
+ viewimage: 'ViewImage',
110
+ write_stdin: 'WriteStdin',
111
+ writestdin: 'WriteStdin',
112
+ web_search: 'WebSearch',
113
+ websearch: 'WebSearch',
114
+ };
115
+
116
+ function canonicalToolName(name) {
117
+ const sourceName = String(name);
118
+ const builtin = BUILTIN_TOOL_ALIASES[sourceName.toLowerCase()];
119
+ if (builtin) return builtin;
120
+ const parts = sourceName.split('__').filter(Boolean);
121
+ if (parts[0] === 'mcp' && parts.length > 2) {
122
+ const providerParts = parts.slice(1, -1);
123
+ if (providerParts[0] === 'codex_apps' && providerParts.length > 1) providerParts.shift();
124
+ return providerParts.join('.') + '.' + parts[parts.length - 1];
125
+ }
126
+ return sourceName;
127
+ }
128
+
129
+ function toolIdentityMatches(expectedName, runtimeName) {
130
+ return expectedName === runtimeName
131
+ || canonicalToolName(expectedName) === canonicalToolName(runtimeName);
132
+ }
133
+
95
134
  function isMockHit(mock, toolName, toolInput) {
96
- if (mock.tool !== '*' && mock.tool !== toolName) return false;
135
+ if (mock.tool !== '*' && !toolIdentityMatches(mock.tool, toolName)) return false;
97
136
  const m = mock.match;
98
137
  if (!m) return true;
99
138
  const ti = toolInput || {};
@@ -18,6 +18,7 @@ import { homedir, tmpdir } from 'node:os';
18
18
  import { dirname, isAbsolute, join, resolve } from 'node:path';
19
19
  import { fileURLToPath } from 'node:url';
20
20
  import { incrementRecordCount, setOwnRecordValue } from '../shared/record-count.js';
21
+ import { toolIdentityMatches } from '../shared/tool-identity.js';
21
22
  // ─── Match logic ────────────────────────────────────────────────────────────
22
23
  function expandHome(p) {
23
24
  if (p.startsWith('~/'))
@@ -112,7 +113,7 @@ function anyStringContains(obj, needle) {
112
113
  }
113
114
  /** 单条 mock 是否命中给定 tool 调用。 */
114
115
  export function isMockHit(mock, toolName, toolInput) {
115
- if (mock.tool !== '*' && mock.tool !== toolName)
116
+ if (mock.tool !== '*' && !toolIdentityMatches(mock.tool, toolName))
116
117
  return false;
117
118
  const m = mock.match;
118
119
  if (!m)
@@ -414,8 +415,44 @@ function anyStringContains(obj, needle) {
414
415
  if (typeof obj === 'object' && obj !== null) return Object.values(obj).some((v) => anyStringContains(v, needle));
415
416
  return false;
416
417
  }
418
+ const BUILTIN_TOOL_ALIASES = {
419
+ bash: 'Bash',
420
+ shell: 'Bash',
421
+ exec_command: 'Bash',
422
+ command_execution: 'Bash',
423
+ read: 'Read',
424
+ file_read: 'Read',
425
+ grep: 'Grep',
426
+ edit: 'Edit',
427
+ apply_patch: 'Edit',
428
+ file_change: 'Edit',
429
+ write: 'Write',
430
+ file_write: 'Write',
431
+ view_image: 'ViewImage',
432
+ viewimage: 'ViewImage',
433
+ write_stdin: 'WriteStdin',
434
+ writestdin: 'WriteStdin',
435
+ web_search: 'WebSearch',
436
+ websearch: 'WebSearch',
437
+ };
438
+ function canonicalToolName(name) {
439
+ const sourceName = String(name);
440
+ const builtin = BUILTIN_TOOL_ALIASES[sourceName.toLowerCase()];
441
+ if (builtin) return builtin;
442
+ const parts = sourceName.split('__').filter(Boolean);
443
+ if (parts[0] === 'mcp' && parts.length > 2) {
444
+ const providerParts = parts.slice(1, -1);
445
+ if (providerParts[0] === 'codex_apps' && providerParts.length > 1) providerParts.shift();
446
+ return providerParts.join('.') + '.' + parts[parts.length - 1];
447
+ }
448
+ return sourceName;
449
+ }
450
+ function toolIdentityMatches(expectedName, runtimeName) {
451
+ return expectedName === runtimeName
452
+ || canonicalToolName(expectedName) === canonicalToolName(runtimeName);
453
+ }
417
454
  function isMockHit(mock, toolName, toolInput) {
418
- if (mock.tool !== '*' && mock.tool !== toolName) return false;
455
+ if (mock.tool !== '*' && !toolIdentityMatches(mock.tool, toolName)) return false;
419
456
  const m = mock.match;
420
457
  if (!m) return true;
421
458
  const ti = toolInput || {};
@@ -1,8 +1,8 @@
1
1
  import type { Artifact, Sample, SampleEnvironment, Task } from '../types/index.js';
2
2
  /**
3
3
  * 把 sample.environment 渲染成自然语言段落,放在用户 prompt 前。
4
- * 让 LLM 读到"环境已就绪",跳过 Glob / find / which / Read 这些环境探测,
5
- * 直接进入 skill 描述的工作流 — 评测信号纯,mock 设计也简化(不用 mock 探测命令)。
4
+ * 这是题设上下文,不会修改 PATH、物化文件或改变 runtime;LLM 可据此跳过
5
+ * which / test -f 等可用性探测,直接进入 skill 描述的工作流。
6
6
  *
7
7
  * 输出 null 表示 sample 没声明 environment,prompt 不变。
8
8
  */
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * 把 sample.environment 渲染成自然语言段落,放在用户 prompt 前。
3
- * 让 LLM 读到"环境已就绪",跳过 Glob / find / which / Read 这些环境探测,
4
- * 直接进入 skill 描述的工作流 — 评测信号纯,mock 设计也简化(不用 mock 探测命令)。
3
+ * 这是题设上下文,不会修改 PATH、物化文件或改变 runtime;LLM 可据此跳过
4
+ * which / test -f 等可用性探测,直接进入 skill 描述的工作流。
5
5
  *
6
6
  * 输出 null 表示 sample 没声明 environment,prompt 不变。
7
7
  */
@@ -10,26 +10,26 @@ export function renderEnvironmentSection(env) {
10
10
  return null;
11
11
  const lines = [];
12
12
  if (env.cli_available && env.cli_available.length > 0) {
13
- lines.push('- 已安装 CLI(已在 PATH,无需 `which` / `command -v` / `type` 探测):');
13
+ lines.push('- 题设声明可用的 CLI(仅作上下文,不修改 PATH):');
14
14
  for (const c of env.cli_available)
15
15
  lines.push(` - \`${c}\``);
16
16
  }
17
17
  if (env.files_available && env.files_available.length > 0) {
18
- lines.push('- 已存在文件(无需 Glob / `Read` / `test -f` 探测):');
18
+ lines.push('- 题设引用的文件路径(仅作上下文,不会在 cwd 物化):');
19
19
  for (const f of env.files_available)
20
20
  lines.push(` - \`${f}\``);
21
21
  }
22
22
  if (env.notes && env.notes.trim()) {
23
- lines.push(`- 备注:${env.notes.trim()}`);
23
+ lines.push(`- 备注:${env.notes.trim()}`);
24
24
  }
25
25
  if (lines.length === 0)
26
26
  return null;
27
27
  return [
28
- '## 评测环境前置(已就绪,无需探测)',
28
+ '## 题设环境声明(仅作上下文)',
29
29
  '',
30
30
  ...lines,
31
31
  '',
32
- '请直接进入 skill 描述的主流程,**不要做环境检查 / `Glob` / `find` / `which` / `test -f` 等探测**。',
32
+ '请按以上题设进入 skill 描述的主流程,**不要额外做 `find` / `which` / `test -f` 等可用性探测**。这些声明不会自动创建文件或修改 runtime 环境。',
33
33
  ].join('\n');
34
34
  }
35
35
  export function buildTasks(samples, variants, skills) {
@@ -28,6 +28,7 @@ import { finalizeSuccessfulRun, initializeEvaluationRunState, persistFailedJob,
28
28
  import { finalizeEvaluationReport } from './evaluation-pipeline/report-finalize.js';
29
29
  import { emitIsolationWarnings, emitPowerWarnings } from './evaluation-pipeline/preflight-warnings.js';
30
30
  import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
31
+ import { assertSamplesCompatibleWithExecutor } from '../executors/capabilities.js';
31
32
  // 兼容 re-export:测试与 run-evaluation.ts 动态 import 仍打 evaluation-pipeline.js
32
33
  export { buildPowerWarnings, buildIsolationWarnings } from './evaluation-pipeline/preflight-warnings.js';
33
34
  export { _computeTestSetHashForTest } from './evaluation-pipeline/test-set-hash.js';
@@ -35,6 +36,7 @@ export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, s
35
36
  // requires 现在由 runEvaluation 上游传给 doctor 处理; eval-pipeline 不再用
36
37
  // 但保留接口字段,避免破 programmatic API (类型层面接收, 内部忽略)
37
38
  requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
39
+ assertSamplesCompatibleWithExecutor(samples, executorName, lang);
38
40
  const variantNames = artifacts.map((artifact) => artifact.name);
39
41
  const runState = await initializeEvaluationRunState({
40
42
  samplesPath,
@@ -1,6 +1,6 @@
1
1
  import { resolve } from 'node:path';
2
2
  import { DEFAULT_OUTPUT_DIR, persistReport } from '../eval-core/evaluation-reporting.js';
3
- import { createExecutor } from '../executors/index.js';
3
+ import { assertSamplesCompatibleWithExecutor, createExecutor, } from '../executors/index.js';
4
4
  import { discoverBatchSkills } from '../inputs/skill-loader.js';
5
5
  import { confidenceInterval, tTest, effectSize } from '../eval-core/statistics.js';
6
6
  import { executeBatchEvaluationRuns, buildBatchVariantSpecs } from './batch-evaluation-workflow.js';
@@ -26,6 +26,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
26
26
  mcpConfig,
27
27
  strictBaseline,
28
28
  });
29
+ assertSamplesCompatibleWithExecutor(samples, executorName, lang);
29
30
  // doctor 强制门禁: skill 静态结构 + 元数据 + 依赖 + 用例契约。
30
31
  // 在 dryRun 分支之前跑, 让 dry-run 也得到 doctor 覆盖(保护 garbage-in 的 verdict)。
31
32
  // 默认强制启用; --skip-doctor 提供 escape hatch,典型场景是评测环境用 mock/stub
@@ -0,0 +1,15 @@
1
+ import type { ExecutorFn, ExecutorInput, Sample } from '../types/index.js';
2
+ export type SampleMockSupport = 'native-hooks' | 'delegated-script' | 'unsupported';
3
+ export interface ExecutorCapabilities {
4
+ sampleMocks: SampleMockSupport;
5
+ }
6
+ /**
7
+ * Custom script executors receive the OMK_MOCK_* protocol environment and own
8
+ * the final adapter. Built-ins are explicit so unsupported runtimes can never
9
+ * silently turn mock assertions into model failures.
10
+ */
11
+ export declare function getExecutorCapabilities(executorName: string): ExecutorCapabilities;
12
+ export declare function executorSupportsSampleMocks(executorName: string): boolean;
13
+ export declare function assertSamplesCompatibleWithExecutor(samples: Sample[], executorName: string, lang?: 'zh' | 'en'): void;
14
+ export declare function assertExecutorInputCapabilities(executorName: string, input: ExecutorInput): void;
15
+ export declare function enforceExecutorCapabilities(executorName: string, executor: ExecutorFn): ExecutorFn;
@@ -0,0 +1,64 @@
1
+ const BUILTIN_CAPABILITIES = {
2
+ claude: { sampleMocks: 'native-hooks' },
3
+ 'claude-sdk': { sampleMocks: 'native-hooks' },
4
+ codex: { sampleMocks: 'unsupported' },
5
+ 'codex-sdk': { sampleMocks: 'unsupported' },
6
+ gemini: { sampleMocks: 'unsupported' },
7
+ 'anthropic-api': { sampleMocks: 'unsupported' },
8
+ 'openai-api': { sampleMocks: 'unsupported' },
9
+ };
10
+ /**
11
+ * Custom script executors receive the OMK_MOCK_* protocol environment and own
12
+ * the final adapter. Built-ins are explicit so unsupported runtimes can never
13
+ * silently turn mock assertions into model failures.
14
+ */
15
+ export function getExecutorCapabilities(executorName) {
16
+ return BUILTIN_CAPABILITIES[executorName]
17
+ ?? { sampleMocks: 'delegated-script' };
18
+ }
19
+ export function executorSupportsSampleMocks(executorName) {
20
+ return getExecutorCapabilities(executorName).sampleMocks !== 'unsupported';
21
+ }
22
+ function unsupportedMocksMessage(executorName, sampleIds, lang) {
23
+ const ids = sampleIds.slice(0, 8).join(', ');
24
+ const overflow = sampleIds.length > 8
25
+ ? lang === 'zh'
26
+ ? ` 等 ${sampleIds.length} 条`
27
+ : ` and ${sampleIds.length - 8} more`
28
+ : '';
29
+ if (lang === 'en') {
30
+ return `Executor "${executorName}" does not support Sample.mocks tool interception. `
31
+ + `Continuing would make mock_hit assertions structurally impossible and create false evidence. `
32
+ + `Affected samples: ${ids}${overflow}. `
33
+ + 'Regenerate them with "omk sample --no-mock", remove mocks/mock_hit, '
34
+ + 'or evaluate with claude/claude-sdk.';
35
+ }
36
+ return `执行器「${executorName}」不支持 Sample.mocks 工具拦截。`
37
+ + `继续运行会让 mock_hit 在结构上必然失败并产生伪证据。`
38
+ + `受影响用例:${ids}${overflow}。`
39
+ + '请用「omk sample --no-mock」重新生成、删除 mocks/mock_hit,'
40
+ + '或改用 claude/claude-sdk 评测。';
41
+ }
42
+ export function assertSamplesCompatibleWithExecutor(samples, executorName, lang = 'zh') {
43
+ if (executorSupportsSampleMocks(executorName))
44
+ return;
45
+ const affected = samples
46
+ .filter((sample) => Array.isArray(sample.mocks) && sample.mocks.length > 0)
47
+ .map((sample) => sample.sample_id);
48
+ if (affected.length === 0)
49
+ return;
50
+ throw new Error(unsupportedMocksMessage(executorName, affected, lang));
51
+ }
52
+ export function assertExecutorInputCapabilities(executorName, input) {
53
+ if (executorSupportsSampleMocks(executorName)
54
+ || !Array.isArray(input.mocks)
55
+ || input.mocks.length === 0)
56
+ return;
57
+ throw new Error(unsupportedMocksMessage(executorName, ['<programmatic-input>'], 'zh'));
58
+ }
59
+ export function enforceExecutorCapabilities(executorName, executor) {
60
+ return async (input) => {
61
+ assertExecutorInputCapabilities(executorName, input);
62
+ return executor(input);
63
+ };
64
+ }
@@ -2,4 +2,5 @@ import type { ExecutorFn } from '../types/index.js';
2
2
  import { extractAgentTrace } from './claude-sdk-trace.js';
3
3
  import { createScriptExecutor } from './script.js';
4
4
  export { extractAgentTrace, createScriptExecutor };
5
+ export { assertExecutorInputCapabilities, assertSamplesCompatibleWithExecutor, executorSupportsSampleMocks, getExecutorCapabilities, } from './capabilities.js';
5
6
  export declare function createExecutor(name: string): ExecutorFn;
@@ -7,6 +7,7 @@ import { codexSdkExecutor } from './codex-sdk.js';
7
7
  import { geminiExecutor } from './gemini.js';
8
8
  import { openAiApiExecutor } from './openai-api.js';
9
9
  import { createScriptExecutor } from './script.js';
10
+ import { enforceExecutorCapabilities } from './capabilities.js';
10
11
  // 命名一致性:provider HTTP 路径统一用 `<vendor>-api`(`anthropic-api` / `openai-api`),
11
12
  // vendor coding agent CLI 用 vendor 名(`claude` / `codex`)。`openai` 这个不带 -api 后缀的
12
13
  // 旧 alias 历史上指 openai-cli 子进程实现,删除后不再设别名 — 用 `--executor openai-api`。
@@ -20,9 +21,11 @@ const EXECUTOR_REGISTRY = {
20
21
  'openai-api': openAiApiExecutor,
21
22
  };
22
23
  export { extractAgentTrace, createScriptExecutor };
24
+ export { assertExecutorInputCapabilities, assertSamplesCompatibleWithExecutor, executorSupportsSampleMocks, getExecutorCapabilities, } from './capabilities.js';
23
25
  export function createExecutor(name) {
24
26
  if (name.trim().length === 0) {
25
27
  throw new Error('executor name or script command is required');
26
28
  }
27
- return EXECUTOR_REGISTRY[name] || createScriptExecutor(name);
29
+ const executor = EXECUTOR_REGISTRY[name] || createScriptExecutor(name);
30
+ return enforceExecutorCapabilities(name, executor);
28
31
  }
@@ -0,0 +1,7 @@
1
+ /** Static extraction for Codex Desktop's JavaScript exec bridge. */
2
+ /**
3
+ * Extract only literal `cmd` values from real `tools.exec_command(...)` calls.
4
+ * The bridge source is never evaluated, and examples inside strings/comments
5
+ * remain ignored.
6
+ */
7
+ export declare function extractCodexExecCommands(source: string): string[];
@@ -0,0 +1,134 @@
1
+ /** Static extraction for Codex Desktop's JavaScript exec bridge. */
2
+ function skipJsString(source, start) {
3
+ const quote = source[start];
4
+ let index = start + 1;
5
+ while (index < source.length) {
6
+ if (source[index] === '\\') {
7
+ index += 2;
8
+ continue;
9
+ }
10
+ if (source[index] === quote)
11
+ return index + 1;
12
+ index += 1;
13
+ }
14
+ return source.length;
15
+ }
16
+ function skipJsTrivia(source, start) {
17
+ let index = start;
18
+ while (index < source.length) {
19
+ if (/\s/.test(source[index])) {
20
+ index += 1;
21
+ continue;
22
+ }
23
+ if (source.startsWith('//', index)) {
24
+ const newline = source.indexOf('\n', index + 2);
25
+ return newline < 0 ? source.length : skipJsTrivia(source, newline + 1);
26
+ }
27
+ if (source.startsWith('/*', index)) {
28
+ const end = source.indexOf('*/', index + 2);
29
+ return end < 0 ? source.length : skipJsTrivia(source, end + 2);
30
+ }
31
+ break;
32
+ }
33
+ return index;
34
+ }
35
+ function commandPropertyEnd(source, index) {
36
+ const bareKey = source.startsWith('cmd', index)
37
+ && !/[\w$]/.test(source[index - 1] ?? '')
38
+ && !/[\w$]/.test(source[index + 3] ?? '');
39
+ if (bareKey)
40
+ return index + 3;
41
+ const quote = source[index];
42
+ if (quote !== '"' && quote !== "'")
43
+ return null;
44
+ const end = skipJsString(source, index);
45
+ return source.slice(index + 1, end - 1) === 'cmd' ? end : null;
46
+ }
47
+ function commandLiteralValue(source, start, end) {
48
+ const literal = source.slice(start, end);
49
+ if (source[start] === '"') {
50
+ try {
51
+ const parsed = JSON.parse(literal);
52
+ if (typeof parsed === 'string')
53
+ return parsed;
54
+ }
55
+ catch {
56
+ // Fall through to the lossless source slice for non-JSON JS escapes.
57
+ }
58
+ }
59
+ return source.slice(start + 1, Math.max(start + 1, end - 1));
60
+ }
61
+ function extractExecCommandLiteral(source, callStart) {
62
+ let index = skipJsTrivia(source, callStart + 'tools.exec_command'.length);
63
+ if (source[index] !== '(')
64
+ return null;
65
+ index = skipJsTrivia(source, index + 1);
66
+ if (source[index] !== '{')
67
+ return null;
68
+ let depth = 1;
69
+ index += 1;
70
+ while (index < source.length && depth > 0) {
71
+ index = skipJsTrivia(source, index);
72
+ const char = source[index];
73
+ if (depth === 1) {
74
+ const keyEnd = commandPropertyEnd(source, index);
75
+ if (keyEnd !== null) {
76
+ let valueStart = skipJsTrivia(source, keyEnd);
77
+ if (source[valueStart] !== ':') {
78
+ index = keyEnd;
79
+ continue;
80
+ }
81
+ valueStart = skipJsTrivia(source, valueStart + 1);
82
+ const quote = source[valueStart];
83
+ if (quote !== '"' && quote !== "'" && quote !== '`')
84
+ return null;
85
+ const end = skipJsString(source, valueStart);
86
+ return commandLiteralValue(source, valueStart, end);
87
+ }
88
+ }
89
+ if (char === '"' || char === "'" || char === '`') {
90
+ index = skipJsString(source, index);
91
+ continue;
92
+ }
93
+ if (char === '{') {
94
+ depth += 1;
95
+ index += 1;
96
+ continue;
97
+ }
98
+ if (char === '}') {
99
+ depth -= 1;
100
+ index += 1;
101
+ continue;
102
+ }
103
+ index += 1;
104
+ }
105
+ return null;
106
+ }
107
+ /**
108
+ * Extract only literal `cmd` values from real `tools.exec_command(...)` calls.
109
+ * The bridge source is never evaluated, and examples inside strings/comments
110
+ * remain ignored.
111
+ */
112
+ export function extractCodexExecCommands(source) {
113
+ const commands = [];
114
+ let index = 0;
115
+ while (index < source.length) {
116
+ index = skipJsTrivia(source, index);
117
+ const char = source[index];
118
+ if (char === '"' || char === "'" || char === '`') {
119
+ index = skipJsString(source, index);
120
+ continue;
121
+ }
122
+ if (source.startsWith('tools.exec_command', index)
123
+ && !/[\w$.]/.test(source[index - 1] ?? '')
124
+ && !/[\w$]/.test(source[index + 'tools.exec_command'.length] ?? '')) {
125
+ const command = extractExecCommandLiteral(source, index);
126
+ if (command)
127
+ commands.push(command);
128
+ index += 'tools.exec_command'.length;
129
+ continue;
130
+ }
131
+ index += 1;
132
+ }
133
+ return commands;
134
+ }
@@ -4,6 +4,7 @@ import { isToolResultFailureText } from './text-signals.js';
4
4
  import { correlateTraceToolEvents, createTraceId, normalizeTraceTimestamp, traceTimestampBounds, } from './trace-ir.js';
5
5
  import { nonNegativeMetric, optionalTokenCount, splitInclusiveInputTokens, tokenCount, } from '../shared/token-usage.js';
6
6
  import { normalizeToolIdentity } from '../shared/tool-identity.js';
7
+ import { extractCodexExecCommands } from './codex-exec-command.js';
7
8
  const CODEX_RESPONSE_ITEM_TYPES = new Set([
8
9
  'message',
9
10
  'reasoning',
@@ -242,8 +243,10 @@ function convertCodexRecords(rawRecords, runId) {
242
243
  const explicitStatus = runtimeOutcome.present
243
244
  ? runtimeOutcome.status
244
245
  : statusFromCodex(payloadStatus);
245
- const inferredFailure = isToolResultFailureText(output) || codexToolOutputFailed(output);
246
- const inferredSuccess = !inferredFailure && codexToolOutputSucceeded(output);
246
+ const bridgeFailure = codexToolOutputFailed(output);
247
+ const bridgeSuccess = !bridgeFailure && codexToolOutputSucceeded(output);
248
+ const inferredFailure = bridgeFailure || (!bridgeSuccess && isToolResultFailureText(output));
249
+ const inferredSuccess = bridgeSuccess;
247
250
  const completedExternalCall = externalEnds.byOccurrence.has(mcpCallOccurrenceKey(callId, externalOccurrence));
248
251
  const inferredStatus = inferredFailure
249
252
  ? 'failure'
@@ -609,6 +612,10 @@ function runtimeToolEndOutcome(end) {
609
612
  }
610
613
  function normalizeCodexTool(sourceName, rawInput, mcpEnd, sourceNamespace) {
611
614
  const input = parseToolInput(rawInput);
615
+ const sourceInput = stringValue(input.input);
616
+ const execCommands = sourceName.toLowerCase() === 'exec' && sourceInput
617
+ ? extractCodexExecCommands(sourceInput)
618
+ : [];
612
619
  // Codex desktop's orchestration wrapper names its JavaScript command bridge
613
620
  // `exec`. This mapping is source-specific: a generic tool named `exec` must not
614
621
  // become shell execution outside the Codex adapter.
@@ -629,10 +636,13 @@ function normalizeCodexTool(sourceName, rawInput, mcpEnd, sourceNamespace) {
629
636
  tool,
630
637
  input: {
631
638
  ...input,
632
- command: stringValue(input.command)
633
- ?? stringValue(input.cmd)
634
- ?? stringValue(input.input)
635
- ?? '',
639
+ command: execCommands.length > 0
640
+ ? execCommands.join('\n')
641
+ : stringValue(input.command)
642
+ ?? stringValue(input.cmd)
643
+ ?? sourceInput
644
+ ?? '',
645
+ ...(execCommands.length > 0 ? { commands: execCommands } : {}),
636
646
  },
637
647
  };
638
648
  }