oh-my-knowledge 0.51.0 → 0.51.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/agent-skills/omk/SKILL.md +2 -0
- package/dist/assets/agent-skills/omk/references/commands.md +2 -2
- package/dist/authoring/generator.d.ts +15 -5
- package/dist/authoring/generator.js +181 -18
- package/dist/authoring/sample-fixer.d.ts +2 -0
- package/dist/authoring/sample-fixer.js +19 -5
- package/dist/cli/commands/sample.js +11 -4
- package/dist/eval-core/dependency-checker.js +30 -18
- package/dist/eval-core/evaluation-execution.js +11 -2
- package/dist/eval-core/fact-checker.d.ts +15 -2
- package/dist/eval-core/fact-checker.js +75 -20
- package/dist/eval-core/mock-hook.cjs +40 -1
- package/dist/eval-core/mocks-runtime.js +39 -2
- package/dist/eval-core/task-planner.d.ts +2 -2
- package/dist/eval-core/task-planner.js +7 -7
- package/dist/eval-workflows/evaluation-pipeline.js +2 -0
- package/dist/eval-workflows/run-evaluation.js +2 -1
- package/dist/executors/capabilities.d.ts +15 -0
- package/dist/executors/capabilities.js +64 -0
- package/dist/executors/index.d.ts +1 -0
- package/dist/executors/index.js +4 -1
- package/dist/observability/codex-exec-command.d.ts +7 -0
- package/dist/observability/codex-exec-command.js +134 -0
- package/dist/observability/codex-trace-adapter.js +16 -6
- package/dist/observability/trace-attribution.js +11 -107
- package/dist/server/skill-insights.js +1 -1
- package/dist/shared/sample-contract.d.ts +1 -0
- package/dist/shared/sample-contract.js +35 -0
- package/dist/shared/tool-identity.d.ts +8 -0
- package/dist/shared/tool-identity.js +13 -0
- package/dist/types/eval.d.ts +7 -6
- package/dist/types/executor.d.ts +3 -2
- package/package.json +4 -4
|
@@ -22,6 +22,10 @@ const IGNORE_PATTERNS = [
|
|
|
22
22
|
/^\.git\//,
|
|
23
23
|
/^index\.\w+$/,
|
|
24
24
|
];
|
|
25
|
+
// Product/runtime names that happen to end in a supported source extension.
|
|
26
|
+
const NON_PATH_REFERENCES = new Set([
|
|
27
|
+
'node.js',
|
|
28
|
+
]);
|
|
25
29
|
/**
|
|
26
30
|
* Extract file path claims from agent output text.
|
|
27
31
|
*/
|
|
@@ -36,35 +40,86 @@ export function extractPathClaims(output) {
|
|
|
36
40
|
path = path.replace(/[.,;:!?))]+$/, '');
|
|
37
41
|
// Clean trailing backtick/quote
|
|
38
42
|
path = path.replace(/[`'"]+$/, '');
|
|
39
|
-
if (path.length > 3
|
|
43
|
+
if (path.length > 3
|
|
44
|
+
&& !NON_PATH_REFERENCES.has(path.toLowerCase())
|
|
45
|
+
&& !IGNORE_PATTERNS.some((p) => p.test(path))) {
|
|
40
46
|
paths.add(path);
|
|
41
47
|
}
|
|
42
48
|
}
|
|
43
49
|
}
|
|
44
50
|
return [...paths];
|
|
45
51
|
}
|
|
52
|
+
function normalizeEvidencePath(path) {
|
|
53
|
+
return path
|
|
54
|
+
.trim()
|
|
55
|
+
.replace(/^['"`]+|['"`]+$/g, '')
|
|
56
|
+
.replace(/\\/g, '/')
|
|
57
|
+
.replace(/^(?:\$SKILL_DIR|~)\//, '')
|
|
58
|
+
.replace(/^\.\//, '')
|
|
59
|
+
.replace(/\/+$/, '');
|
|
60
|
+
}
|
|
61
|
+
function isRootedEvidencePath(path) {
|
|
62
|
+
return path.startsWith('/') || /^[a-zA-Z]:\//.test(path);
|
|
63
|
+
}
|
|
64
|
+
function refersToSamePath(left, right) {
|
|
65
|
+
const a = normalizeEvidencePath(left);
|
|
66
|
+
const b = normalizeEvidencePath(right);
|
|
67
|
+
return a === b
|
|
68
|
+
|| (isRootedEvidencePath(a) && !isRootedEvidencePath(b) && a.endsWith(`/${b}`))
|
|
69
|
+
|| (isRootedEvidencePath(b) && !isRootedEvidencePath(a) && b.endsWith(`/${a}`));
|
|
70
|
+
}
|
|
46
71
|
/**
|
|
47
|
-
* Check facts
|
|
72
|
+
* Check file-path facts against sample-level evidence shared by every arm.
|
|
73
|
+
*
|
|
74
|
+
* A string keeps the legacy direct-filesystem API for callers outside the
|
|
75
|
+
* evaluation pipeline. The pipeline passes structured evidence and deliberately
|
|
76
|
+
* excludes each artifact's execution cwd, because that directory is not a
|
|
77
|
+
* comparable source of truth across control and treatment arms.
|
|
48
78
|
*/
|
|
49
|
-
export function checkFacts(output,
|
|
79
|
+
export function checkFacts(output, evidence) {
|
|
50
80
|
const pathClaims = extractPathClaims(output);
|
|
51
|
-
const
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
:
|
|
66
|
-
|
|
67
|
-
}
|
|
81
|
+
const sources = typeof evidence === 'string'
|
|
82
|
+
? { cwd: evidence }
|
|
83
|
+
: evidence;
|
|
84
|
+
const root = sources.cwd ? resolve(sources.cwd) : null;
|
|
85
|
+
const contextClaims = sources.context
|
|
86
|
+
? extractPathClaims(sources.context)
|
|
87
|
+
: [];
|
|
88
|
+
const declaredFiles = sources.declaredFiles ?? [];
|
|
89
|
+
const claims = pathClaims.flatMap((path) => {
|
|
90
|
+
if (declaredFiles.some((declared) => refersToSamePath(path, declared))) {
|
|
91
|
+
return [{
|
|
92
|
+
type: 'file-path',
|
|
93
|
+
value: path,
|
|
94
|
+
verified: true,
|
|
95
|
+
evidence: 'source=context(sample.environment.files_available)',
|
|
96
|
+
}];
|
|
97
|
+
}
|
|
98
|
+
if (root) {
|
|
99
|
+
const fullPath = resolve(root, path);
|
|
100
|
+
const relativePath = relative(root, fullPath);
|
|
101
|
+
const insideCwd = relativePath === ''
|
|
102
|
+
|| (!relativePath.startsWith('..') && !isAbsolute(relativePath));
|
|
103
|
+
const exists = insideCwd && existsSync(fullPath);
|
|
104
|
+
return [{
|
|
105
|
+
type: 'file-path',
|
|
106
|
+
value: path,
|
|
107
|
+
verified: exists,
|
|
108
|
+
evidence: insideCwd
|
|
109
|
+
? `source=runtime-filesystem; ${fullPath} ${exists ? 'exists' : 'not found'}`
|
|
110
|
+
: `source=runtime-filesystem; ${path} is outside the evaluation cwd`,
|
|
111
|
+
}];
|
|
112
|
+
}
|
|
113
|
+
if (contextClaims.some((contextPath) => refersToSamePath(path, contextPath))) {
|
|
114
|
+
return [{
|
|
115
|
+
type: 'file-path',
|
|
116
|
+
value: path,
|
|
117
|
+
verified: true,
|
|
118
|
+
evidence: 'source=context(sample.context)',
|
|
119
|
+
}];
|
|
120
|
+
}
|
|
121
|
+
// Without a shared fixture, absence from context is unknown rather than false.
|
|
122
|
+
return [];
|
|
68
123
|
});
|
|
69
124
|
const verifiedCount = claims.filter((c) => c.verified).length;
|
|
70
125
|
const totalCount = claims.length;
|
|
@@ -92,8 +92,47 @@ function anyStringContains(obj, needle) {
|
|
|
92
92
|
return false;
|
|
93
93
|
}
|
|
94
94
|
|
|
95
|
+
const BUILTIN_TOOL_ALIASES = {
|
|
96
|
+
bash: 'Bash',
|
|
97
|
+
shell: 'Bash',
|
|
98
|
+
exec_command: 'Bash',
|
|
99
|
+
command_execution: 'Bash',
|
|
100
|
+
read: 'Read',
|
|
101
|
+
file_read: 'Read',
|
|
102
|
+
grep: 'Grep',
|
|
103
|
+
edit: 'Edit',
|
|
104
|
+
apply_patch: 'Edit',
|
|
105
|
+
file_change: 'Edit',
|
|
106
|
+
write: 'Write',
|
|
107
|
+
file_write: 'Write',
|
|
108
|
+
view_image: 'ViewImage',
|
|
109
|
+
viewimage: 'ViewImage',
|
|
110
|
+
write_stdin: 'WriteStdin',
|
|
111
|
+
writestdin: 'WriteStdin',
|
|
112
|
+
web_search: 'WebSearch',
|
|
113
|
+
websearch: 'WebSearch',
|
|
114
|
+
};
|
|
115
|
+
|
|
116
|
+
function canonicalToolName(name) {
|
|
117
|
+
const sourceName = String(name);
|
|
118
|
+
const builtin = BUILTIN_TOOL_ALIASES[sourceName.toLowerCase()];
|
|
119
|
+
if (builtin) return builtin;
|
|
120
|
+
const parts = sourceName.split('__').filter(Boolean);
|
|
121
|
+
if (parts[0] === 'mcp' && parts.length > 2) {
|
|
122
|
+
const providerParts = parts.slice(1, -1);
|
|
123
|
+
if (providerParts[0] === 'codex_apps' && providerParts.length > 1) providerParts.shift();
|
|
124
|
+
return providerParts.join('.') + '.' + parts[parts.length - 1];
|
|
125
|
+
}
|
|
126
|
+
return sourceName;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
function toolIdentityMatches(expectedName, runtimeName) {
|
|
130
|
+
return expectedName === runtimeName
|
|
131
|
+
|| canonicalToolName(expectedName) === canonicalToolName(runtimeName);
|
|
132
|
+
}
|
|
133
|
+
|
|
95
134
|
function isMockHit(mock, toolName, toolInput) {
|
|
96
|
-
if (mock.tool !== '*' && mock.tool
|
|
135
|
+
if (mock.tool !== '*' && !toolIdentityMatches(mock.tool, toolName)) return false;
|
|
97
136
|
const m = mock.match;
|
|
98
137
|
if (!m) return true;
|
|
99
138
|
const ti = toolInput || {};
|
|
@@ -18,6 +18,7 @@ import { homedir, tmpdir } from 'node:os';
|
|
|
18
18
|
import { dirname, isAbsolute, join, resolve } from 'node:path';
|
|
19
19
|
import { fileURLToPath } from 'node:url';
|
|
20
20
|
import { incrementRecordCount, setOwnRecordValue } from '../shared/record-count.js';
|
|
21
|
+
import { toolIdentityMatches } from '../shared/tool-identity.js';
|
|
21
22
|
// ─── Match logic ────────────────────────────────────────────────────────────
|
|
22
23
|
function expandHome(p) {
|
|
23
24
|
if (p.startsWith('~/'))
|
|
@@ -112,7 +113,7 @@ function anyStringContains(obj, needle) {
|
|
|
112
113
|
}
|
|
113
114
|
/** 单条 mock 是否命中给定 tool 调用。 */
|
|
114
115
|
export function isMockHit(mock, toolName, toolInput) {
|
|
115
|
-
if (mock.tool !== '*' && mock.tool
|
|
116
|
+
if (mock.tool !== '*' && !toolIdentityMatches(mock.tool, toolName))
|
|
116
117
|
return false;
|
|
117
118
|
const m = mock.match;
|
|
118
119
|
if (!m)
|
|
@@ -414,8 +415,44 @@ function anyStringContains(obj, needle) {
|
|
|
414
415
|
if (typeof obj === 'object' && obj !== null) return Object.values(obj).some((v) => anyStringContains(v, needle));
|
|
415
416
|
return false;
|
|
416
417
|
}
|
|
418
|
+
const BUILTIN_TOOL_ALIASES = {
|
|
419
|
+
bash: 'Bash',
|
|
420
|
+
shell: 'Bash',
|
|
421
|
+
exec_command: 'Bash',
|
|
422
|
+
command_execution: 'Bash',
|
|
423
|
+
read: 'Read',
|
|
424
|
+
file_read: 'Read',
|
|
425
|
+
grep: 'Grep',
|
|
426
|
+
edit: 'Edit',
|
|
427
|
+
apply_patch: 'Edit',
|
|
428
|
+
file_change: 'Edit',
|
|
429
|
+
write: 'Write',
|
|
430
|
+
file_write: 'Write',
|
|
431
|
+
view_image: 'ViewImage',
|
|
432
|
+
viewimage: 'ViewImage',
|
|
433
|
+
write_stdin: 'WriteStdin',
|
|
434
|
+
writestdin: 'WriteStdin',
|
|
435
|
+
web_search: 'WebSearch',
|
|
436
|
+
websearch: 'WebSearch',
|
|
437
|
+
};
|
|
438
|
+
function canonicalToolName(name) {
|
|
439
|
+
const sourceName = String(name);
|
|
440
|
+
const builtin = BUILTIN_TOOL_ALIASES[sourceName.toLowerCase()];
|
|
441
|
+
if (builtin) return builtin;
|
|
442
|
+
const parts = sourceName.split('__').filter(Boolean);
|
|
443
|
+
if (parts[0] === 'mcp' && parts.length > 2) {
|
|
444
|
+
const providerParts = parts.slice(1, -1);
|
|
445
|
+
if (providerParts[0] === 'codex_apps' && providerParts.length > 1) providerParts.shift();
|
|
446
|
+
return providerParts.join('.') + '.' + parts[parts.length - 1];
|
|
447
|
+
}
|
|
448
|
+
return sourceName;
|
|
449
|
+
}
|
|
450
|
+
function toolIdentityMatches(expectedName, runtimeName) {
|
|
451
|
+
return expectedName === runtimeName
|
|
452
|
+
|| canonicalToolName(expectedName) === canonicalToolName(runtimeName);
|
|
453
|
+
}
|
|
417
454
|
function isMockHit(mock, toolName, toolInput) {
|
|
418
|
-
if (mock.tool !== '*' && mock.tool
|
|
455
|
+
if (mock.tool !== '*' && !toolIdentityMatches(mock.tool, toolName)) return false;
|
|
419
456
|
const m = mock.match;
|
|
420
457
|
if (!m) return true;
|
|
421
458
|
const ti = toolInput || {};
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import type { Artifact, Sample, SampleEnvironment, Task } from '../types/index.js';
|
|
2
2
|
/**
|
|
3
3
|
* 把 sample.environment 渲染成自然语言段落,放在用户 prompt 前。
|
|
4
|
-
*
|
|
5
|
-
*
|
|
4
|
+
* 这是题设上下文,不会修改 PATH、物化文件或改变 runtime;LLM 可据此跳过
|
|
5
|
+
* which / test -f 等可用性探测,直接进入 skill 描述的工作流。
|
|
6
6
|
*
|
|
7
7
|
* 输出 null 表示 sample 没声明 environment,prompt 不变。
|
|
8
8
|
*/
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* 把 sample.environment 渲染成自然语言段落,放在用户 prompt 前。
|
|
3
|
-
*
|
|
4
|
-
*
|
|
3
|
+
* 这是题设上下文,不会修改 PATH、物化文件或改变 runtime;LLM 可据此跳过
|
|
4
|
+
* which / test -f 等可用性探测,直接进入 skill 描述的工作流。
|
|
5
5
|
*
|
|
6
6
|
* 输出 null 表示 sample 没声明 environment,prompt 不变。
|
|
7
7
|
*/
|
|
@@ -10,26 +10,26 @@ export function renderEnvironmentSection(env) {
|
|
|
10
10
|
return null;
|
|
11
11
|
const lines = [];
|
|
12
12
|
if (env.cli_available && env.cli_available.length > 0) {
|
|
13
|
-
lines.push('-
|
|
13
|
+
lines.push('- 题设声明可用的 CLI(仅作上下文,不修改 PATH):');
|
|
14
14
|
for (const c of env.cli_available)
|
|
15
15
|
lines.push(` - \`${c}\``);
|
|
16
16
|
}
|
|
17
17
|
if (env.files_available && env.files_available.length > 0) {
|
|
18
|
-
lines.push('-
|
|
18
|
+
lines.push('- 题设引用的文件路径(仅作上下文,不会在 cwd 物化):');
|
|
19
19
|
for (const f of env.files_available)
|
|
20
20
|
lines.push(` - \`${f}\``);
|
|
21
21
|
}
|
|
22
22
|
if (env.notes && env.notes.trim()) {
|
|
23
|
-
lines.push(`-
|
|
23
|
+
lines.push(`- 备注:${env.notes.trim()}`);
|
|
24
24
|
}
|
|
25
25
|
if (lines.length === 0)
|
|
26
26
|
return null;
|
|
27
27
|
return [
|
|
28
|
-
'##
|
|
28
|
+
'## 题设环境声明(仅作上下文)',
|
|
29
29
|
'',
|
|
30
30
|
...lines,
|
|
31
31
|
'',
|
|
32
|
-
'
|
|
32
|
+
'请按以上题设进入 skill 描述的主流程,**不要额外做 `find` / `which` / `test -f` 等可用性探测**。这些声明不会自动创建文件或修改 runtime 环境。',
|
|
33
33
|
].join('\n');
|
|
34
34
|
}
|
|
35
35
|
export function buildTasks(samples, variants, skills) {
|
|
@@ -28,6 +28,7 @@ import { finalizeSuccessfulRun, initializeEvaluationRunState, persistFailedJob,
|
|
|
28
28
|
import { finalizeEvaluationReport } from './evaluation-pipeline/report-finalize.js';
|
|
29
29
|
import { emitIsolationWarnings, emitPowerWarnings } from './evaluation-pipeline/preflight-warnings.js';
|
|
30
30
|
import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
|
|
31
|
+
import { assertSamplesCompatibleWithExecutor } from '../executors/capabilities.js';
|
|
31
32
|
// 兼容 re-export:测试与 run-evaluation.ts 动态 import 仍打 evaluation-pipeline.js
|
|
32
33
|
export { buildPowerWarnings, buildIsolationWarnings } from './evaluation-pipeline/preflight-warnings.js';
|
|
33
34
|
export { _computeTestSetHashForTest } from './evaluation-pipeline/test-set-hash.js';
|
|
@@ -35,6 +36,7 @@ export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, s
|
|
|
35
36
|
// requires 现在由 runEvaluation 上游传给 doctor 处理; eval-pipeline 不再用
|
|
36
37
|
// 但保留接口字段,避免破 programmatic API (类型层面接收, 内部忽略)
|
|
37
38
|
requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
|
|
39
|
+
assertSamplesCompatibleWithExecutor(samples, executorName, lang);
|
|
38
40
|
const variantNames = artifacts.map((artifact) => artifact.name);
|
|
39
41
|
const runState = await initializeEvaluationRunState({
|
|
40
42
|
samplesPath,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { resolve } from 'node:path';
|
|
2
2
|
import { DEFAULT_OUTPUT_DIR, persistReport } from '../eval-core/evaluation-reporting.js';
|
|
3
|
-
import { createExecutor } from '../executors/index.js';
|
|
3
|
+
import { assertSamplesCompatibleWithExecutor, createExecutor, } from '../executors/index.js';
|
|
4
4
|
import { discoverBatchSkills } from '../inputs/skill-loader.js';
|
|
5
5
|
import { confidenceInterval, tTest, effectSize } from '../eval-core/statistics.js';
|
|
6
6
|
import { executeBatchEvaluationRuns, buildBatchVariantSpecs } from './batch-evaluation-workflow.js';
|
|
@@ -26,6 +26,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
26
26
|
mcpConfig,
|
|
27
27
|
strictBaseline,
|
|
28
28
|
});
|
|
29
|
+
assertSamplesCompatibleWithExecutor(samples, executorName, lang);
|
|
29
30
|
// doctor 强制门禁: skill 静态结构 + 元数据 + 依赖 + 用例契约。
|
|
30
31
|
// 在 dryRun 分支之前跑, 让 dry-run 也得到 doctor 覆盖(保护 garbage-in 的 verdict)。
|
|
31
32
|
// 默认强制启用; --skip-doctor 提供 escape hatch,典型场景是评测环境用 mock/stub
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import type { ExecutorFn, ExecutorInput, Sample } from '../types/index.js';
|
|
2
|
+
export type SampleMockSupport = 'native-hooks' | 'delegated-script' | 'unsupported';
|
|
3
|
+
export interface ExecutorCapabilities {
|
|
4
|
+
sampleMocks: SampleMockSupport;
|
|
5
|
+
}
|
|
6
|
+
/**
|
|
7
|
+
* Custom script executors receive the OMK_MOCK_* protocol environment and own
|
|
8
|
+
* the final adapter. Built-ins are explicit so unsupported runtimes can never
|
|
9
|
+
* silently turn mock assertions into model failures.
|
|
10
|
+
*/
|
|
11
|
+
export declare function getExecutorCapabilities(executorName: string): ExecutorCapabilities;
|
|
12
|
+
export declare function executorSupportsSampleMocks(executorName: string): boolean;
|
|
13
|
+
export declare function assertSamplesCompatibleWithExecutor(samples: Sample[], executorName: string, lang?: 'zh' | 'en'): void;
|
|
14
|
+
export declare function assertExecutorInputCapabilities(executorName: string, input: ExecutorInput): void;
|
|
15
|
+
export declare function enforceExecutorCapabilities(executorName: string, executor: ExecutorFn): ExecutorFn;
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
const BUILTIN_CAPABILITIES = {
|
|
2
|
+
claude: { sampleMocks: 'native-hooks' },
|
|
3
|
+
'claude-sdk': { sampleMocks: 'native-hooks' },
|
|
4
|
+
codex: { sampleMocks: 'unsupported' },
|
|
5
|
+
'codex-sdk': { sampleMocks: 'unsupported' },
|
|
6
|
+
gemini: { sampleMocks: 'unsupported' },
|
|
7
|
+
'anthropic-api': { sampleMocks: 'unsupported' },
|
|
8
|
+
'openai-api': { sampleMocks: 'unsupported' },
|
|
9
|
+
};
|
|
10
|
+
/**
|
|
11
|
+
* Custom script executors receive the OMK_MOCK_* protocol environment and own
|
|
12
|
+
* the final adapter. Built-ins are explicit so unsupported runtimes can never
|
|
13
|
+
* silently turn mock assertions into model failures.
|
|
14
|
+
*/
|
|
15
|
+
export function getExecutorCapabilities(executorName) {
|
|
16
|
+
return BUILTIN_CAPABILITIES[executorName]
|
|
17
|
+
?? { sampleMocks: 'delegated-script' };
|
|
18
|
+
}
|
|
19
|
+
export function executorSupportsSampleMocks(executorName) {
|
|
20
|
+
return getExecutorCapabilities(executorName).sampleMocks !== 'unsupported';
|
|
21
|
+
}
|
|
22
|
+
function unsupportedMocksMessage(executorName, sampleIds, lang) {
|
|
23
|
+
const ids = sampleIds.slice(0, 8).join(', ');
|
|
24
|
+
const overflow = sampleIds.length > 8
|
|
25
|
+
? lang === 'zh'
|
|
26
|
+
? ` 等 ${sampleIds.length} 条`
|
|
27
|
+
: ` and ${sampleIds.length - 8} more`
|
|
28
|
+
: '';
|
|
29
|
+
if (lang === 'en') {
|
|
30
|
+
return `Executor "${executorName}" does not support Sample.mocks tool interception. `
|
|
31
|
+
+ `Continuing would make mock_hit assertions structurally impossible and create false evidence. `
|
|
32
|
+
+ `Affected samples: ${ids}${overflow}. `
|
|
33
|
+
+ 'Regenerate them with "omk sample --no-mock", remove mocks/mock_hit, '
|
|
34
|
+
+ 'or evaluate with claude/claude-sdk.';
|
|
35
|
+
}
|
|
36
|
+
return `执行器「${executorName}」不支持 Sample.mocks 工具拦截。`
|
|
37
|
+
+ `继续运行会让 mock_hit 在结构上必然失败并产生伪证据。`
|
|
38
|
+
+ `受影响用例:${ids}${overflow}。`
|
|
39
|
+
+ '请用「omk sample --no-mock」重新生成、删除 mocks/mock_hit,'
|
|
40
|
+
+ '或改用 claude/claude-sdk 评测。';
|
|
41
|
+
}
|
|
42
|
+
export function assertSamplesCompatibleWithExecutor(samples, executorName, lang = 'zh') {
|
|
43
|
+
if (executorSupportsSampleMocks(executorName))
|
|
44
|
+
return;
|
|
45
|
+
const affected = samples
|
|
46
|
+
.filter((sample) => Array.isArray(sample.mocks) && sample.mocks.length > 0)
|
|
47
|
+
.map((sample) => sample.sample_id);
|
|
48
|
+
if (affected.length === 0)
|
|
49
|
+
return;
|
|
50
|
+
throw new Error(unsupportedMocksMessage(executorName, affected, lang));
|
|
51
|
+
}
|
|
52
|
+
export function assertExecutorInputCapabilities(executorName, input) {
|
|
53
|
+
if (executorSupportsSampleMocks(executorName)
|
|
54
|
+
|| !Array.isArray(input.mocks)
|
|
55
|
+
|| input.mocks.length === 0)
|
|
56
|
+
return;
|
|
57
|
+
throw new Error(unsupportedMocksMessage(executorName, ['<programmatic-input>'], 'zh'));
|
|
58
|
+
}
|
|
59
|
+
export function enforceExecutorCapabilities(executorName, executor) {
|
|
60
|
+
return async (input) => {
|
|
61
|
+
assertExecutorInputCapabilities(executorName, input);
|
|
62
|
+
return executor(input);
|
|
63
|
+
};
|
|
64
|
+
}
|
|
@@ -2,4 +2,5 @@ import type { ExecutorFn } from '../types/index.js';
|
|
|
2
2
|
import { extractAgentTrace } from './claude-sdk-trace.js';
|
|
3
3
|
import { createScriptExecutor } from './script.js';
|
|
4
4
|
export { extractAgentTrace, createScriptExecutor };
|
|
5
|
+
export { assertExecutorInputCapabilities, assertSamplesCompatibleWithExecutor, executorSupportsSampleMocks, getExecutorCapabilities, } from './capabilities.js';
|
|
5
6
|
export declare function createExecutor(name: string): ExecutorFn;
|
package/dist/executors/index.js
CHANGED
|
@@ -7,6 +7,7 @@ import { codexSdkExecutor } from './codex-sdk.js';
|
|
|
7
7
|
import { geminiExecutor } from './gemini.js';
|
|
8
8
|
import { openAiApiExecutor } from './openai-api.js';
|
|
9
9
|
import { createScriptExecutor } from './script.js';
|
|
10
|
+
import { enforceExecutorCapabilities } from './capabilities.js';
|
|
10
11
|
// 命名一致性:provider HTTP 路径统一用 `<vendor>-api`(`anthropic-api` / `openai-api`),
|
|
11
12
|
// vendor coding agent CLI 用 vendor 名(`claude` / `codex`)。`openai` 这个不带 -api 后缀的
|
|
12
13
|
// 旧 alias 历史上指 openai-cli 子进程实现,删除后不再设别名 — 用 `--executor openai-api`。
|
|
@@ -20,9 +21,11 @@ const EXECUTOR_REGISTRY = {
|
|
|
20
21
|
'openai-api': openAiApiExecutor,
|
|
21
22
|
};
|
|
22
23
|
export { extractAgentTrace, createScriptExecutor };
|
|
24
|
+
export { assertExecutorInputCapabilities, assertSamplesCompatibleWithExecutor, executorSupportsSampleMocks, getExecutorCapabilities, } from './capabilities.js';
|
|
23
25
|
export function createExecutor(name) {
|
|
24
26
|
if (name.trim().length === 0) {
|
|
25
27
|
throw new Error('executor name or script command is required');
|
|
26
28
|
}
|
|
27
|
-
|
|
29
|
+
const executor = EXECUTOR_REGISTRY[name] || createScriptExecutor(name);
|
|
30
|
+
return enforceExecutorCapabilities(name, executor);
|
|
28
31
|
}
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
/** Static extraction for Codex Desktop's JavaScript exec bridge. */
|
|
2
|
+
/**
|
|
3
|
+
* Extract only literal `cmd` values from real `tools.exec_command(...)` calls.
|
|
4
|
+
* The bridge source is never evaluated, and examples inside strings/comments
|
|
5
|
+
* remain ignored.
|
|
6
|
+
*/
|
|
7
|
+
export declare function extractCodexExecCommands(source: string): string[];
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
/** Static extraction for Codex Desktop's JavaScript exec bridge. */
|
|
2
|
+
function skipJsString(source, start) {
|
|
3
|
+
const quote = source[start];
|
|
4
|
+
let index = start + 1;
|
|
5
|
+
while (index < source.length) {
|
|
6
|
+
if (source[index] === '\\') {
|
|
7
|
+
index += 2;
|
|
8
|
+
continue;
|
|
9
|
+
}
|
|
10
|
+
if (source[index] === quote)
|
|
11
|
+
return index + 1;
|
|
12
|
+
index += 1;
|
|
13
|
+
}
|
|
14
|
+
return source.length;
|
|
15
|
+
}
|
|
16
|
+
function skipJsTrivia(source, start) {
|
|
17
|
+
let index = start;
|
|
18
|
+
while (index < source.length) {
|
|
19
|
+
if (/\s/.test(source[index])) {
|
|
20
|
+
index += 1;
|
|
21
|
+
continue;
|
|
22
|
+
}
|
|
23
|
+
if (source.startsWith('//', index)) {
|
|
24
|
+
const newline = source.indexOf('\n', index + 2);
|
|
25
|
+
return newline < 0 ? source.length : skipJsTrivia(source, newline + 1);
|
|
26
|
+
}
|
|
27
|
+
if (source.startsWith('/*', index)) {
|
|
28
|
+
const end = source.indexOf('*/', index + 2);
|
|
29
|
+
return end < 0 ? source.length : skipJsTrivia(source, end + 2);
|
|
30
|
+
}
|
|
31
|
+
break;
|
|
32
|
+
}
|
|
33
|
+
return index;
|
|
34
|
+
}
|
|
35
|
+
function commandPropertyEnd(source, index) {
|
|
36
|
+
const bareKey = source.startsWith('cmd', index)
|
|
37
|
+
&& !/[\w$]/.test(source[index - 1] ?? '')
|
|
38
|
+
&& !/[\w$]/.test(source[index + 3] ?? '');
|
|
39
|
+
if (bareKey)
|
|
40
|
+
return index + 3;
|
|
41
|
+
const quote = source[index];
|
|
42
|
+
if (quote !== '"' && quote !== "'")
|
|
43
|
+
return null;
|
|
44
|
+
const end = skipJsString(source, index);
|
|
45
|
+
return source.slice(index + 1, end - 1) === 'cmd' ? end : null;
|
|
46
|
+
}
|
|
47
|
+
function commandLiteralValue(source, start, end) {
|
|
48
|
+
const literal = source.slice(start, end);
|
|
49
|
+
if (source[start] === '"') {
|
|
50
|
+
try {
|
|
51
|
+
const parsed = JSON.parse(literal);
|
|
52
|
+
if (typeof parsed === 'string')
|
|
53
|
+
return parsed;
|
|
54
|
+
}
|
|
55
|
+
catch {
|
|
56
|
+
// Fall through to the lossless source slice for non-JSON JS escapes.
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
return source.slice(start + 1, Math.max(start + 1, end - 1));
|
|
60
|
+
}
|
|
61
|
+
function extractExecCommandLiteral(source, callStart) {
|
|
62
|
+
let index = skipJsTrivia(source, callStart + 'tools.exec_command'.length);
|
|
63
|
+
if (source[index] !== '(')
|
|
64
|
+
return null;
|
|
65
|
+
index = skipJsTrivia(source, index + 1);
|
|
66
|
+
if (source[index] !== '{')
|
|
67
|
+
return null;
|
|
68
|
+
let depth = 1;
|
|
69
|
+
index += 1;
|
|
70
|
+
while (index < source.length && depth > 0) {
|
|
71
|
+
index = skipJsTrivia(source, index);
|
|
72
|
+
const char = source[index];
|
|
73
|
+
if (depth === 1) {
|
|
74
|
+
const keyEnd = commandPropertyEnd(source, index);
|
|
75
|
+
if (keyEnd !== null) {
|
|
76
|
+
let valueStart = skipJsTrivia(source, keyEnd);
|
|
77
|
+
if (source[valueStart] !== ':') {
|
|
78
|
+
index = keyEnd;
|
|
79
|
+
continue;
|
|
80
|
+
}
|
|
81
|
+
valueStart = skipJsTrivia(source, valueStart + 1);
|
|
82
|
+
const quote = source[valueStart];
|
|
83
|
+
if (quote !== '"' && quote !== "'" && quote !== '`')
|
|
84
|
+
return null;
|
|
85
|
+
const end = skipJsString(source, valueStart);
|
|
86
|
+
return commandLiteralValue(source, valueStart, end);
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
if (char === '"' || char === "'" || char === '`') {
|
|
90
|
+
index = skipJsString(source, index);
|
|
91
|
+
continue;
|
|
92
|
+
}
|
|
93
|
+
if (char === '{') {
|
|
94
|
+
depth += 1;
|
|
95
|
+
index += 1;
|
|
96
|
+
continue;
|
|
97
|
+
}
|
|
98
|
+
if (char === '}') {
|
|
99
|
+
depth -= 1;
|
|
100
|
+
index += 1;
|
|
101
|
+
continue;
|
|
102
|
+
}
|
|
103
|
+
index += 1;
|
|
104
|
+
}
|
|
105
|
+
return null;
|
|
106
|
+
}
|
|
107
|
+
/**
|
|
108
|
+
* Extract only literal `cmd` values from real `tools.exec_command(...)` calls.
|
|
109
|
+
* The bridge source is never evaluated, and examples inside strings/comments
|
|
110
|
+
* remain ignored.
|
|
111
|
+
*/
|
|
112
|
+
export function extractCodexExecCommands(source) {
|
|
113
|
+
const commands = [];
|
|
114
|
+
let index = 0;
|
|
115
|
+
while (index < source.length) {
|
|
116
|
+
index = skipJsTrivia(source, index);
|
|
117
|
+
const char = source[index];
|
|
118
|
+
if (char === '"' || char === "'" || char === '`') {
|
|
119
|
+
index = skipJsString(source, index);
|
|
120
|
+
continue;
|
|
121
|
+
}
|
|
122
|
+
if (source.startsWith('tools.exec_command', index)
|
|
123
|
+
&& !/[\w$.]/.test(source[index - 1] ?? '')
|
|
124
|
+
&& !/[\w$]/.test(source[index + 'tools.exec_command'.length] ?? '')) {
|
|
125
|
+
const command = extractExecCommandLiteral(source, index);
|
|
126
|
+
if (command)
|
|
127
|
+
commands.push(command);
|
|
128
|
+
index += 'tools.exec_command'.length;
|
|
129
|
+
continue;
|
|
130
|
+
}
|
|
131
|
+
index += 1;
|
|
132
|
+
}
|
|
133
|
+
return commands;
|
|
134
|
+
}
|
|
@@ -4,6 +4,7 @@ import { isToolResultFailureText } from './text-signals.js';
|
|
|
4
4
|
import { correlateTraceToolEvents, createTraceId, normalizeTraceTimestamp, traceTimestampBounds, } from './trace-ir.js';
|
|
5
5
|
import { nonNegativeMetric, optionalTokenCount, splitInclusiveInputTokens, tokenCount, } from '../shared/token-usage.js';
|
|
6
6
|
import { normalizeToolIdentity } from '../shared/tool-identity.js';
|
|
7
|
+
import { extractCodexExecCommands } from './codex-exec-command.js';
|
|
7
8
|
const CODEX_RESPONSE_ITEM_TYPES = new Set([
|
|
8
9
|
'message',
|
|
9
10
|
'reasoning',
|
|
@@ -242,8 +243,10 @@ function convertCodexRecords(rawRecords, runId) {
|
|
|
242
243
|
const explicitStatus = runtimeOutcome.present
|
|
243
244
|
? runtimeOutcome.status
|
|
244
245
|
: statusFromCodex(payloadStatus);
|
|
245
|
-
const
|
|
246
|
-
const
|
|
246
|
+
const bridgeFailure = codexToolOutputFailed(output);
|
|
247
|
+
const bridgeSuccess = !bridgeFailure && codexToolOutputSucceeded(output);
|
|
248
|
+
const inferredFailure = bridgeFailure || (!bridgeSuccess && isToolResultFailureText(output));
|
|
249
|
+
const inferredSuccess = bridgeSuccess;
|
|
247
250
|
const completedExternalCall = externalEnds.byOccurrence.has(mcpCallOccurrenceKey(callId, externalOccurrence));
|
|
248
251
|
const inferredStatus = inferredFailure
|
|
249
252
|
? 'failure'
|
|
@@ -609,6 +612,10 @@ function runtimeToolEndOutcome(end) {
|
|
|
609
612
|
}
|
|
610
613
|
function normalizeCodexTool(sourceName, rawInput, mcpEnd, sourceNamespace) {
|
|
611
614
|
const input = parseToolInput(rawInput);
|
|
615
|
+
const sourceInput = stringValue(input.input);
|
|
616
|
+
const execCommands = sourceName.toLowerCase() === 'exec' && sourceInput
|
|
617
|
+
? extractCodexExecCommands(sourceInput)
|
|
618
|
+
: [];
|
|
612
619
|
// Codex desktop's orchestration wrapper names its JavaScript command bridge
|
|
613
620
|
// `exec`. This mapping is source-specific: a generic tool named `exec` must not
|
|
614
621
|
// become shell execution outside the Codex adapter.
|
|
@@ -629,10 +636,13 @@ function normalizeCodexTool(sourceName, rawInput, mcpEnd, sourceNamespace) {
|
|
|
629
636
|
tool,
|
|
630
637
|
input: {
|
|
631
638
|
...input,
|
|
632
|
-
command:
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
639
|
+
command: execCommands.length > 0
|
|
640
|
+
? execCommands.join('\n')
|
|
641
|
+
: stringValue(input.command)
|
|
642
|
+
?? stringValue(input.cmd)
|
|
643
|
+
?? sourceInput
|
|
644
|
+
?? '',
|
|
645
|
+
...(execCommands.length > 0 ? { commands: execCommands } : {}),
|
|
636
646
|
},
|
|
637
647
|
};
|
|
638
648
|
}
|