oh-my-knowledge 0.51.1 → 0.52.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +27 -6
- package/README.zh.md +27 -6
- package/dist/assets/agent-skills/omk/SKILL.md +10 -8
- package/dist/assets/agent-skills/omk/references/commands.md +2 -2
- package/dist/authoring/generator.d.ts +15 -5
- package/dist/authoring/generator.js +181 -18
- package/dist/authoring/sample-fixer.d.ts +2 -0
- package/dist/authoring/sample-fixer.js +19 -5
- package/dist/cli/commands/sample.js +11 -4
- package/dist/eval-core/evaluation-execution.js +11 -2
- package/dist/eval-core/fact-checker.d.ts +15 -2
- package/dist/eval-core/fact-checker.js +75 -20
- package/dist/eval-core/mock-hook.cjs +40 -1
- package/dist/eval-core/mocks-runtime.js +39 -2
- package/dist/eval-core/task-planner.d.ts +2 -2
- package/dist/eval-core/task-planner.js +7 -7
- package/dist/eval-workflows/evaluation-pipeline.js +2 -0
- package/dist/eval-workflows/run-evaluation.js +2 -1
- package/dist/executors/capabilities.d.ts +15 -0
- package/dist/executors/capabilities.js +64 -0
- package/dist/executors/index.d.ts +1 -0
- package/dist/executors/index.js +4 -1
- package/dist/observability/codex-conversation-index.d.ts +88 -0
- package/dist/observability/codex-conversation-index.js +566 -0
- package/dist/observability/codex-protocol.d.ts +12 -0
- package/dist/observability/codex-protocol.js +78 -0
- package/dist/observability/codex-tool-status.d.ts +16 -0
- package/dist/observability/codex-tool-status.js +114 -0
- package/dist/observability/codex-trace-adapter.js +342 -119
- package/dist/observability/conversation-catalog.d.ts +38 -0
- package/dist/observability/conversation-catalog.js +573 -0
- package/dist/observability/conversation-index-process.d.ts +1 -0
- package/dist/observability/conversation-index-process.js +31 -0
- package/dist/observability/conversation-view-model.d.ts +5 -0
- package/dist/observability/conversation-view-model.js +100 -0
- package/dist/observability/experience.d.ts +3 -1
- package/dist/observability/experience.js +248 -7
- package/dist/observability/inbox.js +40 -2
- package/dist/observability/knowledge-debugger.d.ts +4 -0
- package/dist/observability/knowledge-debugger.js +364 -0
- package/dist/observability/polling-subscription-hub.d.ts +27 -0
- package/dist/observability/polling-subscription-hub.js +149 -0
- package/dist/observability/source-record-archive.d.ts +13 -0
- package/dist/observability/source-record-archive.js +324 -0
- package/dist/observability/task-window.d.ts +8 -0
- package/dist/observability/task-window.js +36 -0
- package/dist/observability/trace-ir.d.ts +66 -1
- package/dist/observability/trace-source.d.ts +1 -0
- package/dist/observability/trace-source.js +12 -5
- package/dist/observability/turn-index.d.ts +3 -0
- package/dist/observability/turn-index.js +251 -0
- package/dist/renderer/conversation-renderer.d.ts +9 -0
- package/dist/renderer/conversation-renderer.js +410 -0
- package/dist/renderer/icons.js +1 -0
- package/dist/renderer/inline-markdown.d.ts +6 -0
- package/dist/renderer/inline-markdown.js +158 -0
- package/dist/renderer/knowledge-debugger-renderer.d.ts +9 -0
- package/dist/renderer/knowledge-debugger-renderer.js +2280 -0
- package/dist/renderer/observation-inbox-renderer.js +9 -0
- package/dist/renderer/skill-list-renderer.js +2 -1
- package/dist/renderer/trajectory-live.d.ts +42 -0
- package/dist/renderer/trajectory-live.js +258 -0
- package/dist/renderer/trajectory-routing.d.ts +60 -0
- package/dist/renderer/trajectory-routing.js +351 -0
- package/dist/server/report-server.d.ts +4 -1
- package/dist/server/report-server.js +274 -6
- package/dist/server/skill-insights.js +1 -1
- package/dist/shared/sample-contract.d.ts +1 -0
- package/dist/shared/sample-contract.js +35 -0
- package/dist/shared/tool-identity.d.ts +8 -0
- package/dist/shared/tool-identity.js +13 -0
- package/dist/types/eval.d.ts +7 -6
- package/dist/types/executor.d.ts +3 -2
- package/dist/types/observability.d.ts +225 -1
- package/package.json +10 -7
|
@@ -20,8 +20,17 @@ const FIX_SYSTEM_PROMPT = `你是一个评测用例修复专家。根据诊断
|
|
|
20
20
|
输出一个 **JSON 数组**,包含所有待修复 sample(修改过的和原样保留的都要包含)。
|
|
21
21
|
第一字符 \`[\`,最后 \`]\`,不要用 \`\`\`json\`\`\` 围栏,不要寒暄。
|
|
22
22
|
如果判断某条是 LLM 行为问题不需要改,也原样放进数组。`;
|
|
23
|
-
|
|
24
|
-
|
|
23
|
+
const MOCKLESS_FIX_OVERRIDE = `
|
|
24
|
+
|
|
25
|
+
目标执行器不支持工具调用拦截。不得新增或保留 mocks、mocksStrict、
|
|
26
|
+
mock_hit、tools_called、tools_count_min、tool_input_contains、tool_output_contains。
|
|
27
|
+
environment 仅表示题设上下文,不得当作已物化 fixture;负向安全断言可以保留。`;
|
|
28
|
+
function sanitizeFixedSamples(samples, skillContent, mockless) {
|
|
29
|
+
sanitizeGeneratedSamples(samples, {
|
|
30
|
+
skillContent,
|
|
31
|
+
mockless,
|
|
32
|
+
preserveMocklessEnvironment: true,
|
|
33
|
+
});
|
|
25
34
|
return samples;
|
|
26
35
|
}
|
|
27
36
|
const FIXABLE_SAMPLE_FIELDS = new Set([
|
|
@@ -97,7 +106,7 @@ function parseFixedSamples(text) {
|
|
|
97
106
|
return null;
|
|
98
107
|
}
|
|
99
108
|
export async function fixSamples(options) {
|
|
100
|
-
const { skillContent, samples, report, treatmentKey, executor, model, maxAttemptsPerSample = 2 } = options;
|
|
109
|
+
const { skillContent, samples, report, treatmentKey, executor, model, maxAttemptsPerSample = 2, mockless = false, } = options;
|
|
101
110
|
const sampleMap = new Map(samples.map((s) => [s.sample_id, s]));
|
|
102
111
|
// Collect all fixable samples into one batch
|
|
103
112
|
const fixContexts = [];
|
|
@@ -185,7 +194,12 @@ ${sampleSections}
|
|
|
185
194
|
let incurredCostUSD = 0;
|
|
186
195
|
let incurredCostReported = false;
|
|
187
196
|
try {
|
|
188
|
-
const result = await executor({
|
|
197
|
+
const result = await executor({
|
|
198
|
+
model,
|
|
199
|
+
system: mockless ? `${FIX_SYSTEM_PROMPT}${MOCKLESS_FIX_OVERRIDE}` : FIX_SYSTEM_PROMPT,
|
|
200
|
+
prompt,
|
|
201
|
+
timeoutMs: 300_000,
|
|
202
|
+
});
|
|
189
203
|
incurredCostUSD = result.costUSD;
|
|
190
204
|
incurredCostReported = result.costReported !== false;
|
|
191
205
|
if (!result.ok) {
|
|
@@ -226,7 +240,7 @@ ${sampleSections}
|
|
|
226
240
|
})),
|
|
227
241
|
};
|
|
228
242
|
}
|
|
229
|
-
const sanitizedFixedArr = sanitizeFixedSamples(fixedArr, skillContent);
|
|
243
|
+
const sanitizedFixedArr = sanitizeFixedSamples(fixedArr, skillContent, mockless);
|
|
230
244
|
const outOfScope = sanitizedFixedArr.find((fixed) => {
|
|
231
245
|
const sid = fixed.sample_id;
|
|
232
246
|
const original = sampleMap.get(sid);
|
|
@@ -274,7 +274,7 @@ async function runSampleFix(args, flags, lang) {
|
|
|
274
274
|
throw new CliExit(1);
|
|
275
275
|
}
|
|
276
276
|
process.stderr.write(lang === 'zh' ? `🔧 发现 ${sampleDesignCount} 条 sample_design 失败,开始修复...\n` : `🔧 Found ${sampleDesignCount} sample_design failure(s), fixing...\n`);
|
|
277
|
-
const { createExecutor } = await import('../../executors/index.js');
|
|
277
|
+
const { createExecutor, executorSupportsSampleMocks, } = await import('../../executors/index.js');
|
|
278
278
|
const exec = createExecutor(executorName);
|
|
279
279
|
const executorFn = async (opts) => {
|
|
280
280
|
const result = await exec({
|
|
@@ -298,6 +298,7 @@ async function runSampleFix(args, flags, lang) {
|
|
|
298
298
|
treatmentKey: treatmentName,
|
|
299
299
|
executor: executorFn,
|
|
300
300
|
model,
|
|
301
|
+
mockless: !executorSupportsSampleMocks(executorName),
|
|
301
302
|
});
|
|
302
303
|
let writtenFiles = [];
|
|
303
304
|
if (result.fixedCount > 0) {
|
|
@@ -363,7 +364,13 @@ export async function runSampleFromTraces(flags, lang) {
|
|
|
363
364
|
? `🔭 发现 ${items.length} 个${flags.skill ? ` ${flags.skill} 的` : ''}失败信号,正在生成评测用例草稿...\n`
|
|
364
365
|
: `🔭 Found ${items.length}${flags.skill ? ` ${flags.skill}` : ''} failure signal(s); generating regression-sample drafts...\n`);
|
|
365
366
|
try {
|
|
366
|
-
const { samples, costUSD } = await generateSamplesFromTraces({
|
|
367
|
+
const { samples, costUSD } = await generateSamplesFromTraces({
|
|
368
|
+
items,
|
|
369
|
+
count,
|
|
370
|
+
model,
|
|
371
|
+
executorName,
|
|
372
|
+
noMock: flags['no-mock'],
|
|
373
|
+
});
|
|
367
374
|
const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
368
375
|
if (samples.length === 0) {
|
|
369
376
|
// The model conservatively skipped every signal (noise / unreproducible). That's a
|
|
@@ -663,8 +670,8 @@ export default class Sample extends BaseCommand {
|
|
|
663
670
|
}),
|
|
664
671
|
'no-mock': Flags.boolean({
|
|
665
672
|
description: bilingual({
|
|
666
|
-
zh: '不生成 mocks
|
|
667
|
-
en: 'Skip
|
|
673
|
+
zh: '不生成 mocks。执行器不支持工具拦截时会自动启用,避免产生必然失败的 mock_hit。',
|
|
674
|
+
en: 'Skip mocks. Automatically enabled when the executor cannot intercept tools, preventing impossible mock_hit assertions.',
|
|
668
675
|
}),
|
|
669
676
|
default: false,
|
|
670
677
|
}),
|
|
@@ -258,8 +258,17 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
|
|
|
258
258
|
}
|
|
259
259
|
}
|
|
260
260
|
let factCheck;
|
|
261
|
-
if (execResult.ok && execResult.output
|
|
262
|
-
|
|
261
|
+
if (execResult.ok && execResult.output) {
|
|
262
|
+
const sharedEvidence = {
|
|
263
|
+
...(task._sample.context && { context: task._sample.context }),
|
|
264
|
+
...(task._sample.environment?.files_available?.length && {
|
|
265
|
+
declaredFiles: task._sample.environment.files_available,
|
|
266
|
+
}),
|
|
267
|
+
// Resolve exactly like the executor's cwd. samplesBaseDir is for bundle
|
|
268
|
+
// assets such as mocks, not for changing Sample.cwd path semantics.
|
|
269
|
+
...(task._sample.cwd && { cwd: resolve(task._sample.cwd) }),
|
|
270
|
+
};
|
|
271
|
+
factCheck = checkFacts(execResult.output, sharedEvidence);
|
|
263
272
|
}
|
|
264
273
|
const sampleResults = ownRecordValue(results, task.sample_id)
|
|
265
274
|
?? setOwnRecordValue(results, task.sample_id, {});
|
|
@@ -14,11 +14,24 @@ export interface FactCheckResult {
|
|
|
14
14
|
totalCount: number;
|
|
15
15
|
verifiedRate: number;
|
|
16
16
|
}
|
|
17
|
+
export interface FactCheckEvidence {
|
|
18
|
+
/** Shared sample fixture root. Arm-specific execution directories must not be used. */
|
|
19
|
+
cwd?: string;
|
|
20
|
+
/** Facts supplied to both arms in the sample context. */
|
|
21
|
+
context?: string;
|
|
22
|
+
/** Declarative fixture paths from sample.environment.files_available. */
|
|
23
|
+
declaredFiles?: string[];
|
|
24
|
+
}
|
|
17
25
|
/**
|
|
18
26
|
* Extract file path claims from agent output text.
|
|
19
27
|
*/
|
|
20
28
|
export declare function extractPathClaims(output: string): string[];
|
|
21
29
|
/**
|
|
22
|
-
* Check facts
|
|
30
|
+
* Check file-path facts against sample-level evidence shared by every arm.
|
|
31
|
+
*
|
|
32
|
+
* A string keeps the legacy direct-filesystem API for callers outside the
|
|
33
|
+
* evaluation pipeline. The pipeline passes structured evidence and deliberately
|
|
34
|
+
* excludes each artifact's execution cwd, because that directory is not a
|
|
35
|
+
* comparable source of truth across control and treatment arms.
|
|
23
36
|
*/
|
|
24
|
-
export declare function checkFacts(output: string,
|
|
37
|
+
export declare function checkFacts(output: string, evidence: string | FactCheckEvidence): FactCheckResult;
|
|
@@ -22,6 +22,10 @@ const IGNORE_PATTERNS = [
|
|
|
22
22
|
/^\.git\//,
|
|
23
23
|
/^index\.\w+$/,
|
|
24
24
|
];
|
|
25
|
+
// Product/runtime names that happen to end in a supported source extension.
|
|
26
|
+
const NON_PATH_REFERENCES = new Set([
|
|
27
|
+
'node.js',
|
|
28
|
+
]);
|
|
25
29
|
/**
|
|
26
30
|
* Extract file path claims from agent output text.
|
|
27
31
|
*/
|
|
@@ -36,35 +40,86 @@ export function extractPathClaims(output) {
|
|
|
36
40
|
path = path.replace(/[.,;:!?))]+$/, '');
|
|
37
41
|
// Clean trailing backtick/quote
|
|
38
42
|
path = path.replace(/[`'"]+$/, '');
|
|
39
|
-
if (path.length > 3
|
|
43
|
+
if (path.length > 3
|
|
44
|
+
&& !NON_PATH_REFERENCES.has(path.toLowerCase())
|
|
45
|
+
&& !IGNORE_PATTERNS.some((p) => p.test(path))) {
|
|
40
46
|
paths.add(path);
|
|
41
47
|
}
|
|
42
48
|
}
|
|
43
49
|
}
|
|
44
50
|
return [...paths];
|
|
45
51
|
}
|
|
52
|
+
function normalizeEvidencePath(path) {
|
|
53
|
+
return path
|
|
54
|
+
.trim()
|
|
55
|
+
.replace(/^['"`]+|['"`]+$/g, '')
|
|
56
|
+
.replace(/\\/g, '/')
|
|
57
|
+
.replace(/^(?:\$SKILL_DIR|~)\//, '')
|
|
58
|
+
.replace(/^\.\//, '')
|
|
59
|
+
.replace(/\/+$/, '');
|
|
60
|
+
}
|
|
61
|
+
function isRootedEvidencePath(path) {
|
|
62
|
+
return path.startsWith('/') || /^[a-zA-Z]:\//.test(path);
|
|
63
|
+
}
|
|
64
|
+
function refersToSamePath(left, right) {
|
|
65
|
+
const a = normalizeEvidencePath(left);
|
|
66
|
+
const b = normalizeEvidencePath(right);
|
|
67
|
+
return a === b
|
|
68
|
+
|| (isRootedEvidencePath(a) && !isRootedEvidencePath(b) && a.endsWith(`/${b}`))
|
|
69
|
+
|| (isRootedEvidencePath(b) && !isRootedEvidencePath(a) && b.endsWith(`/${a}`));
|
|
70
|
+
}
|
|
46
71
|
/**
|
|
47
|
-
* Check facts
|
|
72
|
+
* Check file-path facts against sample-level evidence shared by every arm.
|
|
73
|
+
*
|
|
74
|
+
* A string keeps the legacy direct-filesystem API for callers outside the
|
|
75
|
+
* evaluation pipeline. The pipeline passes structured evidence and deliberately
|
|
76
|
+
* excludes each artifact's execution cwd, because that directory is not a
|
|
77
|
+
* comparable source of truth across control and treatment arms.
|
|
48
78
|
*/
|
|
49
|
-
export function checkFacts(output,
|
|
79
|
+
export function checkFacts(output, evidence) {
|
|
50
80
|
const pathClaims = extractPathClaims(output);
|
|
51
|
-
const
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
:
|
|
66
|
-
|
|
67
|
-
}
|
|
81
|
+
const sources = typeof evidence === 'string'
|
|
82
|
+
? { cwd: evidence }
|
|
83
|
+
: evidence;
|
|
84
|
+
const root = sources.cwd ? resolve(sources.cwd) : null;
|
|
85
|
+
const contextClaims = sources.context
|
|
86
|
+
? extractPathClaims(sources.context)
|
|
87
|
+
: [];
|
|
88
|
+
const declaredFiles = sources.declaredFiles ?? [];
|
|
89
|
+
const claims = pathClaims.flatMap((path) => {
|
|
90
|
+
if (declaredFiles.some((declared) => refersToSamePath(path, declared))) {
|
|
91
|
+
return [{
|
|
92
|
+
type: 'file-path',
|
|
93
|
+
value: path,
|
|
94
|
+
verified: true,
|
|
95
|
+
evidence: 'source=context(sample.environment.files_available)',
|
|
96
|
+
}];
|
|
97
|
+
}
|
|
98
|
+
if (root) {
|
|
99
|
+
const fullPath = resolve(root, path);
|
|
100
|
+
const relativePath = relative(root, fullPath);
|
|
101
|
+
const insideCwd = relativePath === ''
|
|
102
|
+
|| (!relativePath.startsWith('..') && !isAbsolute(relativePath));
|
|
103
|
+
const exists = insideCwd && existsSync(fullPath);
|
|
104
|
+
return [{
|
|
105
|
+
type: 'file-path',
|
|
106
|
+
value: path,
|
|
107
|
+
verified: exists,
|
|
108
|
+
evidence: insideCwd
|
|
109
|
+
? `source=runtime-filesystem; ${fullPath} ${exists ? 'exists' : 'not found'}`
|
|
110
|
+
: `source=runtime-filesystem; ${path} is outside the evaluation cwd`,
|
|
111
|
+
}];
|
|
112
|
+
}
|
|
113
|
+
if (contextClaims.some((contextPath) => refersToSamePath(path, contextPath))) {
|
|
114
|
+
return [{
|
|
115
|
+
type: 'file-path',
|
|
116
|
+
value: path,
|
|
117
|
+
verified: true,
|
|
118
|
+
evidence: 'source=context(sample.context)',
|
|
119
|
+
}];
|
|
120
|
+
}
|
|
121
|
+
// Without a shared fixture, absence from context is unknown rather than false.
|
|
122
|
+
return [];
|
|
68
123
|
});
|
|
69
124
|
const verifiedCount = claims.filter((c) => c.verified).length;
|
|
70
125
|
const totalCount = claims.length;
|
|
@@ -92,8 +92,47 @@ function anyStringContains(obj, needle) {
|
|
|
92
92
|
return false;
|
|
93
93
|
}
|
|
94
94
|
|
|
95
|
+
const BUILTIN_TOOL_ALIASES = {
|
|
96
|
+
bash: 'Bash',
|
|
97
|
+
shell: 'Bash',
|
|
98
|
+
exec_command: 'Bash',
|
|
99
|
+
command_execution: 'Bash',
|
|
100
|
+
read: 'Read',
|
|
101
|
+
file_read: 'Read',
|
|
102
|
+
grep: 'Grep',
|
|
103
|
+
edit: 'Edit',
|
|
104
|
+
apply_patch: 'Edit',
|
|
105
|
+
file_change: 'Edit',
|
|
106
|
+
write: 'Write',
|
|
107
|
+
file_write: 'Write',
|
|
108
|
+
view_image: 'ViewImage',
|
|
109
|
+
viewimage: 'ViewImage',
|
|
110
|
+
write_stdin: 'WriteStdin',
|
|
111
|
+
writestdin: 'WriteStdin',
|
|
112
|
+
web_search: 'WebSearch',
|
|
113
|
+
websearch: 'WebSearch',
|
|
114
|
+
};
|
|
115
|
+
|
|
116
|
+
function canonicalToolName(name) {
|
|
117
|
+
const sourceName = String(name);
|
|
118
|
+
const builtin = BUILTIN_TOOL_ALIASES[sourceName.toLowerCase()];
|
|
119
|
+
if (builtin) return builtin;
|
|
120
|
+
const parts = sourceName.split('__').filter(Boolean);
|
|
121
|
+
if (parts[0] === 'mcp' && parts.length > 2) {
|
|
122
|
+
const providerParts = parts.slice(1, -1);
|
|
123
|
+
if (providerParts[0] === 'codex_apps' && providerParts.length > 1) providerParts.shift();
|
|
124
|
+
return providerParts.join('.') + '.' + parts[parts.length - 1];
|
|
125
|
+
}
|
|
126
|
+
return sourceName;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
function toolIdentityMatches(expectedName, runtimeName) {
|
|
130
|
+
return expectedName === runtimeName
|
|
131
|
+
|| canonicalToolName(expectedName) === canonicalToolName(runtimeName);
|
|
132
|
+
}
|
|
133
|
+
|
|
95
134
|
function isMockHit(mock, toolName, toolInput) {
|
|
96
|
-
if (mock.tool !== '*' && mock.tool
|
|
135
|
+
if (mock.tool !== '*' && !toolIdentityMatches(mock.tool, toolName)) return false;
|
|
97
136
|
const m = mock.match;
|
|
98
137
|
if (!m) return true;
|
|
99
138
|
const ti = toolInput || {};
|
|
@@ -18,6 +18,7 @@ import { homedir, tmpdir } from 'node:os';
|
|
|
18
18
|
import { dirname, isAbsolute, join, resolve } from 'node:path';
|
|
19
19
|
import { fileURLToPath } from 'node:url';
|
|
20
20
|
import { incrementRecordCount, setOwnRecordValue } from '../shared/record-count.js';
|
|
21
|
+
import { toolIdentityMatches } from '../shared/tool-identity.js';
|
|
21
22
|
// ─── Match logic ────────────────────────────────────────────────────────────
|
|
22
23
|
function expandHome(p) {
|
|
23
24
|
if (p.startsWith('~/'))
|
|
@@ -112,7 +113,7 @@ function anyStringContains(obj, needle) {
|
|
|
112
113
|
}
|
|
113
114
|
/** 单条 mock 是否命中给定 tool 调用。 */
|
|
114
115
|
export function isMockHit(mock, toolName, toolInput) {
|
|
115
|
-
if (mock.tool !== '*' && mock.tool
|
|
116
|
+
if (mock.tool !== '*' && !toolIdentityMatches(mock.tool, toolName))
|
|
116
117
|
return false;
|
|
117
118
|
const m = mock.match;
|
|
118
119
|
if (!m)
|
|
@@ -414,8 +415,44 @@ function anyStringContains(obj, needle) {
|
|
|
414
415
|
if (typeof obj === 'object' && obj !== null) return Object.values(obj).some((v) => anyStringContains(v, needle));
|
|
415
416
|
return false;
|
|
416
417
|
}
|
|
418
|
+
const BUILTIN_TOOL_ALIASES = {
|
|
419
|
+
bash: 'Bash',
|
|
420
|
+
shell: 'Bash',
|
|
421
|
+
exec_command: 'Bash',
|
|
422
|
+
command_execution: 'Bash',
|
|
423
|
+
read: 'Read',
|
|
424
|
+
file_read: 'Read',
|
|
425
|
+
grep: 'Grep',
|
|
426
|
+
edit: 'Edit',
|
|
427
|
+
apply_patch: 'Edit',
|
|
428
|
+
file_change: 'Edit',
|
|
429
|
+
write: 'Write',
|
|
430
|
+
file_write: 'Write',
|
|
431
|
+
view_image: 'ViewImage',
|
|
432
|
+
viewimage: 'ViewImage',
|
|
433
|
+
write_stdin: 'WriteStdin',
|
|
434
|
+
writestdin: 'WriteStdin',
|
|
435
|
+
web_search: 'WebSearch',
|
|
436
|
+
websearch: 'WebSearch',
|
|
437
|
+
};
|
|
438
|
+
function canonicalToolName(name) {
|
|
439
|
+
const sourceName = String(name);
|
|
440
|
+
const builtin = BUILTIN_TOOL_ALIASES[sourceName.toLowerCase()];
|
|
441
|
+
if (builtin) return builtin;
|
|
442
|
+
const parts = sourceName.split('__').filter(Boolean);
|
|
443
|
+
if (parts[0] === 'mcp' && parts.length > 2) {
|
|
444
|
+
const providerParts = parts.slice(1, -1);
|
|
445
|
+
if (providerParts[0] === 'codex_apps' && providerParts.length > 1) providerParts.shift();
|
|
446
|
+
return providerParts.join('.') + '.' + parts[parts.length - 1];
|
|
447
|
+
}
|
|
448
|
+
return sourceName;
|
|
449
|
+
}
|
|
450
|
+
function toolIdentityMatches(expectedName, runtimeName) {
|
|
451
|
+
return expectedName === runtimeName
|
|
452
|
+
|| canonicalToolName(expectedName) === canonicalToolName(runtimeName);
|
|
453
|
+
}
|
|
417
454
|
function isMockHit(mock, toolName, toolInput) {
|
|
418
|
-
if (mock.tool !== '*' && mock.tool
|
|
455
|
+
if (mock.tool !== '*' && !toolIdentityMatches(mock.tool, toolName)) return false;
|
|
419
456
|
const m = mock.match;
|
|
420
457
|
if (!m) return true;
|
|
421
458
|
const ti = toolInput || {};
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import type { Artifact, Sample, SampleEnvironment, Task } from '../types/index.js';
|
|
2
2
|
/**
|
|
3
3
|
* 把 sample.environment 渲染成自然语言段落,放在用户 prompt 前。
|
|
4
|
-
*
|
|
5
|
-
*
|
|
4
|
+
* 这是题设上下文,不会修改 PATH、物化文件或改变 runtime;LLM 可据此跳过
|
|
5
|
+
* which / test -f 等可用性探测,直接进入 skill 描述的工作流。
|
|
6
6
|
*
|
|
7
7
|
* 输出 null 表示 sample 没声明 environment,prompt 不变。
|
|
8
8
|
*/
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* 把 sample.environment 渲染成自然语言段落,放在用户 prompt 前。
|
|
3
|
-
*
|
|
4
|
-
*
|
|
3
|
+
* 这是题设上下文,不会修改 PATH、物化文件或改变 runtime;LLM 可据此跳过
|
|
4
|
+
* which / test -f 等可用性探测,直接进入 skill 描述的工作流。
|
|
5
5
|
*
|
|
6
6
|
* 输出 null 表示 sample 没声明 environment,prompt 不变。
|
|
7
7
|
*/
|
|
@@ -10,26 +10,26 @@ export function renderEnvironmentSection(env) {
|
|
|
10
10
|
return null;
|
|
11
11
|
const lines = [];
|
|
12
12
|
if (env.cli_available && env.cli_available.length > 0) {
|
|
13
|
-
lines.push('-
|
|
13
|
+
lines.push('- 题设声明可用的 CLI(仅作上下文,不修改 PATH):');
|
|
14
14
|
for (const c of env.cli_available)
|
|
15
15
|
lines.push(` - \`${c}\``);
|
|
16
16
|
}
|
|
17
17
|
if (env.files_available && env.files_available.length > 0) {
|
|
18
|
-
lines.push('-
|
|
18
|
+
lines.push('- 题设引用的文件路径(仅作上下文,不会在 cwd 物化):');
|
|
19
19
|
for (const f of env.files_available)
|
|
20
20
|
lines.push(` - \`${f}\``);
|
|
21
21
|
}
|
|
22
22
|
if (env.notes && env.notes.trim()) {
|
|
23
|
-
lines.push(`-
|
|
23
|
+
lines.push(`- 备注:${env.notes.trim()}`);
|
|
24
24
|
}
|
|
25
25
|
if (lines.length === 0)
|
|
26
26
|
return null;
|
|
27
27
|
return [
|
|
28
|
-
'##
|
|
28
|
+
'## 题设环境声明(仅作上下文)',
|
|
29
29
|
'',
|
|
30
30
|
...lines,
|
|
31
31
|
'',
|
|
32
|
-
'
|
|
32
|
+
'请按以上题设进入 skill 描述的主流程,**不要额外做 `find` / `which` / `test -f` 等可用性探测**。这些声明不会自动创建文件或修改 runtime 环境。',
|
|
33
33
|
].join('\n');
|
|
34
34
|
}
|
|
35
35
|
export function buildTasks(samples, variants, skills) {
|
|
@@ -28,6 +28,7 @@ import { finalizeSuccessfulRun, initializeEvaluationRunState, persistFailedJob,
|
|
|
28
28
|
import { finalizeEvaluationReport } from './evaluation-pipeline/report-finalize.js';
|
|
29
29
|
import { emitIsolationWarnings, emitPowerWarnings } from './evaluation-pipeline/preflight-warnings.js';
|
|
30
30
|
import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
|
|
31
|
+
import { assertSamplesCompatibleWithExecutor } from '../executors/capabilities.js';
|
|
31
32
|
// 兼容 re-export:测试与 run-evaluation.ts 动态 import 仍打 evaluation-pipeline.js
|
|
32
33
|
export { buildPowerWarnings, buildIsolationWarnings } from './evaluation-pipeline/preflight-warnings.js';
|
|
33
34
|
export { _computeTestSetHashForTest } from './evaluation-pipeline/test-set-hash.js';
|
|
@@ -35,6 +36,7 @@ export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, s
|
|
|
35
36
|
// requires 现在由 runEvaluation 上游传给 doctor 处理; eval-pipeline 不再用
|
|
36
37
|
// 但保留接口字段,避免破 programmatic API (类型层面接收, 内部忽略)
|
|
37
38
|
requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
|
|
39
|
+
assertSamplesCompatibleWithExecutor(samples, executorName, lang);
|
|
38
40
|
const variantNames = artifacts.map((artifact) => artifact.name);
|
|
39
41
|
const runState = await initializeEvaluationRunState({
|
|
40
42
|
samplesPath,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { resolve } from 'node:path';
|
|
2
2
|
import { DEFAULT_OUTPUT_DIR, persistReport } from '../eval-core/evaluation-reporting.js';
|
|
3
|
-
import { createExecutor } from '../executors/index.js';
|
|
3
|
+
import { assertSamplesCompatibleWithExecutor, createExecutor, } from '../executors/index.js';
|
|
4
4
|
import { discoverBatchSkills } from '../inputs/skill-loader.js';
|
|
5
5
|
import { confidenceInterval, tTest, effectSize } from '../eval-core/statistics.js';
|
|
6
6
|
import { executeBatchEvaluationRuns, buildBatchVariantSpecs } from './batch-evaluation-workflow.js';
|
|
@@ -26,6 +26,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
26
26
|
mcpConfig,
|
|
27
27
|
strictBaseline,
|
|
28
28
|
});
|
|
29
|
+
assertSamplesCompatibleWithExecutor(samples, executorName, lang);
|
|
29
30
|
// doctor 强制门禁: skill 静态结构 + 元数据 + 依赖 + 用例契约。
|
|
30
31
|
// 在 dryRun 分支之前跑, 让 dry-run 也得到 doctor 覆盖(保护 garbage-in 的 verdict)。
|
|
31
32
|
// 默认强制启用; --skip-doctor 提供 escape hatch,典型场景是评测环境用 mock/stub
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import type { ExecutorFn, ExecutorInput, Sample } from '../types/index.js';
|
|
2
|
+
export type SampleMockSupport = 'native-hooks' | 'delegated-script' | 'unsupported';
|
|
3
|
+
export interface ExecutorCapabilities {
|
|
4
|
+
sampleMocks: SampleMockSupport;
|
|
5
|
+
}
|
|
6
|
+
/**
|
|
7
|
+
* Custom script executors receive the OMK_MOCK_* protocol environment and own
|
|
8
|
+
* the final adapter. Built-ins are explicit so unsupported runtimes can never
|
|
9
|
+
* silently turn mock assertions into model failures.
|
|
10
|
+
*/
|
|
11
|
+
export declare function getExecutorCapabilities(executorName: string): ExecutorCapabilities;
|
|
12
|
+
export declare function executorSupportsSampleMocks(executorName: string): boolean;
|
|
13
|
+
export declare function assertSamplesCompatibleWithExecutor(samples: Sample[], executorName: string, lang?: 'zh' | 'en'): void;
|
|
14
|
+
export declare function assertExecutorInputCapabilities(executorName: string, input: ExecutorInput): void;
|
|
15
|
+
export declare function enforceExecutorCapabilities(executorName: string, executor: ExecutorFn): ExecutorFn;
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
const BUILTIN_CAPABILITIES = {
|
|
2
|
+
claude: { sampleMocks: 'native-hooks' },
|
|
3
|
+
'claude-sdk': { sampleMocks: 'native-hooks' },
|
|
4
|
+
codex: { sampleMocks: 'unsupported' },
|
|
5
|
+
'codex-sdk': { sampleMocks: 'unsupported' },
|
|
6
|
+
gemini: { sampleMocks: 'unsupported' },
|
|
7
|
+
'anthropic-api': { sampleMocks: 'unsupported' },
|
|
8
|
+
'openai-api': { sampleMocks: 'unsupported' },
|
|
9
|
+
};
|
|
10
|
+
/**
|
|
11
|
+
* Custom script executors receive the OMK_MOCK_* protocol environment and own
|
|
12
|
+
* the final adapter. Built-ins are explicit so unsupported runtimes can never
|
|
13
|
+
* silently turn mock assertions into model failures.
|
|
14
|
+
*/
|
|
15
|
+
export function getExecutorCapabilities(executorName) {
|
|
16
|
+
return BUILTIN_CAPABILITIES[executorName]
|
|
17
|
+
?? { sampleMocks: 'delegated-script' };
|
|
18
|
+
}
|
|
19
|
+
export function executorSupportsSampleMocks(executorName) {
|
|
20
|
+
return getExecutorCapabilities(executorName).sampleMocks !== 'unsupported';
|
|
21
|
+
}
|
|
22
|
+
function unsupportedMocksMessage(executorName, sampleIds, lang) {
|
|
23
|
+
const ids = sampleIds.slice(0, 8).join(', ');
|
|
24
|
+
const overflow = sampleIds.length > 8
|
|
25
|
+
? lang === 'zh'
|
|
26
|
+
? ` 等 ${sampleIds.length} 条`
|
|
27
|
+
: ` and ${sampleIds.length - 8} more`
|
|
28
|
+
: '';
|
|
29
|
+
if (lang === 'en') {
|
|
30
|
+
return `Executor "${executorName}" does not support Sample.mocks tool interception. `
|
|
31
|
+
+ `Continuing would make mock_hit assertions structurally impossible and create false evidence. `
|
|
32
|
+
+ `Affected samples: ${ids}${overflow}. `
|
|
33
|
+
+ 'Regenerate them with "omk sample --no-mock", remove mocks/mock_hit, '
|
|
34
|
+
+ 'or evaluate with claude/claude-sdk.';
|
|
35
|
+
}
|
|
36
|
+
return `执行器「${executorName}」不支持 Sample.mocks 工具拦截。`
|
|
37
|
+
+ `继续运行会让 mock_hit 在结构上必然失败并产生伪证据。`
|
|
38
|
+
+ `受影响用例:${ids}${overflow}。`
|
|
39
|
+
+ '请用「omk sample --no-mock」重新生成、删除 mocks/mock_hit,'
|
|
40
|
+
+ '或改用 claude/claude-sdk 评测。';
|
|
41
|
+
}
|
|
42
|
+
export function assertSamplesCompatibleWithExecutor(samples, executorName, lang = 'zh') {
|
|
43
|
+
if (executorSupportsSampleMocks(executorName))
|
|
44
|
+
return;
|
|
45
|
+
const affected = samples
|
|
46
|
+
.filter((sample) => Array.isArray(sample.mocks) && sample.mocks.length > 0)
|
|
47
|
+
.map((sample) => sample.sample_id);
|
|
48
|
+
if (affected.length === 0)
|
|
49
|
+
return;
|
|
50
|
+
throw new Error(unsupportedMocksMessage(executorName, affected, lang));
|
|
51
|
+
}
|
|
52
|
+
export function assertExecutorInputCapabilities(executorName, input) {
|
|
53
|
+
if (executorSupportsSampleMocks(executorName)
|
|
54
|
+
|| !Array.isArray(input.mocks)
|
|
55
|
+
|| input.mocks.length === 0)
|
|
56
|
+
return;
|
|
57
|
+
throw new Error(unsupportedMocksMessage(executorName, ['<programmatic-input>'], 'zh'));
|
|
58
|
+
}
|
|
59
|
+
export function enforceExecutorCapabilities(executorName, executor) {
|
|
60
|
+
return async (input) => {
|
|
61
|
+
assertExecutorInputCapabilities(executorName, input);
|
|
62
|
+
return executor(input);
|
|
63
|
+
};
|
|
64
|
+
}
|
|
@@ -2,4 +2,5 @@ import type { ExecutorFn } from '../types/index.js';
|
|
|
2
2
|
import { extractAgentTrace } from './claude-sdk-trace.js';
|
|
3
3
|
import { createScriptExecutor } from './script.js';
|
|
4
4
|
export { extractAgentTrace, createScriptExecutor };
|
|
5
|
+
export { assertExecutorInputCapabilities, assertSamplesCompatibleWithExecutor, executorSupportsSampleMocks, getExecutorCapabilities, } from './capabilities.js';
|
|
5
6
|
export declare function createExecutor(name: string): ExecutorFn;
|
package/dist/executors/index.js
CHANGED
|
@@ -7,6 +7,7 @@ import { codexSdkExecutor } from './codex-sdk.js';
|
|
|
7
7
|
import { geminiExecutor } from './gemini.js';
|
|
8
8
|
import { openAiApiExecutor } from './openai-api.js';
|
|
9
9
|
import { createScriptExecutor } from './script.js';
|
|
10
|
+
import { enforceExecutorCapabilities } from './capabilities.js';
|
|
10
11
|
// 命名一致性:provider HTTP 路径统一用 `<vendor>-api`(`anthropic-api` / `openai-api`),
|
|
11
12
|
// vendor coding agent CLI 用 vendor 名(`claude` / `codex`)。`openai` 这个不带 -api 后缀的
|
|
12
13
|
// 旧 alias 历史上指 openai-cli 子进程实现,删除后不再设别名 — 用 `--executor openai-api`。
|
|
@@ -20,9 +21,11 @@ const EXECUTOR_REGISTRY = {
|
|
|
20
21
|
'openai-api': openAiApiExecutor,
|
|
21
22
|
};
|
|
22
23
|
export { extractAgentTrace, createScriptExecutor };
|
|
24
|
+
export { assertExecutorInputCapabilities, assertSamplesCompatibleWithExecutor, executorSupportsSampleMocks, getExecutorCapabilities, } from './capabilities.js';
|
|
23
25
|
export function createExecutor(name) {
|
|
24
26
|
if (name.trim().length === 0) {
|
|
25
27
|
throw new Error('executor name or script command is required');
|
|
26
28
|
}
|
|
27
|
-
|
|
29
|
+
const executor = EXECUTOR_REGISTRY[name] || createScriptExecutor(name);
|
|
30
|
+
return enforceExecutorCapabilities(name, executor);
|
|
28
31
|
}
|