oh-my-knowledge 0.52.3 → 0.54.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +21 -1
- package/README.zh.md +21 -1
- package/dist/assets/agent-skills/omk/SKILL.md +6 -0
- package/dist/assets/agent-skills/omk/references/commands.md +1 -1
- package/dist/authoring/evolver.js +1 -1
- package/dist/authoring/generator.js +1 -1
- package/dist/cli/commands/eval/index.js +7 -6
- package/dist/cli/commands/install.js +5 -1
- package/dist/cli/lib/generation-failure-hint.js +6 -8
- package/dist/cli/lib/runtime-defaults.d.ts +2 -0
- package/dist/cli/lib/runtime-defaults.js +7 -4
- package/dist/dsh-plugin/cordis.patch.yml +3 -0
- package/dist/dsh-plugin/host-executor.d.ts +93 -0
- package/dist/dsh-plugin/host-executor.js +232 -0
- package/dist/dsh-plugin/index.d.ts +30 -0
- package/dist/dsh-plugin/index.js +275 -0
- package/dist/dsh-plugin/observe.d.ts +47 -0
- package/dist/dsh-plugin/observe.js +359 -0
- package/dist/dsh-plugin/protocol.d.ts +21 -0
- package/dist/dsh-plugin/protocol.js +229 -0
- package/dist/dsh-plugin/trace-adapter.d.ts +48 -0
- package/dist/dsh-plugin/trace-adapter.js +590 -0
- package/dist/eval-core/comparability.js +3 -0
- package/dist/eval-core/evaluation-execution.js +5 -13
- package/dist/eval-core/evaluation-reporting.d.ts +7 -4
- package/dist/eval-core/evaluation-reporting.js +39 -18
- package/dist/eval-core/judge-independence.d.ts +1 -1
- package/dist/eval-core/judge-independence.js +1 -1
- package/dist/eval-core/report-document.js +6 -0
- package/dist/eval-core/resume-compatibility.d.ts +1 -0
- package/dist/eval-core/resume-compatibility.js +5 -4
- package/dist/eval-workflows/batch-evaluation-workflow.js +1 -1
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +4 -2
- package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
- package/dist/eval-workflows/evaluation-pipeline.js +7 -3
- package/dist/eval-workflows/run-evaluation.d.ts +4 -2
- package/dist/eval-workflows/run-evaluation.js +7 -3
- package/dist/executors/{anthropic-api.d.ts → anthropic/api.d.ts} +1 -1
- package/dist/executors/{anthropic-api.js → anthropic/api.js} +4 -2
- package/dist/executors/{claude-cli.d.ts → anthropic/claude/cli.d.ts} +1 -1
- package/dist/executors/{claude-cli.js → anthropic/claude/cli.js} +5 -3
- package/dist/executors/anthropic/claude/protocol.d.ts +87 -0
- package/dist/executors/{claude-protocol.js → anthropic/claude/protocol.js} +6 -6
- package/dist/executors/{claude-sdk.d.ts → anthropic/claude/sdk.d.ts} +2 -2
- package/dist/executors/{claude-sdk.js → anthropic/claude/sdk.js} +8 -5
- package/dist/executors/anthropic/claude/trace.d.ts +9 -0
- package/dist/executors/{claude-sdk-trace.js → anthropic/claude/trace.js} +5 -5
- package/dist/executors/{capabilities.d.ts → core/capabilities.d.ts} +3 -5
- package/dist/executors/{capabilities.js → core/capabilities.js} +4 -11
- package/dist/executors/core/http.d.ts +6 -0
- package/dist/executors/core/http.js +19 -0
- package/dist/executors/core/limits.d.ts +2 -0
- package/dist/executors/core/limits.js +2 -0
- package/dist/executors/core/optional-dependencies.d.ts +7 -0
- package/dist/executors/core/optional-dependencies.js +35 -0
- package/dist/executors/core/registry.d.ts +145 -0
- package/dist/executors/core/registry.js +127 -0
- package/dist/executors/core/runtime-fingerprint.d.ts +13 -0
- package/dist/executors/{runtime-fingerprint.js → core/runtime-fingerprint.js} +119 -59
- package/dist/executors/core/runtime.d.ts +12 -0
- package/dist/executors/core/runtime.js +61 -0
- package/dist/executors/core/subprocess.d.ts +44 -0
- package/dist/executors/{shared.js → core/subprocess.js} +20 -156
- package/dist/executors/index.d.ts +4 -4
- package/dist/executors/index.js +26 -15
- package/dist/executors/{openai-api.d.ts → openai/api.d.ts} +1 -1
- package/dist/executors/{openai-api.js → openai/api.js} +4 -2
- package/dist/executors/{codex-cli.d.ts → openai/codex/cli.d.ts} +3 -3
- package/dist/executors/{codex-cli.js → openai/codex/cli.js} +5 -3
- package/dist/executors/openai/codex/protocol.d.ts +72 -0
- package/dist/executors/{codex-protocol.js → openai/codex/protocol.js} +33 -2
- package/dist/executors/{codex-sdk.d.ts → openai/codex/sdk.d.ts} +18 -4
- package/dist/executors/{codex-sdk.js → openai/codex/sdk.js} +7 -4
- package/dist/executors/{codex-cli-trace.d.ts → openai/codex/trace.d.ts} +2 -2
- package/dist/executors/{codex-cli-trace.js → openai/codex/trace.js} +4 -4
- package/dist/executors/{script.d.ts → script/index.d.ts} +1 -1
- package/dist/executors/{script.js → script/index.js} +6 -4
- package/dist/grading/judge.d.ts +1 -1
- package/dist/grading/judge.js +1 -1
- package/dist/observability/conversation-catalog.js +1 -1
- package/dist/observability/experience.js +4 -0
- package/dist/observability/inbox.d.ts +4 -1
- package/dist/observability/inbox.js +7 -2
- package/dist/observability/trace-ir.d.ts +6 -3
- package/dist/observability/turn-index.js +4 -0
- package/dist/renderer/conversation-renderer.js +20 -6
- package/dist/renderer/html-renderer.js +30 -11
- package/dist/renderer/knowledge-debugger-renderer.js +6 -1
- package/dist/renderer/observation-inbox-renderer.js +2 -2
- package/dist/renderer/trajectory-live.d.ts +1 -0
- package/dist/renderer/trajectory-live.js +5 -2
- package/dist/shared/trace-source-kind.js +1 -0
- package/dist/types/executor.d.ts +13 -2
- package/dist/types/judge.d.ts +2 -2
- package/dist/types/observability.d.ts +1 -1
- package/dist/types/trace.d.ts +1 -1
- package/package.json +25 -5
- package/dist/executors/claude-protocol.d.ts +0 -28
- package/dist/executors/claude-sdk-trace.d.ts +0 -9
- package/dist/executors/codex-protocol.d.ts +0 -24
- package/dist/executors/gemini.d.ts +0 -2
- package/dist/executors/gemini.js +0 -156
- package/dist/executors/runtime-fingerprint.d.ts +0 -6
- package/dist/executors/shared.d.ts +0 -226
- /package/dist/executors/{script-command.d.ts → script/command.d.ts} +0 -0
- /package/dist/executors/{script-command.js → script/command.js} +0 -0
|
@@ -13,7 +13,7 @@ import { buildVariantConfig, resolveExecutionStrategy } from './execution-strate
|
|
|
13
13
|
import { getJudgePromptHash } from '../grading/judge.js';
|
|
14
14
|
import { getDiagnosticPromptHash, resolveDiagnosticTarget, } from '../grading/diagnostic.js';
|
|
15
15
|
import { bootstrapMeanCI, bootstrapPairedDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES, } from './bootstrap.js';
|
|
16
|
-
import {
|
|
16
|
+
import { resolveExecutorRuntimeFingerprint, } from '../executors/core/runtime-fingerprint.js';
|
|
17
17
|
import { ownRecordValue, setOwnRecordValue, } from '../shared/record-count.js';
|
|
18
18
|
import { writeJsonFileAtomic } from '../shared/atomic-json.js';
|
|
19
19
|
import { hashSample } from './sample-fingerprint.js';
|
|
@@ -39,15 +39,25 @@ export { hashSample } from './sample-fingerprint.js';
|
|
|
39
39
|
export function getCliVersion() {
|
|
40
40
|
return PKG.version;
|
|
41
41
|
}
|
|
42
|
-
export function getGitInfo() {
|
|
43
|
-
//
|
|
44
|
-
//
|
|
45
|
-
//
|
|
42
|
+
export function getGitInfo(cwd = process.cwd()) {
|
|
43
|
+
// porcelain v2 的 branch header 一次返回 commit / branch,后续状态行同时表达 dirty。
|
|
44
|
+
// 相比 rev-parse ×2 + status,少启动两个同步 Git 子进程;报告字段语义不变。
|
|
45
|
+
// stdio 静默 stderr:非 git 目录仍返回 null,不把 fatal 噪声泄漏给用户。
|
|
46
46
|
const gitProbeStdio = ['ignore', 'pipe', 'ignore'];
|
|
47
47
|
try {
|
|
48
|
-
const
|
|
49
|
-
|
|
50
|
-
|
|
48
|
+
const output = execFileSync('git', ['status', '--porcelain=v2', '--branch'], {
|
|
49
|
+
cwd,
|
|
50
|
+
encoding: 'utf-8',
|
|
51
|
+
stdio: gitProbeStdio,
|
|
52
|
+
});
|
|
53
|
+
const lines = output.split('\n');
|
|
54
|
+
const oid = lines.find((line) => line.startsWith('# branch.oid '))?.slice('# branch.oid '.length).trim();
|
|
55
|
+
const head = lines.find((line) => line.startsWith('# branch.head '))?.slice('# branch.head '.length).trim();
|
|
56
|
+
if (!oid || oid === '(initial)' || !head)
|
|
57
|
+
return null;
|
|
58
|
+
const commit = oid;
|
|
59
|
+
const branch = head === '(detached)' ? 'HEAD' : head;
|
|
60
|
+
const dirty = lines.some((line) => line.length > 0 && !line.startsWith('# '));
|
|
51
61
|
return { commit, commitShort: commit.slice(0, 7), branch, dirty };
|
|
52
62
|
}
|
|
53
63
|
catch {
|
|
@@ -64,15 +74,15 @@ function commonRuntime(runtimes) {
|
|
|
64
74
|
function representativeRuntime(runtimes) {
|
|
65
75
|
return Object.values(runtimes)[0];
|
|
66
76
|
}
|
|
67
|
-
export function buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request, }) {
|
|
77
|
+
export function buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request, executor, }) {
|
|
68
78
|
const runtimes = {};
|
|
69
79
|
for (const task of tasks) {
|
|
70
80
|
if (ownRecordValue(runtimes, task.variant))
|
|
71
81
|
continue;
|
|
72
82
|
const executionPlan = resolveExecutionStrategy(task, model, request?.timeoutMs, false);
|
|
73
|
-
setOwnRecordValue(runtimes, task.variant,
|
|
83
|
+
setOwnRecordValue(runtimes, task.variant, resolveExecutorRuntimeFingerprint(executorName, model, {
|
|
74
84
|
skillDir: executionPlan.input.skillDir,
|
|
75
|
-
}));
|
|
85
|
+
}, executor));
|
|
76
86
|
}
|
|
77
87
|
for (const variant of variants) {
|
|
78
88
|
if (ownRecordValue(runtimes, variant))
|
|
@@ -86,13 +96,13 @@ export function buildExecutorRuntimesByVariant({ variants, model, executorName,
|
|
|
86
96
|
: artifact?.locator
|
|
87
97
|
? dirname(artifact.locator)
|
|
88
98
|
: request?.skillDir);
|
|
89
|
-
setOwnRecordValue(runtimes, variant,
|
|
99
|
+
setOwnRecordValue(runtimes, variant, resolveExecutorRuntimeFingerprint(executorName, model, {
|
|
90
100
|
skillDir: fallbackSkillDir,
|
|
91
|
-
}));
|
|
101
|
+
}, executor));
|
|
92
102
|
}
|
|
93
103
|
return runtimes;
|
|
94
104
|
}
|
|
95
|
-
export function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, samplesBaseDir, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, }) {
|
|
105
|
+
export function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, samplesBaseDir, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, executor, judgeExecutors, }) {
|
|
96
106
|
const summary = {};
|
|
97
107
|
for (const variant of variants) {
|
|
98
108
|
const entries = Object.values(results)
|
|
@@ -172,10 +182,18 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
|
|
|
172
182
|
const sampleHashes = Object.fromEntries(samples.map((sample) => [sample.sample_id, hashSample(sample, samplesBaseDir)]));
|
|
173
183
|
const judgeRepeat = request?.judgeRepeat && request.judgeRepeat > 1 ? request.judgeRepeat : undefined;
|
|
174
184
|
const runtimeOptions = { skillDir: request?.skillDir };
|
|
175
|
-
const executorRuntimes = buildExecutorRuntimesByVariant({
|
|
185
|
+
const executorRuntimes = buildExecutorRuntimesByVariant({
|
|
186
|
+
variants,
|
|
187
|
+
model,
|
|
188
|
+
executorName,
|
|
189
|
+
tasks,
|
|
190
|
+
artifacts,
|
|
191
|
+
request,
|
|
192
|
+
executor,
|
|
193
|
+
});
|
|
176
194
|
const executorRuntime = commonRuntime(executorRuntimes)
|
|
177
195
|
?? representativeRuntime(executorRuntimes)
|
|
178
|
-
??
|
|
196
|
+
?? resolveExecutorRuntimeFingerprint(executorName, model, runtimeOptions, executor);
|
|
179
197
|
// request.judgeModels is the authoritative source (always non-empty in new schema).
|
|
180
198
|
// Fallback synthesizes a 1-entry from positional judgeModel/executorName for any
|
|
181
199
|
// legacy caller not yet migrated to the array. noJudge ⇒ runtime undefined per entry.
|
|
@@ -183,7 +201,9 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
|
|
|
183
201
|
const judgeModelsMeta = requestJudges.map((jc) => ({
|
|
184
202
|
executor: jc.executor,
|
|
185
203
|
model: jc.model,
|
|
186
|
-
...(noJudge ? {} : {
|
|
204
|
+
...(noJudge ? {} : {
|
|
205
|
+
runtime: resolveExecutorRuntimeFingerprint(jc.executor, jc.model, runtimeOptions, judgeExecutors?.[jc.executor]),
|
|
206
|
+
}),
|
|
187
207
|
}));
|
|
188
208
|
const diagnosticEnabled = request?.noDiagnostic !== true;
|
|
189
209
|
const diagnosticTarget = resolveDiagnosticTarget(requestJudges, executorName, model);
|
|
@@ -192,7 +212,8 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
|
|
|
192
212
|
enabled: true,
|
|
193
213
|
executor: diagnosticTarget.executor,
|
|
194
214
|
model: diagnosticTarget.model,
|
|
195
|
-
runtime:
|
|
215
|
+
runtime: resolveExecutorRuntimeFingerprint(diagnosticTarget.executor, diagnosticTarget.model, {}, judgeExecutors?.[diagnosticTarget.executor]
|
|
216
|
+
?? (diagnosticTarget.executor === executorName ? executor : undefined)),
|
|
196
217
|
promptHash: getDiagnosticPromptHash(),
|
|
197
218
|
}
|
|
198
219
|
: { enabled: false };
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { executorVendor } from '../executors/
|
|
1
|
+
import { executorVendor } from '../executors/core/registry.js';
|
|
2
2
|
export function analyzeJudgeIndependence(report) {
|
|
3
3
|
const judges = report.meta?.judgeModels ?? [];
|
|
4
4
|
const judgeVendors = judges.map((j) => executorVendor(j.executor));
|
|
@@ -71,6 +71,12 @@ function isRuntimeFingerprint(value) {
|
|
|
71
71
|
|| !['reported', 'not-reported', 'unknown'].includes(String(value.capabilities.costUSD))
|
|
72
72
|
|| !['native', 'best-effort', 'none', 'unknown'].includes(String(value.capabilities.trace))
|
|
73
73
|
|| !['full', 'full-no-partial', 'cwd-only', 'none', 'unknown'].includes(String(value.capabilities.skillIsolation))
|
|
74
|
+
|| (value.auditability !== undefined
|
|
75
|
+
&& (!isRecord(value.auditability)
|
|
76
|
+
|| !['complete', 'partial'].includes(String(value.auditability.status))
|
|
77
|
+
|| (value.auditability.reasons !== undefined
|
|
78
|
+
&& (!Array.isArray(value.auditability.reasons)
|
|
79
|
+
|| !value.auditability.reasons.every((reason) => typeof reason === 'string')))))
|
|
74
80
|
|| (value.sdk !== undefined && !isRuntimePackage(value.sdk)))
|
|
75
81
|
return false;
|
|
76
82
|
if (value.binary === undefined)
|
|
@@ -17,6 +17,7 @@ export interface ResumeCompatibilityInput {
|
|
|
17
17
|
samplesBaseDir?: string;
|
|
18
18
|
tasks: Task[];
|
|
19
19
|
artifacts: Artifact[];
|
|
20
|
+
executorOverrides?: Readonly<Record<string, import('../types/index.js').ExecutorFn>>;
|
|
20
21
|
}
|
|
21
22
|
export interface ResumeCompatibilityResult {
|
|
22
23
|
compatible: boolean;
|
|
@@ -3,7 +3,7 @@ import { buildExecutorRuntimesByVariant, EVALUATION_REPORT_SCHEMA_VERSION, getCl
|
|
|
3
3
|
import { hashSample } from './sample-fingerprint.js';
|
|
4
4
|
import { getJudgePromptHash } from '../grading/judge.js';
|
|
5
5
|
import { getDiagnosticPromptHash, resolveDiagnosticTarget, } from '../grading/diagnostic.js';
|
|
6
|
-
import {
|
|
6
|
+
import { resolveExecutorRuntimeFingerprint } from '../executors/core/runtime-fingerprint.js';
|
|
7
7
|
function canonicalStringify(value) {
|
|
8
8
|
if (value === undefined)
|
|
9
9
|
return 'undefined';
|
|
@@ -73,6 +73,7 @@ export function checkResumeCompatibility(report, current) {
|
|
|
73
73
|
skillDir: current.skillDir,
|
|
74
74
|
timeoutMs: current.timeoutMs,
|
|
75
75
|
},
|
|
76
|
+
executor: current.executorOverrides?.[current.executorName],
|
|
76
77
|
});
|
|
77
78
|
check('meta.executorRuntimes', runtimeFingerprints(report.meta.executorRuntimes), runtimeFingerprints(expectedExecutorRuntimes));
|
|
78
79
|
const actualJudges = report.meta.judgeModels.map((judge) => ({
|
|
@@ -85,9 +86,9 @@ export function checkResumeCompatibility(report, current) {
|
|
|
85
86
|
model: judge.model,
|
|
86
87
|
...(!current.noJudge
|
|
87
88
|
? {
|
|
88
|
-
runtime:
|
|
89
|
+
runtime: resolveExecutorRuntimeFingerprint(judge.executor, judge.model, {
|
|
89
90
|
skillDir: current.skillDir,
|
|
90
|
-
}).fingerprint,
|
|
91
|
+
}, current.executorOverrides?.[judge.executor]).fingerprint,
|
|
91
92
|
}
|
|
92
93
|
: {}),
|
|
93
94
|
}));
|
|
@@ -102,7 +103,7 @@ export function checkResumeCompatibility(report, current) {
|
|
|
102
103
|
enabled: true,
|
|
103
104
|
executor: diagnosticTarget.executor,
|
|
104
105
|
model: diagnosticTarget.model,
|
|
105
|
-
runtime:
|
|
106
|
+
runtime: resolveExecutorRuntimeFingerprint(diagnosticTarget.executor, diagnosticTarget.model, {}, current.executorOverrides?.[diagnosticTarget.executor]).fingerprint,
|
|
106
107
|
promptHash: getDiagnosticPromptHash(),
|
|
107
108
|
}
|
|
108
109
|
: { enabled: false };
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { dirname } from 'node:path';
|
|
2
2
|
import { DEFAULT_OUTPUT_DIR, EVALUATION_REPORT_SCHEMA_VERSION, generateRunId, getCliVersion, getGitInfo, persistReport, } from '../eval-core/evaluation-reporting.js';
|
|
3
3
|
import { buildEvaluationRequest, createEvaluationRun, createSucceededJob, finalizeEvaluationRun } from '../eval-core/evaluation-job.js';
|
|
4
|
-
import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
|
|
4
|
+
import { getExecutorRuntimeFingerprint } from '../executors/core/runtime-fingerprint.js';
|
|
5
5
|
import { createFileJobStore } from '../server/job-store.js';
|
|
6
6
|
import { DEFAULT_JOBS_DIR } from '../eval-core/default-dirs.js';
|
|
7
7
|
import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
|
|
@@ -16,6 +16,7 @@
|
|
|
16
16
|
import { existsSync, readdirSync } from 'node:fs';
|
|
17
17
|
import { homedir } from 'node:os';
|
|
18
18
|
import { join, relative } from 'node:path';
|
|
19
|
+
import { executorFamily } from '../../executors/core/registry.js';
|
|
19
20
|
import { tEvalWorkflowMessage } from '../messages.js';
|
|
20
21
|
export function buildPowerWarnings(sampleCount, repeat, lang = 'zh') {
|
|
21
22
|
const warnings = [];
|
|
@@ -43,11 +44,12 @@ export function buildIsolationWarnings(artifacts, strictBaseline, options) {
|
|
|
43
44
|
if (!hasUnisolatedBaseline)
|
|
44
45
|
return [];
|
|
45
46
|
const executorName = options.executorName;
|
|
47
|
+
const family = executorFamily(executorName);
|
|
46
48
|
const home = options.homeDir ?? homedir();
|
|
47
49
|
const cwd = options.cwd ?? process.cwd();
|
|
48
|
-
const roots =
|
|
50
|
+
const roots = family === 'claude'
|
|
49
51
|
? [join(home, '.claude', 'skills'), join(cwd, '.claude', 'skills')]
|
|
50
|
-
:
|
|
52
|
+
: family === 'codex'
|
|
51
53
|
? [
|
|
52
54
|
join(home, '.agents', 'skills'),
|
|
53
55
|
join(home, '.codex', 'skills'),
|
|
@@ -43,6 +43,8 @@ export interface EvaluationPipelineOptions {
|
|
|
43
43
|
judgeExecutorName: string;
|
|
44
44
|
executor: ExecutorFn;
|
|
45
45
|
judgeExecutor: ExecutorFn;
|
|
46
|
+
/** Embedding-host executors that must not be reconstructed from a CLI name. */
|
|
47
|
+
executorOverrides?: Readonly<Record<string, ExecutorFn>>;
|
|
46
48
|
outputDir?: string | null;
|
|
47
49
|
project?: string;
|
|
48
50
|
owner?: string;
|
|
@@ -91,7 +93,7 @@ export interface EvaluationPipelineOptions {
|
|
|
91
93
|
noDiagnostic?: boolean;
|
|
92
94
|
}
|
|
93
95
|
type VariantResult = import('../types/index.js').VariantResult;
|
|
94
|
-
export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
|
|
96
|
+
export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, executorOverrides, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
|
|
95
97
|
report: Report;
|
|
96
98
|
filePath: string | null;
|
|
97
99
|
}>;
|
|
@@ -28,11 +28,11 @@ import { finalizeSuccessfulRun, initializeEvaluationRunState, persistFailedJob,
|
|
|
28
28
|
import { finalizeEvaluationReport } from './evaluation-pipeline/report-finalize.js';
|
|
29
29
|
import { emitIsolationWarnings, emitPowerWarnings } from './evaluation-pipeline/preflight-warnings.js';
|
|
30
30
|
import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
|
|
31
|
-
import { assertSamplesCompatibleWithExecutor } from '../executors/capabilities.js';
|
|
31
|
+
import { assertSamplesCompatibleWithExecutor } from '../executors/core/capabilities.js';
|
|
32
32
|
// 兼容 re-export:测试与 run-evaluation.ts 动态 import 仍打 evaluation-pipeline.js
|
|
33
33
|
export { buildPowerWarnings, buildIsolationWarnings } from './evaluation-pipeline/preflight-warnings.js';
|
|
34
34
|
export { _computeTestSetHashForTest } from './evaluation-pipeline/test-set-hash.js';
|
|
35
|
-
export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, concurrency = 1, timeoutMs, noCache = false, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, verbose = false, retry = 0, existingResults,
|
|
35
|
+
export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, executorOverrides, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, concurrency = 1, timeoutMs, noCache = false, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, verbose = false, retry = 0, existingResults,
|
|
36
36
|
// requires 现在由 runEvaluation 上游传给 doctor 处理; eval-pipeline 不再用
|
|
37
37
|
// 但保留接口字段,避免破 programmatic API (类型层面接收, 内部忽略)
|
|
38
38
|
requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
|
|
@@ -87,7 +87,9 @@ requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRep
|
|
|
87
87
|
resolvedJudgeExecutors = Object.create(null);
|
|
88
88
|
for (const jc of resolvedJudgeModels) {
|
|
89
89
|
if (!ownRecordValue(resolvedJudgeExecutors, jc.executor)) {
|
|
90
|
-
setOwnRecordValue(resolvedJudgeExecutors, jc.executor,
|
|
90
|
+
setOwnRecordValue(resolvedJudgeExecutors, jc.executor, executorOverrides?.[jc.executor]
|
|
91
|
+
?? (jc.executor === executorName ? executor : undefined)
|
|
92
|
+
?? createExecutor(jc.executor));
|
|
91
93
|
}
|
|
92
94
|
}
|
|
93
95
|
}
|
|
@@ -156,6 +158,8 @@ requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRep
|
|
|
156
158
|
run,
|
|
157
159
|
job,
|
|
158
160
|
layeredStats,
|
|
161
|
+
executor,
|
|
162
|
+
judgeExecutors: resolvedJudgeExecutors,
|
|
159
163
|
}),
|
|
160
164
|
results,
|
|
161
165
|
artifacts,
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { Artifact, BatchEvaluationReport, EvaluationReport, JobStore, ProgressCallback, Report, VarianceData, VariantSpec } from '../types/index.js';
|
|
1
|
+
import type { Artifact, BatchEvaluationReport, EvaluationReport, ExecutorFn, JobStore, ProgressCallback, Report, VarianceData, VariantSpec } from '../types/index.js';
|
|
2
2
|
export interface SkillProgressInfo {
|
|
3
3
|
phase: string;
|
|
4
4
|
skill: string;
|
|
@@ -17,6 +17,8 @@ interface CommonEvaluationOptions {
|
|
|
17
17
|
timeoutMs?: number;
|
|
18
18
|
/** Core API is runtime-neutral: CLI/default resolution happens before this boundary. */
|
|
19
19
|
executorName: string;
|
|
20
|
+
/** Same-process executors supplied by an embedding host, keyed by executor name. */
|
|
21
|
+
executorOverrides?: Readonly<Record<string, ExecutorFn>>;
|
|
20
22
|
jobStore?: JobStore | null;
|
|
21
23
|
persistJob?: boolean;
|
|
22
24
|
onProgress?: ProgressCallback | null;
|
|
@@ -122,7 +124,7 @@ export interface DryRunReport extends DryRunBase {
|
|
|
122
124
|
samplesPath: string;
|
|
123
125
|
tasks: DryRunTask[];
|
|
124
126
|
}
|
|
125
|
-
export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
|
|
127
|
+
export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, executorOverrides, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
|
|
126
128
|
report: Report | DryRunReport;
|
|
127
129
|
filePath: string | null;
|
|
128
130
|
}>;
|
|
@@ -10,7 +10,7 @@ import { checkResumeCompatibility } from '../eval-core/resume-compatibility.js';
|
|
|
10
10
|
import { findSaturationPoint } from '../analysis/saturation.js';
|
|
11
11
|
import { bootstrapMeanCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
|
|
12
12
|
import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
|
|
13
|
-
export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
|
|
13
|
+
export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName, executorOverrides, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
|
|
14
14
|
// Unified judgeModels → derive single-judge fields for downstream pipeline / grading
|
|
15
15
|
// (which still operate on string `judgeModel` + `judgeExecutorName` fields per call).
|
|
16
16
|
const effectiveJudgeModels = judgeModels && judgeModels.length > 0
|
|
@@ -123,6 +123,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
123
123
|
samplesBaseDir,
|
|
124
124
|
tasks,
|
|
125
125
|
artifacts: resolvedArtifacts,
|
|
126
|
+
executorOverrides,
|
|
126
127
|
});
|
|
127
128
|
if (compatibility.compatible) {
|
|
128
129
|
const sourceBySample = new Map(existing.results.map((entry) => [entry.sample_id, entry.variants]));
|
|
@@ -159,8 +160,10 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
159
160
|
const effectiveSkipConnectivity = existingResults !== undefined
|
|
160
161
|
? true
|
|
161
162
|
: skipConnectivity;
|
|
162
|
-
const executor = createExecutor(executorName);
|
|
163
|
-
const
|
|
163
|
+
const executor = executorOverrides?.[executorName] ?? createExecutor(executorName);
|
|
164
|
+
const effectiveJudgeExecutorName = judgeExecutorName || executorName;
|
|
165
|
+
const judgeExecutor = executorOverrides?.[effectiveJudgeExecutorName]
|
|
166
|
+
?? createExecutor(effectiveJudgeExecutorName);
|
|
164
167
|
return executeEvaluationPipeline({
|
|
165
168
|
samplesPath,
|
|
166
169
|
samplesBaseDir,
|
|
@@ -176,6 +179,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
176
179
|
judgeExecutorName: judgeExecutorName || executorName,
|
|
177
180
|
executor,
|
|
178
181
|
judgeExecutor,
|
|
182
|
+
executorOverrides,
|
|
179
183
|
outputDir,
|
|
180
184
|
project,
|
|
181
185
|
owner,
|
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
import type { ExecResult, ExecutorInput } from '
|
|
1
|
+
import type { ExecResult, ExecutorInput } from '../../types/index.js';
|
|
2
2
|
export declare function anthropicApiExecutor({ model, system, prompt, timeoutMs }: ExecutorInput): Promise<ExecResult>;
|
|
@@ -1,5 +1,7 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
1
|
+
import { optionalTokenCount } from '../../shared/token-usage.js';
|
|
2
|
+
import { DEFAULT_TIMEOUT_MS } from '../core/limits.js';
|
|
3
|
+
import { readJsonResponse, responseBodyPreview } from '../core/http.js';
|
|
4
|
+
import { asErrorLike, errorMessage } from '../core/runtime.js';
|
|
3
5
|
export async function anthropicApiExecutor({ model, system, prompt, timeoutMs = DEFAULT_TIMEOUT_MS }) {
|
|
4
6
|
const apiKey = process.env.ANTHROPIC_API_KEY;
|
|
5
7
|
if (!apiKey)
|
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
import type { ExecResult, ExecutorInput } from '
|
|
1
|
+
import type { ExecResult, ExecutorInput } from '../../../types/index.js';
|
|
2
2
|
export declare function claudeCliExecutor({ model, system, prompt, cwd, skillDir, timeoutMs, allowedSkills, mocks, mocksBaseDir, mocksStrict, lean, effort }: ExecutorInput): Promise<ExecResult>;
|
|
@@ -1,6 +1,8 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
3
|
-
import {
|
|
1
|
+
import { materializeForCliConfigDir } from '../../../eval-core/mocks-runtime.js';
|
|
2
|
+
import { buildClaudeResult, parseClaudeStreamJson } from './protocol.js';
|
|
3
|
+
import { DEFAULT_TIMEOUT_MS, MAX_BUFFER } from '../../core/limits.js';
|
|
4
|
+
import { buildExecEnv, errorMessage, interruptedExecResult, timeoutExecResult, } from '../../core/runtime.js';
|
|
5
|
+
import { spawnWithSigintPropagation } from '../../core/subprocess.js';
|
|
4
6
|
// claude CLI 用 `--disable-slash-commands` (文档:"Disable all skills") +
|
|
5
7
|
// `--disallowedTools Skill` 实现与 SDK 等价的完全隔离。非空 skill 白名单不再支持
|
|
6
8
|
// (它无法真正隔离,见 claude-sdk.ts buildSdkIsolationOptions),所以 [name1, ...] 必须 throw。
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
import type { ExecResult } from '../../../types/index.js';
|
|
2
|
+
interface ClaudeTokenUsage {
|
|
3
|
+
input_tokens?: number;
|
|
4
|
+
output_tokens?: number;
|
|
5
|
+
cache_read_input_tokens?: number;
|
|
6
|
+
cache_creation_input_tokens?: number;
|
|
7
|
+
}
|
|
8
|
+
export interface ClaudeSdkQueryOptions {
|
|
9
|
+
model?: string;
|
|
10
|
+
systemPrompt?: string;
|
|
11
|
+
cwd: string;
|
|
12
|
+
permissionMode: 'bypassPermissions';
|
|
13
|
+
allowDangerouslySkipPermissions: true;
|
|
14
|
+
abortController: AbortController;
|
|
15
|
+
env: NodeJS.ProcessEnv;
|
|
16
|
+
}
|
|
17
|
+
export interface ClaudeSdkQueryInput {
|
|
18
|
+
prompt: string;
|
|
19
|
+
options: ClaudeSdkQueryOptions;
|
|
20
|
+
}
|
|
21
|
+
export interface ClaudeMessage {
|
|
22
|
+
type: string;
|
|
23
|
+
message?: {
|
|
24
|
+
role?: string;
|
|
25
|
+
content?: Array<{
|
|
26
|
+
type: string;
|
|
27
|
+
text?: string;
|
|
28
|
+
id?: string;
|
|
29
|
+
name?: string;
|
|
30
|
+
input?: unknown;
|
|
31
|
+
}>;
|
|
32
|
+
};
|
|
33
|
+
tool_use_id?: string;
|
|
34
|
+
content?: string | Array<{
|
|
35
|
+
type: string;
|
|
36
|
+
text?: string;
|
|
37
|
+
}>;
|
|
38
|
+
is_error?: boolean;
|
|
39
|
+
}
|
|
40
|
+
export interface ClaudeResultMessage extends ClaudeMessage {
|
|
41
|
+
type: 'result';
|
|
42
|
+
result?: string;
|
|
43
|
+
usage?: ClaudeTokenUsage;
|
|
44
|
+
total_cost_usd?: number;
|
|
45
|
+
duration_api_ms?: number;
|
|
46
|
+
duration_ms?: number;
|
|
47
|
+
num_turns?: number;
|
|
48
|
+
stop_reason?: string | null;
|
|
49
|
+
modelUsage?: Record<string, {
|
|
50
|
+
inputTokens?: number;
|
|
51
|
+
outputTokens?: number;
|
|
52
|
+
cacheReadInputTokens?: number;
|
|
53
|
+
cacheCreationInputTokens?: number;
|
|
54
|
+
}>;
|
|
55
|
+
subtype?: string;
|
|
56
|
+
errors?: string[];
|
|
57
|
+
}
|
|
58
|
+
export interface ClaudeSdkModule {
|
|
59
|
+
query: (opts: ClaudeSdkQueryInput) => AsyncIterable<ClaudeMessage>;
|
|
60
|
+
}
|
|
61
|
+
export interface ClaudeMeasurements {
|
|
62
|
+
durationMs: number;
|
|
63
|
+
durationApiMs: number;
|
|
64
|
+
inputTokens: number;
|
|
65
|
+
outputTokens: number;
|
|
66
|
+
cacheReadTokens: number;
|
|
67
|
+
cacheCreationTokens: number;
|
|
68
|
+
costUSD: number;
|
|
69
|
+
numTurns: number;
|
|
70
|
+
}
|
|
71
|
+
export declare function normalizeClaudeMeasurements(result: ClaudeResultMessage): ClaudeMeasurements | {
|
|
72
|
+
error: string;
|
|
73
|
+
};
|
|
74
|
+
export interface ClaudeStreamParseResult {
|
|
75
|
+
messages: ClaudeMessage[];
|
|
76
|
+
malformedLineCount: number;
|
|
77
|
+
}
|
|
78
|
+
export declare function parseClaudeStreamJson(stdout: string): ClaudeStreamParseResult;
|
|
79
|
+
export declare function buildClaudeResult(options: {
|
|
80
|
+
messages: ClaudeMessage[];
|
|
81
|
+
wallClockDurationMs: number;
|
|
82
|
+
source: 'claude stream-json' | 'claude-sdk';
|
|
83
|
+
malformedLineCount?: number;
|
|
84
|
+
forcedError?: string;
|
|
85
|
+
messageTimestamps?: number[];
|
|
86
|
+
}): ExecResult;
|
|
87
|
+
export {};
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { checkedSumTokenCounts, nonNegativeMetric, optionalTokenCount, } from '
|
|
2
|
-
import {
|
|
3
|
-
export function
|
|
1
|
+
import { checkedSumTokenCounts, nonNegativeMetric, optionalTokenCount, } from '../../../shared/token-usage.js';
|
|
2
|
+
import { extractClaudeTrace, isClaudeResultMessage } from './trace.js';
|
|
3
|
+
export function normalizeClaudeMeasurements(result) {
|
|
4
4
|
const durationMs = optionalTokenCount(result.duration_ms);
|
|
5
5
|
const durationApiMs = optionalTokenCount(result.duration_api_ms);
|
|
6
6
|
const numTurns = optionalTokenCount(result.num_turns);
|
|
@@ -96,8 +96,8 @@ export function parseClaudeStreamJson(stdout) {
|
|
|
96
96
|
}
|
|
97
97
|
export function buildClaudeResult(options) {
|
|
98
98
|
const { messages, wallClockDurationMs, source, malformedLineCount = 0, forcedError, messageTimestamps, } = options;
|
|
99
|
-
const resultMessages = messages.filter(
|
|
100
|
-
const trace =
|
|
99
|
+
const resultMessages = messages.filter(isClaudeResultMessage);
|
|
100
|
+
const trace = extractClaudeTrace(messages, messageTimestamps);
|
|
101
101
|
const traceFields = {
|
|
102
102
|
fullNumTurns: trace.fullNumTurns,
|
|
103
103
|
numSubAgents: trace.numSubAgents,
|
|
@@ -133,7 +133,7 @@ export function buildClaudeResult(options) {
|
|
|
133
133
|
};
|
|
134
134
|
}
|
|
135
135
|
const result = resultMessages[0];
|
|
136
|
-
const measurements =
|
|
136
|
+
const measurements = normalizeClaudeMeasurements(result);
|
|
137
137
|
if ('error' in measurements) {
|
|
138
138
|
errors.push(measurements.error);
|
|
139
139
|
return {
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import type { ExecResult, ExecutorInput } from '
|
|
2
|
-
export {
|
|
1
|
+
import type { ExecResult, ExecutorInput } from '../../../types/index.js';
|
|
2
|
+
export { normalizeClaudeMeasurements } from './protocol.js';
|
|
3
3
|
/**
|
|
4
4
|
* Map ExecutorInput.allowedSkills to SDK query options for skill isolation.
|
|
5
5
|
* undefined → {} (SDK default: full ~/.claude/skills/ discovery)
|
|
@@ -1,11 +1,14 @@
|
|
|
1
1
|
import { join } from 'node:path';
|
|
2
2
|
import { mkdirSync, writeFileSync } from 'node:fs';
|
|
3
3
|
import { tmpdir } from 'node:os';
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
6
|
-
import {
|
|
7
|
-
|
|
4
|
+
import { DEFAULT_TIMEOUT_MS } from '../../core/limits.js';
|
|
5
|
+
import { asErrorLike, buildExecEnv, errorMessage, interruptedExecResult, timeoutExecResult, } from '../../core/runtime.js';
|
|
6
|
+
import { registerSigintSubscriber } from '../../core/subprocess.js';
|
|
7
|
+
import { buildSdkHookCallback } from '../../../eval-core/mocks-runtime.js';
|
|
8
|
+
import { buildClaudeResult } from './protocol.js';
|
|
9
|
+
export { normalizeClaudeMeasurements } from './protocol.js';
|
|
8
10
|
let sdkQuery = null;
|
|
11
|
+
const CLAUDE_AGENT_SDK_PACKAGE = '@anthropic-ai/claude-agent-sdk';
|
|
9
12
|
/**
|
|
10
13
|
* Map ExecutorInput.allowedSkills to SDK query options for skill isolation.
|
|
11
14
|
* undefined → {} (SDK default: full ~/.claude/skills/ discovery)
|
|
@@ -28,7 +31,7 @@ export function buildSdkIsolationOptions(allowedSkills) {
|
|
|
28
31
|
}
|
|
29
32
|
async function getSdkQuery() {
|
|
30
33
|
if (!sdkQuery) {
|
|
31
|
-
const sdk = await import(
|
|
34
|
+
const sdk = await import(CLAUDE_AGENT_SDK_PACKAGE);
|
|
32
35
|
sdkQuery = sdk.query;
|
|
33
36
|
}
|
|
34
37
|
return sdkQuery;
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
import type { ToolCallInfo, TurnInfo } from '../../../types/index.js';
|
|
2
|
+
import type { ClaudeMessage } from './protocol.js';
|
|
3
|
+
export declare function isClaudeResultMessage(message: ClaudeMessage): boolean;
|
|
4
|
+
export declare function extractClaudeTrace(messages: ClaudeMessage[], timestamps?: number[]): {
|
|
5
|
+
turns: TurnInfo[];
|
|
6
|
+
toolCalls: ToolCallInfo[];
|
|
7
|
+
fullNumTurns: number;
|
|
8
|
+
numSubAgents: number;
|
|
9
|
+
};
|
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
import { safeSliceForJson } from '
|
|
2
|
-
import { isToolResultFailureText } from '
|
|
3
|
-
import { normalizeToolIdentity } from '
|
|
4
|
-
export function
|
|
1
|
+
import { safeSliceForJson } from '../../../util/safe-slice.js';
|
|
2
|
+
import { isToolResultFailureText } from '../../../observability/text-signals.js';
|
|
3
|
+
import { normalizeToolIdentity } from '../../../shared/tool-identity.js';
|
|
4
|
+
export function isClaudeResultMessage(message) {
|
|
5
5
|
return message.type === 'result';
|
|
6
6
|
}
|
|
7
|
-
export function
|
|
7
|
+
export function extractClaudeTrace(messages, timestamps) {
|
|
8
8
|
const turns = [];
|
|
9
9
|
const toolCalls = [];
|
|
10
10
|
const pendingToolUse = new Map();
|
|
@@ -1,8 +1,6 @@
|
|
|
1
|
-
import type { ExecutorFn, ExecutorInput, Sample } from '
|
|
2
|
-
|
|
3
|
-
export
|
|
4
|
-
sampleMocks: SampleMockSupport;
|
|
5
|
-
}
|
|
1
|
+
import type { ExecutorFn, ExecutorInput, Sample } from '../../types/index.js';
|
|
2
|
+
import { type ExecutorCapabilities } from './registry.js';
|
|
3
|
+
export type { ExecutorCapabilities, SampleMockSupport } from './registry.js';
|
|
6
4
|
/**
|
|
7
5
|
* Custom script executors receive the OMK_MOCK_* protocol environment and own
|
|
8
6
|
* the final adapter. Built-ins are explicit so unsupported runtimes can never
|
|
@@ -1,20 +1,13 @@
|
|
|
1
|
-
|
|
2
|
-
claude: { sampleMocks: 'native-hooks' },
|
|
3
|
-
'claude-sdk': { sampleMocks: 'native-hooks' },
|
|
4
|
-
codex: { sampleMocks: 'unsupported' },
|
|
5
|
-
'codex-sdk': { sampleMocks: 'unsupported' },
|
|
6
|
-
gemini: { sampleMocks: 'unsupported' },
|
|
7
|
-
'anthropic-api': { sampleMocks: 'unsupported' },
|
|
8
|
-
'openai-api': { sampleMocks: 'unsupported' },
|
|
9
|
-
};
|
|
1
|
+
import { getExecutorDescriptor, } from './registry.js';
|
|
10
2
|
/**
|
|
11
3
|
* Custom script executors receive the OMK_MOCK_* protocol environment and own
|
|
12
4
|
* the final adapter. Built-ins are explicit so unsupported runtimes can never
|
|
13
5
|
* silently turn mock assertions into model failures.
|
|
14
6
|
*/
|
|
15
7
|
export function getExecutorCapabilities(executorName) {
|
|
16
|
-
return
|
|
17
|
-
|
|
8
|
+
return {
|
|
9
|
+
sampleMocks: getExecutorDescriptor(executorName)?.sampleMocks ?? 'delegated-script',
|
|
10
|
+
};
|
|
18
11
|
}
|
|
19
12
|
export function executorSupportsSampleMocks(executorName) {
|
|
20
13
|
return getExecutorCapabilities(executorName).sampleMocks !== 'unsupported';
|