oh-my-knowledge 0.52.3 → 0.53.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -1
- package/README.zh.md +13 -1
- package/dist/assets/agent-skills/omk/SKILL.md +4 -0
- package/dist/assets/agent-skills/omk/references/commands.md +1 -1
- package/dist/authoring/evolver.js +1 -1
- package/dist/authoring/generator.js +1 -1
- package/dist/cli/commands/eval/index.js +7 -6
- package/dist/cli/commands/install.js +5 -1
- package/dist/cli/lib/generation-failure-hint.js +6 -8
- package/dist/cli/lib/runtime-defaults.d.ts +2 -0
- package/dist/cli/lib/runtime-defaults.js +7 -4
- package/dist/dsh-plugin/cordis.patch.yml +3 -0
- package/dist/dsh-plugin/host-executor.d.ts +93 -0
- package/dist/dsh-plugin/host-executor.js +232 -0
- package/dist/dsh-plugin/index.d.ts +28 -0
- package/dist/dsh-plugin/index.js +156 -0
- package/dist/dsh-plugin/protocol.d.ts +21 -0
- package/dist/dsh-plugin/protocol.js +229 -0
- package/dist/eval-core/comparability.js +3 -0
- package/dist/eval-core/evaluation-execution.js +5 -13
- package/dist/eval-core/evaluation-reporting.d.ts +7 -4
- package/dist/eval-core/evaluation-reporting.js +39 -18
- package/dist/eval-core/judge-independence.d.ts +1 -1
- package/dist/eval-core/judge-independence.js +1 -1
- package/dist/eval-core/report-document.js +6 -0
- package/dist/eval-core/resume-compatibility.d.ts +1 -0
- package/dist/eval-core/resume-compatibility.js +5 -4
- package/dist/eval-workflows/batch-evaluation-workflow.js +1 -1
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +4 -2
- package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
- package/dist/eval-workflows/evaluation-pipeline.js +7 -3
- package/dist/eval-workflows/run-evaluation.d.ts +4 -2
- package/dist/eval-workflows/run-evaluation.js +7 -3
- package/dist/executors/{anthropic-api.d.ts → anthropic/api.d.ts} +1 -1
- package/dist/executors/{anthropic-api.js → anthropic/api.js} +4 -2
- package/dist/executors/{claude-cli.d.ts → anthropic/claude/cli.d.ts} +1 -1
- package/dist/executors/{claude-cli.js → anthropic/claude/cli.js} +5 -3
- package/dist/executors/anthropic/claude/protocol.d.ts +87 -0
- package/dist/executors/{claude-protocol.js → anthropic/claude/protocol.js} +6 -6
- package/dist/executors/{claude-sdk.d.ts → anthropic/claude/sdk.d.ts} +2 -2
- package/dist/executors/{claude-sdk.js → anthropic/claude/sdk.js} +8 -5
- package/dist/executors/anthropic/claude/trace.d.ts +9 -0
- package/dist/executors/{claude-sdk-trace.js → anthropic/claude/trace.js} +5 -5
- package/dist/executors/{capabilities.d.ts → core/capabilities.d.ts} +3 -5
- package/dist/executors/{capabilities.js → core/capabilities.js} +4 -11
- package/dist/executors/core/http.d.ts +6 -0
- package/dist/executors/core/http.js +19 -0
- package/dist/executors/core/limits.d.ts +2 -0
- package/dist/executors/core/limits.js +2 -0
- package/dist/executors/core/optional-dependencies.d.ts +7 -0
- package/dist/executors/core/optional-dependencies.js +35 -0
- package/dist/executors/core/registry.d.ts +145 -0
- package/dist/executors/core/registry.js +127 -0
- package/dist/executors/core/runtime-fingerprint.d.ts +13 -0
- package/dist/executors/{runtime-fingerprint.js → core/runtime-fingerprint.js} +119 -59
- package/dist/executors/core/runtime.d.ts +12 -0
- package/dist/executors/core/runtime.js +61 -0
- package/dist/executors/core/subprocess.d.ts +44 -0
- package/dist/executors/{shared.js → core/subprocess.js} +20 -156
- package/dist/executors/index.d.ts +4 -4
- package/dist/executors/index.js +26 -15
- package/dist/executors/{openai-api.d.ts → openai/api.d.ts} +1 -1
- package/dist/executors/{openai-api.js → openai/api.js} +4 -2
- package/dist/executors/{codex-cli.d.ts → openai/codex/cli.d.ts} +3 -3
- package/dist/executors/{codex-cli.js → openai/codex/cli.js} +5 -3
- package/dist/executors/openai/codex/protocol.d.ts +72 -0
- package/dist/executors/{codex-protocol.js → openai/codex/protocol.js} +33 -2
- package/dist/executors/{codex-sdk.d.ts → openai/codex/sdk.d.ts} +18 -4
- package/dist/executors/{codex-sdk.js → openai/codex/sdk.js} +7 -4
- package/dist/executors/{codex-cli-trace.d.ts → openai/codex/trace.d.ts} +2 -2
- package/dist/executors/{codex-cli-trace.js → openai/codex/trace.js} +4 -4
- package/dist/executors/{script.d.ts → script/index.d.ts} +1 -1
- package/dist/executors/{script.js → script/index.js} +6 -4
- package/dist/grading/judge.d.ts +1 -1
- package/dist/grading/judge.js +1 -1
- package/dist/renderer/html-renderer.js +30 -11
- package/dist/types/executor.d.ts +13 -2
- package/dist/types/judge.d.ts +2 -2
- package/package.json +21 -5
- package/dist/executors/claude-protocol.d.ts +0 -28
- package/dist/executors/claude-sdk-trace.d.ts +0 -9
- package/dist/executors/codex-protocol.d.ts +0 -24
- package/dist/executors/gemini.d.ts +0 -2
- package/dist/executors/gemini.js +0 -156
- package/dist/executors/runtime-fingerprint.d.ts +0 -6
- package/dist/executors/shared.d.ts +0 -226
- /package/dist/executors/{script-command.d.ts → script/command.d.ts} +0 -0
- /package/dist/executors/{script-command.js → script/command.js} +0 -0
|
@@ -3,7 +3,7 @@ import { buildExecutorRuntimesByVariant, EVALUATION_REPORT_SCHEMA_VERSION, getCl
|
|
|
3
3
|
import { hashSample } from './sample-fingerprint.js';
|
|
4
4
|
import { getJudgePromptHash } from '../grading/judge.js';
|
|
5
5
|
import { getDiagnosticPromptHash, resolveDiagnosticTarget, } from '../grading/diagnostic.js';
|
|
6
|
-
import {
|
|
6
|
+
import { resolveExecutorRuntimeFingerprint } from '../executors/core/runtime-fingerprint.js';
|
|
7
7
|
function canonicalStringify(value) {
|
|
8
8
|
if (value === undefined)
|
|
9
9
|
return 'undefined';
|
|
@@ -73,6 +73,7 @@ export function checkResumeCompatibility(report, current) {
|
|
|
73
73
|
skillDir: current.skillDir,
|
|
74
74
|
timeoutMs: current.timeoutMs,
|
|
75
75
|
},
|
|
76
|
+
executor: current.executorOverrides?.[current.executorName],
|
|
76
77
|
});
|
|
77
78
|
check('meta.executorRuntimes', runtimeFingerprints(report.meta.executorRuntimes), runtimeFingerprints(expectedExecutorRuntimes));
|
|
78
79
|
const actualJudges = report.meta.judgeModels.map((judge) => ({
|
|
@@ -85,9 +86,9 @@ export function checkResumeCompatibility(report, current) {
|
|
|
85
86
|
model: judge.model,
|
|
86
87
|
...(!current.noJudge
|
|
87
88
|
? {
|
|
88
|
-
runtime:
|
|
89
|
+
runtime: resolveExecutorRuntimeFingerprint(judge.executor, judge.model, {
|
|
89
90
|
skillDir: current.skillDir,
|
|
90
|
-
}).fingerprint,
|
|
91
|
+
}, current.executorOverrides?.[judge.executor]).fingerprint,
|
|
91
92
|
}
|
|
92
93
|
: {}),
|
|
93
94
|
}));
|
|
@@ -102,7 +103,7 @@ export function checkResumeCompatibility(report, current) {
|
|
|
102
103
|
enabled: true,
|
|
103
104
|
executor: diagnosticTarget.executor,
|
|
104
105
|
model: diagnosticTarget.model,
|
|
105
|
-
runtime:
|
|
106
|
+
runtime: resolveExecutorRuntimeFingerprint(diagnosticTarget.executor, diagnosticTarget.model, {}, current.executorOverrides?.[diagnosticTarget.executor]).fingerprint,
|
|
106
107
|
promptHash: getDiagnosticPromptHash(),
|
|
107
108
|
}
|
|
108
109
|
: { enabled: false };
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { dirname } from 'node:path';
|
|
2
2
|
import { DEFAULT_OUTPUT_DIR, EVALUATION_REPORT_SCHEMA_VERSION, generateRunId, getCliVersion, getGitInfo, persistReport, } from '../eval-core/evaluation-reporting.js';
|
|
3
3
|
import { buildEvaluationRequest, createEvaluationRun, createSucceededJob, finalizeEvaluationRun } from '../eval-core/evaluation-job.js';
|
|
4
|
-
import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
|
|
4
|
+
import { getExecutorRuntimeFingerprint } from '../executors/core/runtime-fingerprint.js';
|
|
5
5
|
import { createFileJobStore } from '../server/job-store.js';
|
|
6
6
|
import { DEFAULT_JOBS_DIR } from '../eval-core/default-dirs.js';
|
|
7
7
|
import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
|
|
@@ -16,6 +16,7 @@
|
|
|
16
16
|
import { existsSync, readdirSync } from 'node:fs';
|
|
17
17
|
import { homedir } from 'node:os';
|
|
18
18
|
import { join, relative } from 'node:path';
|
|
19
|
+
import { executorFamily } from '../../executors/core/registry.js';
|
|
19
20
|
import { tEvalWorkflowMessage } from '../messages.js';
|
|
20
21
|
export function buildPowerWarnings(sampleCount, repeat, lang = 'zh') {
|
|
21
22
|
const warnings = [];
|
|
@@ -43,11 +44,12 @@ export function buildIsolationWarnings(artifacts, strictBaseline, options) {
|
|
|
43
44
|
if (!hasUnisolatedBaseline)
|
|
44
45
|
return [];
|
|
45
46
|
const executorName = options.executorName;
|
|
47
|
+
const family = executorFamily(executorName);
|
|
46
48
|
const home = options.homeDir ?? homedir();
|
|
47
49
|
const cwd = options.cwd ?? process.cwd();
|
|
48
|
-
const roots =
|
|
50
|
+
const roots = family === 'claude'
|
|
49
51
|
? [join(home, '.claude', 'skills'), join(cwd, '.claude', 'skills')]
|
|
50
|
-
:
|
|
52
|
+
: family === 'codex'
|
|
51
53
|
? [
|
|
52
54
|
join(home, '.agents', 'skills'),
|
|
53
55
|
join(home, '.codex', 'skills'),
|
|
@@ -43,6 +43,8 @@ export interface EvaluationPipelineOptions {
|
|
|
43
43
|
judgeExecutorName: string;
|
|
44
44
|
executor: ExecutorFn;
|
|
45
45
|
judgeExecutor: ExecutorFn;
|
|
46
|
+
/** Embedding-host executors that must not be reconstructed from a CLI name. */
|
|
47
|
+
executorOverrides?: Readonly<Record<string, ExecutorFn>>;
|
|
46
48
|
outputDir?: string | null;
|
|
47
49
|
project?: string;
|
|
48
50
|
owner?: string;
|
|
@@ -91,7 +93,7 @@ export interface EvaluationPipelineOptions {
|
|
|
91
93
|
noDiagnostic?: boolean;
|
|
92
94
|
}
|
|
93
95
|
type VariantResult = import('../types/index.js').VariantResult;
|
|
94
|
-
export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
|
|
96
|
+
export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, executorOverrides, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
|
|
95
97
|
report: Report;
|
|
96
98
|
filePath: string | null;
|
|
97
99
|
}>;
|
|
@@ -28,11 +28,11 @@ import { finalizeSuccessfulRun, initializeEvaluationRunState, persistFailedJob,
|
|
|
28
28
|
import { finalizeEvaluationReport } from './evaluation-pipeline/report-finalize.js';
|
|
29
29
|
import { emitIsolationWarnings, emitPowerWarnings } from './evaluation-pipeline/preflight-warnings.js';
|
|
30
30
|
import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
|
|
31
|
-
import { assertSamplesCompatibleWithExecutor } from '../executors/capabilities.js';
|
|
31
|
+
import { assertSamplesCompatibleWithExecutor } from '../executors/core/capabilities.js';
|
|
32
32
|
// 兼容 re-export:测试与 run-evaluation.ts 动态 import 仍打 evaluation-pipeline.js
|
|
33
33
|
export { buildPowerWarnings, buildIsolationWarnings } from './evaluation-pipeline/preflight-warnings.js';
|
|
34
34
|
export { _computeTestSetHashForTest } from './evaluation-pipeline/test-set-hash.js';
|
|
35
|
-
export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, concurrency = 1, timeoutMs, noCache = false, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, verbose = false, retry = 0, existingResults,
|
|
35
|
+
export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, executorOverrides, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, concurrency = 1, timeoutMs, noCache = false, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, verbose = false, retry = 0, existingResults,
|
|
36
36
|
// requires 现在由 runEvaluation 上游传给 doctor 处理; eval-pipeline 不再用
|
|
37
37
|
// 但保留接口字段,避免破 programmatic API (类型层面接收, 内部忽略)
|
|
38
38
|
requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
|
|
@@ -87,7 +87,9 @@ requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRep
|
|
|
87
87
|
resolvedJudgeExecutors = Object.create(null);
|
|
88
88
|
for (const jc of resolvedJudgeModels) {
|
|
89
89
|
if (!ownRecordValue(resolvedJudgeExecutors, jc.executor)) {
|
|
90
|
-
setOwnRecordValue(resolvedJudgeExecutors, jc.executor,
|
|
90
|
+
setOwnRecordValue(resolvedJudgeExecutors, jc.executor, executorOverrides?.[jc.executor]
|
|
91
|
+
?? (jc.executor === executorName ? executor : undefined)
|
|
92
|
+
?? createExecutor(jc.executor));
|
|
91
93
|
}
|
|
92
94
|
}
|
|
93
95
|
}
|
|
@@ -156,6 +158,8 @@ requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRep
|
|
|
156
158
|
run,
|
|
157
159
|
job,
|
|
158
160
|
layeredStats,
|
|
161
|
+
executor,
|
|
162
|
+
judgeExecutors: resolvedJudgeExecutors,
|
|
159
163
|
}),
|
|
160
164
|
results,
|
|
161
165
|
artifacts,
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { Artifact, BatchEvaluationReport, EvaluationReport, JobStore, ProgressCallback, Report, VarianceData, VariantSpec } from '../types/index.js';
|
|
1
|
+
import type { Artifact, BatchEvaluationReport, EvaluationReport, ExecutorFn, JobStore, ProgressCallback, Report, VarianceData, VariantSpec } from '../types/index.js';
|
|
2
2
|
export interface SkillProgressInfo {
|
|
3
3
|
phase: string;
|
|
4
4
|
skill: string;
|
|
@@ -17,6 +17,8 @@ interface CommonEvaluationOptions {
|
|
|
17
17
|
timeoutMs?: number;
|
|
18
18
|
/** Core API is runtime-neutral: CLI/default resolution happens before this boundary. */
|
|
19
19
|
executorName: string;
|
|
20
|
+
/** Same-process executors supplied by an embedding host, keyed by executor name. */
|
|
21
|
+
executorOverrides?: Readonly<Record<string, ExecutorFn>>;
|
|
20
22
|
jobStore?: JobStore | null;
|
|
21
23
|
persistJob?: boolean;
|
|
22
24
|
onProgress?: ProgressCallback | null;
|
|
@@ -122,7 +124,7 @@ export interface DryRunReport extends DryRunBase {
|
|
|
122
124
|
samplesPath: string;
|
|
123
125
|
tasks: DryRunTask[];
|
|
124
126
|
}
|
|
125
|
-
export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
|
|
127
|
+
export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, executorOverrides, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
|
|
126
128
|
report: Report | DryRunReport;
|
|
127
129
|
filePath: string | null;
|
|
128
130
|
}>;
|
|
@@ -10,7 +10,7 @@ import { checkResumeCompatibility } from '../eval-core/resume-compatibility.js';
|
|
|
10
10
|
import { findSaturationPoint } from '../analysis/saturation.js';
|
|
11
11
|
import { bootstrapMeanCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
|
|
12
12
|
import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
|
|
13
|
-
export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
|
|
13
|
+
export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName, executorOverrides, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
|
|
14
14
|
// Unified judgeModels → derive single-judge fields for downstream pipeline / grading
|
|
15
15
|
// (which still operate on string `judgeModel` + `judgeExecutorName` fields per call).
|
|
16
16
|
const effectiveJudgeModels = judgeModels && judgeModels.length > 0
|
|
@@ -123,6 +123,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
123
123
|
samplesBaseDir,
|
|
124
124
|
tasks,
|
|
125
125
|
artifacts: resolvedArtifacts,
|
|
126
|
+
executorOverrides,
|
|
126
127
|
});
|
|
127
128
|
if (compatibility.compatible) {
|
|
128
129
|
const sourceBySample = new Map(existing.results.map((entry) => [entry.sample_id, entry.variants]));
|
|
@@ -159,8 +160,10 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
159
160
|
const effectiveSkipConnectivity = existingResults !== undefined
|
|
160
161
|
? true
|
|
161
162
|
: skipConnectivity;
|
|
162
|
-
const executor = createExecutor(executorName);
|
|
163
|
-
const
|
|
163
|
+
const executor = executorOverrides?.[executorName] ?? createExecutor(executorName);
|
|
164
|
+
const effectiveJudgeExecutorName = judgeExecutorName || executorName;
|
|
165
|
+
const judgeExecutor = executorOverrides?.[effectiveJudgeExecutorName]
|
|
166
|
+
?? createExecutor(effectiveJudgeExecutorName);
|
|
164
167
|
return executeEvaluationPipeline({
|
|
165
168
|
samplesPath,
|
|
166
169
|
samplesBaseDir,
|
|
@@ -176,6 +179,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
176
179
|
judgeExecutorName: judgeExecutorName || executorName,
|
|
177
180
|
executor,
|
|
178
181
|
judgeExecutor,
|
|
182
|
+
executorOverrides,
|
|
179
183
|
outputDir,
|
|
180
184
|
project,
|
|
181
185
|
owner,
|
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
import type { ExecResult, ExecutorInput } from '
|
|
1
|
+
import type { ExecResult, ExecutorInput } from '../../types/index.js';
|
|
2
2
|
export declare function anthropicApiExecutor({ model, system, prompt, timeoutMs }: ExecutorInput): Promise<ExecResult>;
|
|
@@ -1,5 +1,7 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
1
|
+
import { optionalTokenCount } from '../../shared/token-usage.js';
|
|
2
|
+
import { DEFAULT_TIMEOUT_MS } from '../core/limits.js';
|
|
3
|
+
import { readJsonResponse, responseBodyPreview } from '../core/http.js';
|
|
4
|
+
import { asErrorLike, errorMessage } from '../core/runtime.js';
|
|
3
5
|
export async function anthropicApiExecutor({ model, system, prompt, timeoutMs = DEFAULT_TIMEOUT_MS }) {
|
|
4
6
|
const apiKey = process.env.ANTHROPIC_API_KEY;
|
|
5
7
|
if (!apiKey)
|
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
import type { ExecResult, ExecutorInput } from '
|
|
1
|
+
import type { ExecResult, ExecutorInput } from '../../../types/index.js';
|
|
2
2
|
export declare function claudeCliExecutor({ model, system, prompt, cwd, skillDir, timeoutMs, allowedSkills, mocks, mocksBaseDir, mocksStrict, lean, effort }: ExecutorInput): Promise<ExecResult>;
|
|
@@ -1,6 +1,8 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
3
|
-
import {
|
|
1
|
+
import { materializeForCliConfigDir } from '../../../eval-core/mocks-runtime.js';
|
|
2
|
+
import { buildClaudeResult, parseClaudeStreamJson } from './protocol.js';
|
|
3
|
+
import { DEFAULT_TIMEOUT_MS, MAX_BUFFER } from '../../core/limits.js';
|
|
4
|
+
import { buildExecEnv, errorMessage, interruptedExecResult, timeoutExecResult, } from '../../core/runtime.js';
|
|
5
|
+
import { spawnWithSigintPropagation } from '../../core/subprocess.js';
|
|
4
6
|
// claude CLI 用 `--disable-slash-commands` (文档:"Disable all skills") +
|
|
5
7
|
// `--disallowedTools Skill` 实现与 SDK 等价的完全隔离。非空 skill 白名单不再支持
|
|
6
8
|
// (它无法真正隔离,见 claude-sdk.ts buildSdkIsolationOptions),所以 [name1, ...] 必须 throw。
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
import type { ExecResult } from '../../../types/index.js';
|
|
2
|
+
interface ClaudeTokenUsage {
|
|
3
|
+
input_tokens?: number;
|
|
4
|
+
output_tokens?: number;
|
|
5
|
+
cache_read_input_tokens?: number;
|
|
6
|
+
cache_creation_input_tokens?: number;
|
|
7
|
+
}
|
|
8
|
+
export interface ClaudeSdkQueryOptions {
|
|
9
|
+
model?: string;
|
|
10
|
+
systemPrompt?: string;
|
|
11
|
+
cwd: string;
|
|
12
|
+
permissionMode: 'bypassPermissions';
|
|
13
|
+
allowDangerouslySkipPermissions: true;
|
|
14
|
+
abortController: AbortController;
|
|
15
|
+
env: NodeJS.ProcessEnv;
|
|
16
|
+
}
|
|
17
|
+
export interface ClaudeSdkQueryInput {
|
|
18
|
+
prompt: string;
|
|
19
|
+
options: ClaudeSdkQueryOptions;
|
|
20
|
+
}
|
|
21
|
+
export interface ClaudeMessage {
|
|
22
|
+
type: string;
|
|
23
|
+
message?: {
|
|
24
|
+
role?: string;
|
|
25
|
+
content?: Array<{
|
|
26
|
+
type: string;
|
|
27
|
+
text?: string;
|
|
28
|
+
id?: string;
|
|
29
|
+
name?: string;
|
|
30
|
+
input?: unknown;
|
|
31
|
+
}>;
|
|
32
|
+
};
|
|
33
|
+
tool_use_id?: string;
|
|
34
|
+
content?: string | Array<{
|
|
35
|
+
type: string;
|
|
36
|
+
text?: string;
|
|
37
|
+
}>;
|
|
38
|
+
is_error?: boolean;
|
|
39
|
+
}
|
|
40
|
+
export interface ClaudeResultMessage extends ClaudeMessage {
|
|
41
|
+
type: 'result';
|
|
42
|
+
result?: string;
|
|
43
|
+
usage?: ClaudeTokenUsage;
|
|
44
|
+
total_cost_usd?: number;
|
|
45
|
+
duration_api_ms?: number;
|
|
46
|
+
duration_ms?: number;
|
|
47
|
+
num_turns?: number;
|
|
48
|
+
stop_reason?: string | null;
|
|
49
|
+
modelUsage?: Record<string, {
|
|
50
|
+
inputTokens?: number;
|
|
51
|
+
outputTokens?: number;
|
|
52
|
+
cacheReadInputTokens?: number;
|
|
53
|
+
cacheCreationInputTokens?: number;
|
|
54
|
+
}>;
|
|
55
|
+
subtype?: string;
|
|
56
|
+
errors?: string[];
|
|
57
|
+
}
|
|
58
|
+
export interface ClaudeSdkModule {
|
|
59
|
+
query: (opts: ClaudeSdkQueryInput) => AsyncIterable<ClaudeMessage>;
|
|
60
|
+
}
|
|
61
|
+
export interface ClaudeMeasurements {
|
|
62
|
+
durationMs: number;
|
|
63
|
+
durationApiMs: number;
|
|
64
|
+
inputTokens: number;
|
|
65
|
+
outputTokens: number;
|
|
66
|
+
cacheReadTokens: number;
|
|
67
|
+
cacheCreationTokens: number;
|
|
68
|
+
costUSD: number;
|
|
69
|
+
numTurns: number;
|
|
70
|
+
}
|
|
71
|
+
export declare function normalizeClaudeMeasurements(result: ClaudeResultMessage): ClaudeMeasurements | {
|
|
72
|
+
error: string;
|
|
73
|
+
};
|
|
74
|
+
export interface ClaudeStreamParseResult {
|
|
75
|
+
messages: ClaudeMessage[];
|
|
76
|
+
malformedLineCount: number;
|
|
77
|
+
}
|
|
78
|
+
export declare function parseClaudeStreamJson(stdout: string): ClaudeStreamParseResult;
|
|
79
|
+
export declare function buildClaudeResult(options: {
|
|
80
|
+
messages: ClaudeMessage[];
|
|
81
|
+
wallClockDurationMs: number;
|
|
82
|
+
source: 'claude stream-json' | 'claude-sdk';
|
|
83
|
+
malformedLineCount?: number;
|
|
84
|
+
forcedError?: string;
|
|
85
|
+
messageTimestamps?: number[];
|
|
86
|
+
}): ExecResult;
|
|
87
|
+
export {};
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { checkedSumTokenCounts, nonNegativeMetric, optionalTokenCount, } from '
|
|
2
|
-
import {
|
|
3
|
-
export function
|
|
1
|
+
import { checkedSumTokenCounts, nonNegativeMetric, optionalTokenCount, } from '../../../shared/token-usage.js';
|
|
2
|
+
import { extractClaudeTrace, isClaudeResultMessage } from './trace.js';
|
|
3
|
+
export function normalizeClaudeMeasurements(result) {
|
|
4
4
|
const durationMs = optionalTokenCount(result.duration_ms);
|
|
5
5
|
const durationApiMs = optionalTokenCount(result.duration_api_ms);
|
|
6
6
|
const numTurns = optionalTokenCount(result.num_turns);
|
|
@@ -96,8 +96,8 @@ export function parseClaudeStreamJson(stdout) {
|
|
|
96
96
|
}
|
|
97
97
|
export function buildClaudeResult(options) {
|
|
98
98
|
const { messages, wallClockDurationMs, source, malformedLineCount = 0, forcedError, messageTimestamps, } = options;
|
|
99
|
-
const resultMessages = messages.filter(
|
|
100
|
-
const trace =
|
|
99
|
+
const resultMessages = messages.filter(isClaudeResultMessage);
|
|
100
|
+
const trace = extractClaudeTrace(messages, messageTimestamps);
|
|
101
101
|
const traceFields = {
|
|
102
102
|
fullNumTurns: trace.fullNumTurns,
|
|
103
103
|
numSubAgents: trace.numSubAgents,
|
|
@@ -133,7 +133,7 @@ export function buildClaudeResult(options) {
|
|
|
133
133
|
};
|
|
134
134
|
}
|
|
135
135
|
const result = resultMessages[0];
|
|
136
|
-
const measurements =
|
|
136
|
+
const measurements = normalizeClaudeMeasurements(result);
|
|
137
137
|
if ('error' in measurements) {
|
|
138
138
|
errors.push(measurements.error);
|
|
139
139
|
return {
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import type { ExecResult, ExecutorInput } from '
|
|
2
|
-
export {
|
|
1
|
+
import type { ExecResult, ExecutorInput } from '../../../types/index.js';
|
|
2
|
+
export { normalizeClaudeMeasurements } from './protocol.js';
|
|
3
3
|
/**
|
|
4
4
|
* Map ExecutorInput.allowedSkills to SDK query options for skill isolation.
|
|
5
5
|
* undefined → {} (SDK default: full ~/.claude/skills/ discovery)
|
|
@@ -1,11 +1,14 @@
|
|
|
1
1
|
import { join } from 'node:path';
|
|
2
2
|
import { mkdirSync, writeFileSync } from 'node:fs';
|
|
3
3
|
import { tmpdir } from 'node:os';
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
6
|
-
import {
|
|
7
|
-
|
|
4
|
+
import { DEFAULT_TIMEOUT_MS } from '../../core/limits.js';
|
|
5
|
+
import { asErrorLike, buildExecEnv, errorMessage, interruptedExecResult, timeoutExecResult, } from '../../core/runtime.js';
|
|
6
|
+
import { registerSigintSubscriber } from '../../core/subprocess.js';
|
|
7
|
+
import { buildSdkHookCallback } from '../../../eval-core/mocks-runtime.js';
|
|
8
|
+
import { buildClaudeResult } from './protocol.js';
|
|
9
|
+
export { normalizeClaudeMeasurements } from './protocol.js';
|
|
8
10
|
let sdkQuery = null;
|
|
11
|
+
const CLAUDE_AGENT_SDK_PACKAGE = '@anthropic-ai/claude-agent-sdk';
|
|
9
12
|
/**
|
|
10
13
|
* Map ExecutorInput.allowedSkills to SDK query options for skill isolation.
|
|
11
14
|
* undefined → {} (SDK default: full ~/.claude/skills/ discovery)
|
|
@@ -28,7 +31,7 @@ export function buildSdkIsolationOptions(allowedSkills) {
|
|
|
28
31
|
}
|
|
29
32
|
async function getSdkQuery() {
|
|
30
33
|
if (!sdkQuery) {
|
|
31
|
-
const sdk = await import(
|
|
34
|
+
const sdk = await import(CLAUDE_AGENT_SDK_PACKAGE);
|
|
32
35
|
sdkQuery = sdk.query;
|
|
33
36
|
}
|
|
34
37
|
return sdkQuery;
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
import type { ToolCallInfo, TurnInfo } from '../../../types/index.js';
|
|
2
|
+
import type { ClaudeMessage } from './protocol.js';
|
|
3
|
+
export declare function isClaudeResultMessage(message: ClaudeMessage): boolean;
|
|
4
|
+
export declare function extractClaudeTrace(messages: ClaudeMessage[], timestamps?: number[]): {
|
|
5
|
+
turns: TurnInfo[];
|
|
6
|
+
toolCalls: ToolCallInfo[];
|
|
7
|
+
fullNumTurns: number;
|
|
8
|
+
numSubAgents: number;
|
|
9
|
+
};
|
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
import { safeSliceForJson } from '
|
|
2
|
-
import { isToolResultFailureText } from '
|
|
3
|
-
import { normalizeToolIdentity } from '
|
|
4
|
-
export function
|
|
1
|
+
import { safeSliceForJson } from '../../../util/safe-slice.js';
|
|
2
|
+
import { isToolResultFailureText } from '../../../observability/text-signals.js';
|
|
3
|
+
import { normalizeToolIdentity } from '../../../shared/tool-identity.js';
|
|
4
|
+
export function isClaudeResultMessage(message) {
|
|
5
5
|
return message.type === 'result';
|
|
6
6
|
}
|
|
7
|
-
export function
|
|
7
|
+
export function extractClaudeTrace(messages, timestamps) {
|
|
8
8
|
const turns = [];
|
|
9
9
|
const toolCalls = [];
|
|
10
10
|
const pendingToolUse = new Map();
|
|
@@ -1,8 +1,6 @@
|
|
|
1
|
-
import type { ExecutorFn, ExecutorInput, Sample } from '
|
|
2
|
-
|
|
3
|
-
export
|
|
4
|
-
sampleMocks: SampleMockSupport;
|
|
5
|
-
}
|
|
1
|
+
import type { ExecutorFn, ExecutorInput, Sample } from '../../types/index.js';
|
|
2
|
+
import { type ExecutorCapabilities } from './registry.js';
|
|
3
|
+
export type { ExecutorCapabilities, SampleMockSupport } from './registry.js';
|
|
6
4
|
/**
|
|
7
5
|
* Custom script executors receive the OMK_MOCK_* protocol environment and own
|
|
8
6
|
* the final adapter. Built-ins are explicit so unsupported runtimes can never
|
|
@@ -1,20 +1,13 @@
|
|
|
1
|
-
|
|
2
|
-
claude: { sampleMocks: 'native-hooks' },
|
|
3
|
-
'claude-sdk': { sampleMocks: 'native-hooks' },
|
|
4
|
-
codex: { sampleMocks: 'unsupported' },
|
|
5
|
-
'codex-sdk': { sampleMocks: 'unsupported' },
|
|
6
|
-
gemini: { sampleMocks: 'unsupported' },
|
|
7
|
-
'anthropic-api': { sampleMocks: 'unsupported' },
|
|
8
|
-
'openai-api': { sampleMocks: 'unsupported' },
|
|
9
|
-
};
|
|
1
|
+
import { getExecutorDescriptor, } from './registry.js';
|
|
10
2
|
/**
|
|
11
3
|
* Custom script executors receive the OMK_MOCK_* protocol environment and own
|
|
12
4
|
* the final adapter. Built-ins are explicit so unsupported runtimes can never
|
|
13
5
|
* silently turn mock assertions into model failures.
|
|
14
6
|
*/
|
|
15
7
|
export function getExecutorCapabilities(executorName) {
|
|
16
|
-
return
|
|
17
|
-
|
|
8
|
+
return {
|
|
9
|
+
sampleMocks: getExecutorDescriptor(executorName)?.sampleMocks ?? 'delegated-script',
|
|
10
|
+
};
|
|
18
11
|
}
|
|
19
12
|
export function executorSupportsSampleMocks(executorName) {
|
|
20
13
|
return getExecutorCapabilities(executorName).sampleMocks !== 'unsupported';
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
export async function readJsonResponse(response) {
|
|
2
|
+
const rawBody = await response.text();
|
|
3
|
+
if (!rawBody.trim())
|
|
4
|
+
return { data: null, rawBody };
|
|
5
|
+
try {
|
|
6
|
+
return { data: JSON.parse(rawBody), rawBody };
|
|
7
|
+
}
|
|
8
|
+
catch {
|
|
9
|
+
return { data: null, rawBody };
|
|
10
|
+
}
|
|
11
|
+
}
|
|
12
|
+
export function responseBodyPreview(rawBody, maxLength = 500) {
|
|
13
|
+
const normalized = rawBody.replace(/\s+/g, ' ').trim();
|
|
14
|
+
if (!normalized)
|
|
15
|
+
return '';
|
|
16
|
+
return normalized.length > maxLength
|
|
17
|
+
? `${normalized.slice(0, maxLength)}...`
|
|
18
|
+
: normalized;
|
|
19
|
+
}
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
import type { ExecutableExecutorName } from './registry.js';
|
|
2
|
+
export interface OptionalExecutorDependency {
|
|
3
|
+
readonly packageName: string;
|
|
4
|
+
readonly installSpec: string;
|
|
5
|
+
}
|
|
6
|
+
export declare function getOptionalExecutorDependency(executorName: ExecutableExecutorName): OptionalExecutorDependency | undefined;
|
|
7
|
+
export declare function assertOptionalExecutorDependency(executorName: ExecutableExecutorName, available?: (packageName: string) => boolean): void;
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
const OPTIONAL_EXECUTOR_DEPENDENCIES = {
|
|
2
|
+
'claude-sdk': {
|
|
3
|
+
packageName: '@anthropic-ai/claude-agent-sdk',
|
|
4
|
+
installSpec: '@anthropic-ai/claude-agent-sdk@^0.3.143',
|
|
5
|
+
},
|
|
6
|
+
'codex-sdk': {
|
|
7
|
+
packageName: '@openai/codex-sdk',
|
|
8
|
+
installSpec: '@openai/codex-sdk@^0.149.0',
|
|
9
|
+
},
|
|
10
|
+
};
|
|
11
|
+
export function getOptionalExecutorDependency(executorName) {
|
|
12
|
+
return OPTIONAL_EXECUTOR_DEPENDENCIES[executorName];
|
|
13
|
+
}
|
|
14
|
+
function packageAvailable(packageName) {
|
|
15
|
+
try {
|
|
16
|
+
// Use ESM resolution because @openai/codex-sdk only exposes an `import`
|
|
17
|
+
// condition; createRequire().resolve() reports ERR_PACKAGE_PATH_NOT_EXPORTED
|
|
18
|
+
// even when that optional peer is correctly installed.
|
|
19
|
+
import.meta.resolve(packageName);
|
|
20
|
+
return true;
|
|
21
|
+
}
|
|
22
|
+
catch {
|
|
23
|
+
return false;
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
export function assertOptionalExecutorDependency(executorName, available = packageAvailable) {
|
|
27
|
+
const dependency = getOptionalExecutorDependency(executorName);
|
|
28
|
+
if (!dependency || available(dependency.packageName))
|
|
29
|
+
return;
|
|
30
|
+
throw new Error([
|
|
31
|
+
`执行器「${executorName}」需要可选依赖「${dependency.packageName}」,但当前 OMK 安装作用域中未找到。`,
|
|
32
|
+
`本地安装:npm install ${dependency.installSpec}`,
|
|
33
|
+
`全局安装:npm install -g ${dependency.installSpec}`,
|
|
34
|
+
].join('\n'));
|
|
35
|
+
}
|