oh-my-knowledge 0.52.3 → 0.53.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/README.md +13 -1
  2. package/README.zh.md +13 -1
  3. package/dist/assets/agent-skills/omk/SKILL.md +4 -0
  4. package/dist/assets/agent-skills/omk/references/commands.md +1 -1
  5. package/dist/authoring/evolver.js +1 -1
  6. package/dist/authoring/generator.js +1 -1
  7. package/dist/cli/commands/eval/index.js +7 -6
  8. package/dist/cli/commands/install.js +5 -1
  9. package/dist/cli/lib/generation-failure-hint.js +6 -8
  10. package/dist/cli/lib/runtime-defaults.d.ts +2 -0
  11. package/dist/cli/lib/runtime-defaults.js +7 -4
  12. package/dist/dsh-plugin/cordis.patch.yml +3 -0
  13. package/dist/dsh-plugin/host-executor.d.ts +93 -0
  14. package/dist/dsh-plugin/host-executor.js +232 -0
  15. package/dist/dsh-plugin/index.d.ts +28 -0
  16. package/dist/dsh-plugin/index.js +156 -0
  17. package/dist/dsh-plugin/protocol.d.ts +21 -0
  18. package/dist/dsh-plugin/protocol.js +229 -0
  19. package/dist/eval-core/comparability.js +3 -0
  20. package/dist/eval-core/evaluation-execution.js +5 -13
  21. package/dist/eval-core/evaluation-reporting.d.ts +7 -4
  22. package/dist/eval-core/evaluation-reporting.js +39 -18
  23. package/dist/eval-core/judge-independence.d.ts +1 -1
  24. package/dist/eval-core/judge-independence.js +1 -1
  25. package/dist/eval-core/report-document.js +6 -0
  26. package/dist/eval-core/resume-compatibility.d.ts +1 -0
  27. package/dist/eval-core/resume-compatibility.js +5 -4
  28. package/dist/eval-workflows/batch-evaluation-workflow.js +1 -1
  29. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +4 -2
  30. package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
  31. package/dist/eval-workflows/evaluation-pipeline.js +7 -3
  32. package/dist/eval-workflows/run-evaluation.d.ts +4 -2
  33. package/dist/eval-workflows/run-evaluation.js +7 -3
  34. package/dist/executors/{anthropic-api.d.ts → anthropic/api.d.ts} +1 -1
  35. package/dist/executors/{anthropic-api.js → anthropic/api.js} +4 -2
  36. package/dist/executors/{claude-cli.d.ts → anthropic/claude/cli.d.ts} +1 -1
  37. package/dist/executors/{claude-cli.js → anthropic/claude/cli.js} +5 -3
  38. package/dist/executors/anthropic/claude/protocol.d.ts +87 -0
  39. package/dist/executors/{claude-protocol.js → anthropic/claude/protocol.js} +6 -6
  40. package/dist/executors/{claude-sdk.d.ts → anthropic/claude/sdk.d.ts} +2 -2
  41. package/dist/executors/{claude-sdk.js → anthropic/claude/sdk.js} +8 -5
  42. package/dist/executors/anthropic/claude/trace.d.ts +9 -0
  43. package/dist/executors/{claude-sdk-trace.js → anthropic/claude/trace.js} +5 -5
  44. package/dist/executors/{capabilities.d.ts → core/capabilities.d.ts} +3 -5
  45. package/dist/executors/{capabilities.js → core/capabilities.js} +4 -11
  46. package/dist/executors/core/http.d.ts +6 -0
  47. package/dist/executors/core/http.js +19 -0
  48. package/dist/executors/core/limits.d.ts +2 -0
  49. package/dist/executors/core/limits.js +2 -0
  50. package/dist/executors/core/optional-dependencies.d.ts +7 -0
  51. package/dist/executors/core/optional-dependencies.js +35 -0
  52. package/dist/executors/core/registry.d.ts +145 -0
  53. package/dist/executors/core/registry.js +127 -0
  54. package/dist/executors/core/runtime-fingerprint.d.ts +13 -0
  55. package/dist/executors/{runtime-fingerprint.js → core/runtime-fingerprint.js} +119 -59
  56. package/dist/executors/core/runtime.d.ts +12 -0
  57. package/dist/executors/core/runtime.js +61 -0
  58. package/dist/executors/core/subprocess.d.ts +44 -0
  59. package/dist/executors/{shared.js → core/subprocess.js} +20 -156
  60. package/dist/executors/index.d.ts +4 -4
  61. package/dist/executors/index.js +26 -15
  62. package/dist/executors/{openai-api.d.ts → openai/api.d.ts} +1 -1
  63. package/dist/executors/{openai-api.js → openai/api.js} +4 -2
  64. package/dist/executors/{codex-cli.d.ts → openai/codex/cli.d.ts} +3 -3
  65. package/dist/executors/{codex-cli.js → openai/codex/cli.js} +5 -3
  66. package/dist/executors/openai/codex/protocol.d.ts +72 -0
  67. package/dist/executors/{codex-protocol.js → openai/codex/protocol.js} +33 -2
  68. package/dist/executors/{codex-sdk.d.ts → openai/codex/sdk.d.ts} +18 -4
  69. package/dist/executors/{codex-sdk.js → openai/codex/sdk.js} +7 -4
  70. package/dist/executors/{codex-cli-trace.d.ts → openai/codex/trace.d.ts} +2 -2
  71. package/dist/executors/{codex-cli-trace.js → openai/codex/trace.js} +4 -4
  72. package/dist/executors/{script.d.ts → script/index.d.ts} +1 -1
  73. package/dist/executors/{script.js → script/index.js} +6 -4
  74. package/dist/grading/judge.d.ts +1 -1
  75. package/dist/grading/judge.js +1 -1
  76. package/dist/renderer/html-renderer.js +30 -11
  77. package/dist/types/executor.d.ts +13 -2
  78. package/dist/types/judge.d.ts +2 -2
  79. package/package.json +21 -5
  80. package/dist/executors/claude-protocol.d.ts +0 -28
  81. package/dist/executors/claude-sdk-trace.d.ts +0 -9
  82. package/dist/executors/codex-protocol.d.ts +0 -24
  83. package/dist/executors/gemini.d.ts +0 -2
  84. package/dist/executors/gemini.js +0 -156
  85. package/dist/executors/runtime-fingerprint.d.ts +0 -6
  86. package/dist/executors/shared.d.ts +0 -226
  87. /package/dist/executors/{script-command.d.ts → script/command.d.ts} +0 -0
  88. /package/dist/executors/{script-command.js → script/command.js} +0 -0
@@ -3,7 +3,7 @@ import { buildExecutorRuntimesByVariant, EVALUATION_REPORT_SCHEMA_VERSION, getCl
3
3
  import { hashSample } from './sample-fingerprint.js';
4
4
  import { getJudgePromptHash } from '../grading/judge.js';
5
5
  import { getDiagnosticPromptHash, resolveDiagnosticTarget, } from '../grading/diagnostic.js';
6
- import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
6
+ import { resolveExecutorRuntimeFingerprint } from '../executors/core/runtime-fingerprint.js';
7
7
  function canonicalStringify(value) {
8
8
  if (value === undefined)
9
9
  return 'undefined';
@@ -73,6 +73,7 @@ export function checkResumeCompatibility(report, current) {
73
73
  skillDir: current.skillDir,
74
74
  timeoutMs: current.timeoutMs,
75
75
  },
76
+ executor: current.executorOverrides?.[current.executorName],
76
77
  });
77
78
  check('meta.executorRuntimes', runtimeFingerprints(report.meta.executorRuntimes), runtimeFingerprints(expectedExecutorRuntimes));
78
79
  const actualJudges = report.meta.judgeModels.map((judge) => ({
@@ -85,9 +86,9 @@ export function checkResumeCompatibility(report, current) {
85
86
  model: judge.model,
86
87
  ...(!current.noJudge
87
88
  ? {
88
- runtime: getExecutorRuntimeFingerprint(judge.executor, judge.model, {
89
+ runtime: resolveExecutorRuntimeFingerprint(judge.executor, judge.model, {
89
90
  skillDir: current.skillDir,
90
- }).fingerprint,
91
+ }, current.executorOverrides?.[judge.executor]).fingerprint,
91
92
  }
92
93
  : {}),
93
94
  }));
@@ -102,7 +103,7 @@ export function checkResumeCompatibility(report, current) {
102
103
  enabled: true,
103
104
  executor: diagnosticTarget.executor,
104
105
  model: diagnosticTarget.model,
105
- runtime: getExecutorRuntimeFingerprint(diagnosticTarget.executor, diagnosticTarget.model).fingerprint,
106
+ runtime: resolveExecutorRuntimeFingerprint(diagnosticTarget.executor, diagnosticTarget.model, {}, current.executorOverrides?.[diagnosticTarget.executor]).fingerprint,
106
107
  promptHash: getDiagnosticPromptHash(),
107
108
  }
108
109
  : { enabled: false };
@@ -1,7 +1,7 @@
1
1
  import { dirname } from 'node:path';
2
2
  import { DEFAULT_OUTPUT_DIR, EVALUATION_REPORT_SCHEMA_VERSION, generateRunId, getCliVersion, getGitInfo, persistReport, } from '../eval-core/evaluation-reporting.js';
3
3
  import { buildEvaluationRequest, createEvaluationRun, createSucceededJob, finalizeEvaluationRun } from '../eval-core/evaluation-job.js';
4
- import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
4
+ import { getExecutorRuntimeFingerprint } from '../executors/core/runtime-fingerprint.js';
5
5
  import { createFileJobStore } from '../server/job-store.js';
6
6
  import { DEFAULT_JOBS_DIR } from '../eval-core/default-dirs.js';
7
7
  import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
@@ -16,6 +16,7 @@
16
16
  import { existsSync, readdirSync } from 'node:fs';
17
17
  import { homedir } from 'node:os';
18
18
  import { join, relative } from 'node:path';
19
+ import { executorFamily } from '../../executors/core/registry.js';
19
20
  import { tEvalWorkflowMessage } from '../messages.js';
20
21
  export function buildPowerWarnings(sampleCount, repeat, lang = 'zh') {
21
22
  const warnings = [];
@@ -43,11 +44,12 @@ export function buildIsolationWarnings(artifacts, strictBaseline, options) {
43
44
  if (!hasUnisolatedBaseline)
44
45
  return [];
45
46
  const executorName = options.executorName;
47
+ const family = executorFamily(executorName);
46
48
  const home = options.homeDir ?? homedir();
47
49
  const cwd = options.cwd ?? process.cwd();
48
- const roots = executorName === 'claude' || executorName === 'claude-sdk'
50
+ const roots = family === 'claude'
49
51
  ? [join(home, '.claude', 'skills'), join(cwd, '.claude', 'skills')]
50
- : executorName === 'codex' || executorName === 'codex-sdk'
52
+ : family === 'codex'
51
53
  ? [
52
54
  join(home, '.agents', 'skills'),
53
55
  join(home, '.codex', 'skills'),
@@ -43,6 +43,8 @@ export interface EvaluationPipelineOptions {
43
43
  judgeExecutorName: string;
44
44
  executor: ExecutorFn;
45
45
  judgeExecutor: ExecutorFn;
46
+ /** Embedding-host executors that must not be reconstructed from a CLI name. */
47
+ executorOverrides?: Readonly<Record<string, ExecutorFn>>;
46
48
  outputDir?: string | null;
47
49
  project?: string;
48
50
  owner?: string;
@@ -91,7 +93,7 @@ export interface EvaluationPipelineOptions {
91
93
  noDiagnostic?: boolean;
92
94
  }
93
95
  type VariantResult = import('../types/index.js').VariantResult;
94
- export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
96
+ export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, executorOverrides, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
95
97
  report: Report;
96
98
  filePath: string | null;
97
99
  }>;
@@ -28,11 +28,11 @@ import { finalizeSuccessfulRun, initializeEvaluationRunState, persistFailedJob,
28
28
  import { finalizeEvaluationReport } from './evaluation-pipeline/report-finalize.js';
29
29
  import { emitIsolationWarnings, emitPowerWarnings } from './evaluation-pipeline/preflight-warnings.js';
30
30
  import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
31
- import { assertSamplesCompatibleWithExecutor } from '../executors/capabilities.js';
31
+ import { assertSamplesCompatibleWithExecutor } from '../executors/core/capabilities.js';
32
32
  // 兼容 re-export:测试与 run-evaluation.ts 动态 import 仍打 evaluation-pipeline.js
33
33
  export { buildPowerWarnings, buildIsolationWarnings } from './evaluation-pipeline/preflight-warnings.js';
34
34
  export { _computeTestSetHashForTest } from './evaluation-pipeline/test-set-hash.js';
35
- export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, concurrency = 1, timeoutMs, noCache = false, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, verbose = false, retry = 0, existingResults,
35
+ export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, executorOverrides, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, concurrency = 1, timeoutMs, noCache = false, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, verbose = false, retry = 0, existingResults,
36
36
  // requires 现在由 runEvaluation 上游传给 doctor 处理; eval-pipeline 不再用
37
37
  // 但保留接口字段,避免破 programmatic API (类型层面接收, 内部忽略)
38
38
  requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
@@ -87,7 +87,9 @@ requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRep
87
87
  resolvedJudgeExecutors = Object.create(null);
88
88
  for (const jc of resolvedJudgeModels) {
89
89
  if (!ownRecordValue(resolvedJudgeExecutors, jc.executor)) {
90
- setOwnRecordValue(resolvedJudgeExecutors, jc.executor, createExecutor(jc.executor));
90
+ setOwnRecordValue(resolvedJudgeExecutors, jc.executor, executorOverrides?.[jc.executor]
91
+ ?? (jc.executor === executorName ? executor : undefined)
92
+ ?? createExecutor(jc.executor));
91
93
  }
92
94
  }
93
95
  }
@@ -156,6 +158,8 @@ requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRep
156
158
  run,
157
159
  job,
158
160
  layeredStats,
161
+ executor,
162
+ judgeExecutors: resolvedJudgeExecutors,
159
163
  }),
160
164
  results,
161
165
  artifacts,
@@ -1,4 +1,4 @@
1
- import type { Artifact, BatchEvaluationReport, EvaluationReport, JobStore, ProgressCallback, Report, VarianceData, VariantSpec } from '../types/index.js';
1
+ import type { Artifact, BatchEvaluationReport, EvaluationReport, ExecutorFn, JobStore, ProgressCallback, Report, VarianceData, VariantSpec } from '../types/index.js';
2
2
  export interface SkillProgressInfo {
3
3
  phase: string;
4
4
  skill: string;
@@ -17,6 +17,8 @@ interface CommonEvaluationOptions {
17
17
  timeoutMs?: number;
18
18
  /** Core API is runtime-neutral: CLI/default resolution happens before this boundary. */
19
19
  executorName: string;
20
+ /** Same-process executors supplied by an embedding host, keyed by executor name. */
21
+ executorOverrides?: Readonly<Record<string, ExecutorFn>>;
20
22
  jobStore?: JobStore | null;
21
23
  persistJob?: boolean;
22
24
  onProgress?: ProgressCallback | null;
@@ -122,7 +124,7 @@ export interface DryRunReport extends DryRunBase {
122
124
  samplesPath: string;
123
125
  tasks: DryRunTask[];
124
126
  }
125
- export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
127
+ export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, executorOverrides, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
126
128
  report: Report | DryRunReport;
127
129
  filePath: string | null;
128
130
  }>;
@@ -10,7 +10,7 @@ import { checkResumeCompatibility } from '../eval-core/resume-compatibility.js';
10
10
  import { findSaturationPoint } from '../analysis/saturation.js';
11
11
  import { bootstrapMeanCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
12
12
  import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
13
- export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
13
+ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName, executorOverrides, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
14
14
  // Unified judgeModels → derive single-judge fields for downstream pipeline / grading
15
15
  // (which still operate on string `judgeModel` + `judgeExecutorName` fields per call).
16
16
  const effectiveJudgeModels = judgeModels && judgeModels.length > 0
@@ -123,6 +123,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
123
123
  samplesBaseDir,
124
124
  tasks,
125
125
  artifacts: resolvedArtifacts,
126
+ executorOverrides,
126
127
  });
127
128
  if (compatibility.compatible) {
128
129
  const sourceBySample = new Map(existing.results.map((entry) => [entry.sample_id, entry.variants]));
@@ -159,8 +160,10 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
159
160
  const effectiveSkipConnectivity = existingResults !== undefined
160
161
  ? true
161
162
  : skipConnectivity;
162
- const executor = createExecutor(executorName);
163
- const judgeExecutor = createExecutor(judgeExecutorName || executorName);
163
+ const executor = executorOverrides?.[executorName] ?? createExecutor(executorName);
164
+ const effectiveJudgeExecutorName = judgeExecutorName || executorName;
165
+ const judgeExecutor = executorOverrides?.[effectiveJudgeExecutorName]
166
+ ?? createExecutor(effectiveJudgeExecutorName);
164
167
  return executeEvaluationPipeline({
165
168
  samplesPath,
166
169
  samplesBaseDir,
@@ -176,6 +179,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
176
179
  judgeExecutorName: judgeExecutorName || executorName,
177
180
  executor,
178
181
  judgeExecutor,
182
+ executorOverrides,
179
183
  outputDir,
180
184
  project,
181
185
  owner,
@@ -1,2 +1,2 @@
1
- import type { ExecResult, ExecutorInput } from '../types/index.js';
1
+ import type { ExecResult, ExecutorInput } from '../../types/index.js';
2
2
  export declare function anthropicApiExecutor({ model, system, prompt, timeoutMs }: ExecutorInput): Promise<ExecResult>;
@@ -1,5 +1,7 @@
1
- import { asErrorLike, DEFAULT_TIMEOUT_MS, errorMessage, readJsonResponse, responseBodyPreview, } from './shared.js';
2
- import { optionalTokenCount } from '../shared/token-usage.js';
1
+ import { optionalTokenCount } from '../../shared/token-usage.js';
2
+ import { DEFAULT_TIMEOUT_MS } from '../core/limits.js';
3
+ import { readJsonResponse, responseBodyPreview } from '../core/http.js';
4
+ import { asErrorLike, errorMessage } from '../core/runtime.js';
3
5
  export async function anthropicApiExecutor({ model, system, prompt, timeoutMs = DEFAULT_TIMEOUT_MS }) {
4
6
  const apiKey = process.env.ANTHROPIC_API_KEY;
5
7
  if (!apiKey)
@@ -1,2 +1,2 @@
1
- import type { ExecResult, ExecutorInput } from '../types/index.js';
1
+ import type { ExecResult, ExecutorInput } from '../../../types/index.js';
2
2
  export declare function claudeCliExecutor({ model, system, prompt, cwd, skillDir, timeoutMs, allowedSkills, mocks, mocksBaseDir, mocksStrict, lean, effort }: ExecutorInput): Promise<ExecResult>;
@@ -1,6 +1,8 @@
1
- import { buildExecEnv, DEFAULT_TIMEOUT_MS, errorMessage, interruptedExecResult, MAX_BUFFER, spawnWithSigintPropagation, timeoutExecResult, } from './shared.js';
2
- import { materializeForCliConfigDir } from '../eval-core/mocks-runtime.js';
3
- import { buildClaudeResult, parseClaudeStreamJson } from './claude-protocol.js';
1
+ import { materializeForCliConfigDir } from '../../../eval-core/mocks-runtime.js';
2
+ import { buildClaudeResult, parseClaudeStreamJson } from './protocol.js';
3
+ import { DEFAULT_TIMEOUT_MS, MAX_BUFFER } from '../../core/limits.js';
4
+ import { buildExecEnv, errorMessage, interruptedExecResult, timeoutExecResult, } from '../../core/runtime.js';
5
+ import { spawnWithSigintPropagation } from '../../core/subprocess.js';
4
6
  // claude CLI 用 `--disable-slash-commands` (文档:"Disable all skills") +
5
7
  // `--disallowedTools Skill` 实现与 SDK 等价的完全隔离。非空 skill 白名单不再支持
6
8
  // (它无法真正隔离,见 claude-sdk.ts buildSdkIsolationOptions),所以 [name1, ...] 必须 throw。
@@ -0,0 +1,87 @@
1
+ import type { ExecResult } from '../../../types/index.js';
2
+ interface ClaudeTokenUsage {
3
+ input_tokens?: number;
4
+ output_tokens?: number;
5
+ cache_read_input_tokens?: number;
6
+ cache_creation_input_tokens?: number;
7
+ }
8
+ export interface ClaudeSdkQueryOptions {
9
+ model?: string;
10
+ systemPrompt?: string;
11
+ cwd: string;
12
+ permissionMode: 'bypassPermissions';
13
+ allowDangerouslySkipPermissions: true;
14
+ abortController: AbortController;
15
+ env: NodeJS.ProcessEnv;
16
+ }
17
+ export interface ClaudeSdkQueryInput {
18
+ prompt: string;
19
+ options: ClaudeSdkQueryOptions;
20
+ }
21
+ export interface ClaudeMessage {
22
+ type: string;
23
+ message?: {
24
+ role?: string;
25
+ content?: Array<{
26
+ type: string;
27
+ text?: string;
28
+ id?: string;
29
+ name?: string;
30
+ input?: unknown;
31
+ }>;
32
+ };
33
+ tool_use_id?: string;
34
+ content?: string | Array<{
35
+ type: string;
36
+ text?: string;
37
+ }>;
38
+ is_error?: boolean;
39
+ }
40
+ export interface ClaudeResultMessage extends ClaudeMessage {
41
+ type: 'result';
42
+ result?: string;
43
+ usage?: ClaudeTokenUsage;
44
+ total_cost_usd?: number;
45
+ duration_api_ms?: number;
46
+ duration_ms?: number;
47
+ num_turns?: number;
48
+ stop_reason?: string | null;
49
+ modelUsage?: Record<string, {
50
+ inputTokens?: number;
51
+ outputTokens?: number;
52
+ cacheReadInputTokens?: number;
53
+ cacheCreationInputTokens?: number;
54
+ }>;
55
+ subtype?: string;
56
+ errors?: string[];
57
+ }
58
+ export interface ClaudeSdkModule {
59
+ query: (opts: ClaudeSdkQueryInput) => AsyncIterable<ClaudeMessage>;
60
+ }
61
+ export interface ClaudeMeasurements {
62
+ durationMs: number;
63
+ durationApiMs: number;
64
+ inputTokens: number;
65
+ outputTokens: number;
66
+ cacheReadTokens: number;
67
+ cacheCreationTokens: number;
68
+ costUSD: number;
69
+ numTurns: number;
70
+ }
71
+ export declare function normalizeClaudeMeasurements(result: ClaudeResultMessage): ClaudeMeasurements | {
72
+ error: string;
73
+ };
74
+ export interface ClaudeStreamParseResult {
75
+ messages: ClaudeMessage[];
76
+ malformedLineCount: number;
77
+ }
78
+ export declare function parseClaudeStreamJson(stdout: string): ClaudeStreamParseResult;
79
+ export declare function buildClaudeResult(options: {
80
+ messages: ClaudeMessage[];
81
+ wallClockDurationMs: number;
82
+ source: 'claude stream-json' | 'claude-sdk';
83
+ malformedLineCount?: number;
84
+ forcedError?: string;
85
+ messageTimestamps?: number[];
86
+ }): ExecResult;
87
+ export {};
@@ -1,6 +1,6 @@
1
- import { checkedSumTokenCounts, nonNegativeMetric, optionalTokenCount, } from '../shared/token-usage.js';
2
- import { extractAgentTrace, isClaudeSdkResultMessage } from './claude-sdk-trace.js';
3
- export function normalizeClaudeSdkMeasurements(result) {
1
+ import { checkedSumTokenCounts, nonNegativeMetric, optionalTokenCount, } from '../../../shared/token-usage.js';
2
+ import { extractClaudeTrace, isClaudeResultMessage } from './trace.js';
3
+ export function normalizeClaudeMeasurements(result) {
4
4
  const durationMs = optionalTokenCount(result.duration_ms);
5
5
  const durationApiMs = optionalTokenCount(result.duration_api_ms);
6
6
  const numTurns = optionalTokenCount(result.num_turns);
@@ -96,8 +96,8 @@ export function parseClaudeStreamJson(stdout) {
96
96
  }
97
97
  export function buildClaudeResult(options) {
98
98
  const { messages, wallClockDurationMs, source, malformedLineCount = 0, forcedError, messageTimestamps, } = options;
99
- const resultMessages = messages.filter(isClaudeSdkResultMessage);
100
- const trace = extractAgentTrace(messages, messageTimestamps);
99
+ const resultMessages = messages.filter(isClaudeResultMessage);
100
+ const trace = extractClaudeTrace(messages, messageTimestamps);
101
101
  const traceFields = {
102
102
  fullNumTurns: trace.fullNumTurns,
103
103
  numSubAgents: trace.numSubAgents,
@@ -133,7 +133,7 @@ export function buildClaudeResult(options) {
133
133
  };
134
134
  }
135
135
  const result = resultMessages[0];
136
- const measurements = normalizeClaudeSdkMeasurements(result);
136
+ const measurements = normalizeClaudeMeasurements(result);
137
137
  if ('error' in measurements) {
138
138
  errors.push(measurements.error);
139
139
  return {
@@ -1,5 +1,5 @@
1
- import type { ExecResult, ExecutorInput } from '../types/index.js';
2
- export { normalizeClaudeSdkMeasurements } from './claude-protocol.js';
1
+ import type { ExecResult, ExecutorInput } from '../../../types/index.js';
2
+ export { normalizeClaudeMeasurements } from './protocol.js';
3
3
  /**
4
4
  * Map ExecutorInput.allowedSkills to SDK query options for skill isolation.
5
5
  * undefined → {} (SDK default: full ~/.claude/skills/ discovery)
@@ -1,11 +1,14 @@
1
1
  import { join } from 'node:path';
2
2
  import { mkdirSync, writeFileSync } from 'node:fs';
3
3
  import { tmpdir } from 'node:os';
4
- import { asErrorLike, buildExecEnv, DEFAULT_TIMEOUT_MS, errorMessage, interruptedExecResult, registerSigintSubscriber, timeoutExecResult, } from './shared.js';
5
- import { buildSdkHookCallback } from '../eval-core/mocks-runtime.js';
6
- import { buildClaudeResult } from './claude-protocol.js';
7
- export { normalizeClaudeSdkMeasurements } from './claude-protocol.js';
4
+ import { DEFAULT_TIMEOUT_MS } from '../../core/limits.js';
5
+ import { asErrorLike, buildExecEnv, errorMessage, interruptedExecResult, timeoutExecResult, } from '../../core/runtime.js';
6
+ import { registerSigintSubscriber } from '../../core/subprocess.js';
7
+ import { buildSdkHookCallback } from '../../../eval-core/mocks-runtime.js';
8
+ import { buildClaudeResult } from './protocol.js';
9
+ export { normalizeClaudeMeasurements } from './protocol.js';
8
10
  let sdkQuery = null;
11
+ const CLAUDE_AGENT_SDK_PACKAGE = '@anthropic-ai/claude-agent-sdk';
9
12
  /**
10
13
  * Map ExecutorInput.allowedSkills to SDK query options for skill isolation.
11
14
  * undefined → {} (SDK default: full ~/.claude/skills/ discovery)
@@ -28,7 +31,7 @@ export function buildSdkIsolationOptions(allowedSkills) {
28
31
  }
29
32
  async function getSdkQuery() {
30
33
  if (!sdkQuery) {
31
- const sdk = await import('@anthropic-ai/claude-agent-sdk');
34
+ const sdk = await import(CLAUDE_AGENT_SDK_PACKAGE);
32
35
  sdkQuery = sdk.query;
33
36
  }
34
37
  return sdkQuery;
@@ -0,0 +1,9 @@
1
+ import type { ToolCallInfo, TurnInfo } from '../../../types/index.js';
2
+ import type { ClaudeMessage } from './protocol.js';
3
+ export declare function isClaudeResultMessage(message: ClaudeMessage): boolean;
4
+ export declare function extractClaudeTrace(messages: ClaudeMessage[], timestamps?: number[]): {
5
+ turns: TurnInfo[];
6
+ toolCalls: ToolCallInfo[];
7
+ fullNumTurns: number;
8
+ numSubAgents: number;
9
+ };
@@ -1,10 +1,10 @@
1
- import { safeSliceForJson } from '../util/safe-slice.js';
2
- import { isToolResultFailureText } from '../observability/text-signals.js';
3
- import { normalizeToolIdentity } from '../shared/tool-identity.js';
4
- export function isClaudeSdkResultMessage(message) {
1
+ import { safeSliceForJson } from '../../../util/safe-slice.js';
2
+ import { isToolResultFailureText } from '../../../observability/text-signals.js';
3
+ import { normalizeToolIdentity } from '../../../shared/tool-identity.js';
4
+ export function isClaudeResultMessage(message) {
5
5
  return message.type === 'result';
6
6
  }
7
- export function extractAgentTrace(messages, timestamps) {
7
+ export function extractClaudeTrace(messages, timestamps) {
8
8
  const turns = [];
9
9
  const toolCalls = [];
10
10
  const pendingToolUse = new Map();
@@ -1,8 +1,6 @@
1
- import type { ExecutorFn, ExecutorInput, Sample } from '../types/index.js';
2
- export type SampleMockSupport = 'native-hooks' | 'delegated-script' | 'unsupported';
3
- export interface ExecutorCapabilities {
4
- sampleMocks: SampleMockSupport;
5
- }
1
+ import type { ExecutorFn, ExecutorInput, Sample } from '../../types/index.js';
2
+ import { type ExecutorCapabilities } from './registry.js';
3
+ export type { ExecutorCapabilities, SampleMockSupport } from './registry.js';
6
4
  /**
7
5
  * Custom script executors receive the OMK_MOCK_* protocol environment and own
8
6
  * the final adapter. Built-ins are explicit so unsupported runtimes can never
@@ -1,20 +1,13 @@
1
- const BUILTIN_CAPABILITIES = {
2
- claude: { sampleMocks: 'native-hooks' },
3
- 'claude-sdk': { sampleMocks: 'native-hooks' },
4
- codex: { sampleMocks: 'unsupported' },
5
- 'codex-sdk': { sampleMocks: 'unsupported' },
6
- gemini: { sampleMocks: 'unsupported' },
7
- 'anthropic-api': { sampleMocks: 'unsupported' },
8
- 'openai-api': { sampleMocks: 'unsupported' },
9
- };
1
+ import { getExecutorDescriptor, } from './registry.js';
10
2
  /**
11
3
  * Custom script executors receive the OMK_MOCK_* protocol environment and own
12
4
  * the final adapter. Built-ins are explicit so unsupported runtimes can never
13
5
  * silently turn mock assertions into model failures.
14
6
  */
15
7
  export function getExecutorCapabilities(executorName) {
16
- return BUILTIN_CAPABILITIES[executorName]
17
- ?? { sampleMocks: 'delegated-script' };
8
+ return {
9
+ sampleMocks: getExecutorDescriptor(executorName)?.sampleMocks ?? 'delegated-script',
10
+ };
18
11
  }
19
12
  export function executorSupportsSampleMocks(executorName) {
20
13
  return getExecutorCapabilities(executorName).sampleMocks !== 'unsupported';
@@ -0,0 +1,6 @@
1
+ export interface JsonResponseBody<T> {
2
+ data: T | null;
3
+ rawBody: string;
4
+ }
5
+ export declare function readJsonResponse<T>(response: Response): Promise<JsonResponseBody<T>>;
6
+ export declare function responseBodyPreview(rawBody: string, maxLength?: number): string;
@@ -0,0 +1,19 @@
1
+ export async function readJsonResponse(response) {
2
+ const rawBody = await response.text();
3
+ if (!rawBody.trim())
4
+ return { data: null, rawBody };
5
+ try {
6
+ return { data: JSON.parse(rawBody), rawBody };
7
+ }
8
+ catch {
9
+ return { data: null, rawBody };
10
+ }
11
+ }
12
+ export function responseBodyPreview(rawBody, maxLength = 500) {
13
+ const normalized = rawBody.replace(/\s+/g, ' ').trim();
14
+ if (!normalized)
15
+ return '';
16
+ return normalized.length > maxLength
17
+ ? `${normalized.slice(0, maxLength)}...`
18
+ : normalized;
19
+ }
@@ -0,0 +1,2 @@
1
+ export declare const DEFAULT_TIMEOUT_MS = 600000;
2
+ export declare const MAX_BUFFER: number;
@@ -0,0 +1,2 @@
1
+ export const DEFAULT_TIMEOUT_MS = 600_000;
2
+ export const MAX_BUFFER = 10 * 1024 * 1024;
@@ -0,0 +1,7 @@
1
+ import type { ExecutableExecutorName } from './registry.js';
2
+ export interface OptionalExecutorDependency {
3
+ readonly packageName: string;
4
+ readonly installSpec: string;
5
+ }
6
+ export declare function getOptionalExecutorDependency(executorName: ExecutableExecutorName): OptionalExecutorDependency | undefined;
7
+ export declare function assertOptionalExecutorDependency(executorName: ExecutableExecutorName, available?: (packageName: string) => boolean): void;
@@ -0,0 +1,35 @@
1
+ const OPTIONAL_EXECUTOR_DEPENDENCIES = {
2
+ 'claude-sdk': {
3
+ packageName: '@anthropic-ai/claude-agent-sdk',
4
+ installSpec: '@anthropic-ai/claude-agent-sdk@^0.3.143',
5
+ },
6
+ 'codex-sdk': {
7
+ packageName: '@openai/codex-sdk',
8
+ installSpec: '@openai/codex-sdk@^0.149.0',
9
+ },
10
+ };
11
+ export function getOptionalExecutorDependency(executorName) {
12
+ return OPTIONAL_EXECUTOR_DEPENDENCIES[executorName];
13
+ }
14
+ function packageAvailable(packageName) {
15
+ try {
16
+ // Use ESM resolution because @openai/codex-sdk only exposes an `import`
17
+ // condition; createRequire().resolve() reports ERR_PACKAGE_PATH_NOT_EXPORTED
18
+ // even when that optional peer is correctly installed.
19
+ import.meta.resolve(packageName);
20
+ return true;
21
+ }
22
+ catch {
23
+ return false;
24
+ }
25
+ }
26
+ export function assertOptionalExecutorDependency(executorName, available = packageAvailable) {
27
+ const dependency = getOptionalExecutorDependency(executorName);
28
+ if (!dependency || available(dependency.packageName))
29
+ return;
30
+ throw new Error([
31
+ `执行器「${executorName}」需要可选依赖「${dependency.packageName}」,但当前 OMK 安装作用域中未找到。`,
32
+ `本地安装:npm install ${dependency.installSpec}`,
33
+ `全局安装:npm install -g ${dependency.installSpec}`,
34
+ ].join('\n'));
35
+ }