oh-my-knowledge 0.52.3 → 0.54.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/README.md +21 -1
  2. package/README.zh.md +21 -1
  3. package/dist/assets/agent-skills/omk/SKILL.md +6 -0
  4. package/dist/assets/agent-skills/omk/references/commands.md +1 -1
  5. package/dist/authoring/evolver.js +1 -1
  6. package/dist/authoring/generator.js +1 -1
  7. package/dist/cli/commands/eval/index.js +7 -6
  8. package/dist/cli/commands/install.js +5 -1
  9. package/dist/cli/lib/generation-failure-hint.js +6 -8
  10. package/dist/cli/lib/runtime-defaults.d.ts +2 -0
  11. package/dist/cli/lib/runtime-defaults.js +7 -4
  12. package/dist/dsh-plugin/cordis.patch.yml +3 -0
  13. package/dist/dsh-plugin/host-executor.d.ts +93 -0
  14. package/dist/dsh-plugin/host-executor.js +232 -0
  15. package/dist/dsh-plugin/index.d.ts +30 -0
  16. package/dist/dsh-plugin/index.js +275 -0
  17. package/dist/dsh-plugin/observe.d.ts +47 -0
  18. package/dist/dsh-plugin/observe.js +359 -0
  19. package/dist/dsh-plugin/protocol.d.ts +21 -0
  20. package/dist/dsh-plugin/protocol.js +229 -0
  21. package/dist/dsh-plugin/trace-adapter.d.ts +48 -0
  22. package/dist/dsh-plugin/trace-adapter.js +590 -0
  23. package/dist/eval-core/comparability.js +3 -0
  24. package/dist/eval-core/evaluation-execution.js +5 -13
  25. package/dist/eval-core/evaluation-reporting.d.ts +7 -4
  26. package/dist/eval-core/evaluation-reporting.js +39 -18
  27. package/dist/eval-core/judge-independence.d.ts +1 -1
  28. package/dist/eval-core/judge-independence.js +1 -1
  29. package/dist/eval-core/report-document.js +6 -0
  30. package/dist/eval-core/resume-compatibility.d.ts +1 -0
  31. package/dist/eval-core/resume-compatibility.js +5 -4
  32. package/dist/eval-workflows/batch-evaluation-workflow.js +1 -1
  33. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +4 -2
  34. package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
  35. package/dist/eval-workflows/evaluation-pipeline.js +7 -3
  36. package/dist/eval-workflows/run-evaluation.d.ts +4 -2
  37. package/dist/eval-workflows/run-evaluation.js +7 -3
  38. package/dist/executors/{anthropic-api.d.ts → anthropic/api.d.ts} +1 -1
  39. package/dist/executors/{anthropic-api.js → anthropic/api.js} +4 -2
  40. package/dist/executors/{claude-cli.d.ts → anthropic/claude/cli.d.ts} +1 -1
  41. package/dist/executors/{claude-cli.js → anthropic/claude/cli.js} +5 -3
  42. package/dist/executors/anthropic/claude/protocol.d.ts +87 -0
  43. package/dist/executors/{claude-protocol.js → anthropic/claude/protocol.js} +6 -6
  44. package/dist/executors/{claude-sdk.d.ts → anthropic/claude/sdk.d.ts} +2 -2
  45. package/dist/executors/{claude-sdk.js → anthropic/claude/sdk.js} +8 -5
  46. package/dist/executors/anthropic/claude/trace.d.ts +9 -0
  47. package/dist/executors/{claude-sdk-trace.js → anthropic/claude/trace.js} +5 -5
  48. package/dist/executors/{capabilities.d.ts → core/capabilities.d.ts} +3 -5
  49. package/dist/executors/{capabilities.js → core/capabilities.js} +4 -11
  50. package/dist/executors/core/http.d.ts +6 -0
  51. package/dist/executors/core/http.js +19 -0
  52. package/dist/executors/core/limits.d.ts +2 -0
  53. package/dist/executors/core/limits.js +2 -0
  54. package/dist/executors/core/optional-dependencies.d.ts +7 -0
  55. package/dist/executors/core/optional-dependencies.js +35 -0
  56. package/dist/executors/core/registry.d.ts +145 -0
  57. package/dist/executors/core/registry.js +127 -0
  58. package/dist/executors/core/runtime-fingerprint.d.ts +13 -0
  59. package/dist/executors/{runtime-fingerprint.js → core/runtime-fingerprint.js} +119 -59
  60. package/dist/executors/core/runtime.d.ts +12 -0
  61. package/dist/executors/core/runtime.js +61 -0
  62. package/dist/executors/core/subprocess.d.ts +44 -0
  63. package/dist/executors/{shared.js → core/subprocess.js} +20 -156
  64. package/dist/executors/index.d.ts +4 -4
  65. package/dist/executors/index.js +26 -15
  66. package/dist/executors/{openai-api.d.ts → openai/api.d.ts} +1 -1
  67. package/dist/executors/{openai-api.js → openai/api.js} +4 -2
  68. package/dist/executors/{codex-cli.d.ts → openai/codex/cli.d.ts} +3 -3
  69. package/dist/executors/{codex-cli.js → openai/codex/cli.js} +5 -3
  70. package/dist/executors/openai/codex/protocol.d.ts +72 -0
  71. package/dist/executors/{codex-protocol.js → openai/codex/protocol.js} +33 -2
  72. package/dist/executors/{codex-sdk.d.ts → openai/codex/sdk.d.ts} +18 -4
  73. package/dist/executors/{codex-sdk.js → openai/codex/sdk.js} +7 -4
  74. package/dist/executors/{codex-cli-trace.d.ts → openai/codex/trace.d.ts} +2 -2
  75. package/dist/executors/{codex-cli-trace.js → openai/codex/trace.js} +4 -4
  76. package/dist/executors/{script.d.ts → script/index.d.ts} +1 -1
  77. package/dist/executors/{script.js → script/index.js} +6 -4
  78. package/dist/grading/judge.d.ts +1 -1
  79. package/dist/grading/judge.js +1 -1
  80. package/dist/observability/conversation-catalog.js +1 -1
  81. package/dist/observability/experience.js +4 -0
  82. package/dist/observability/inbox.d.ts +4 -1
  83. package/dist/observability/inbox.js +7 -2
  84. package/dist/observability/trace-ir.d.ts +6 -3
  85. package/dist/observability/turn-index.js +4 -0
  86. package/dist/renderer/conversation-renderer.js +20 -6
  87. package/dist/renderer/html-renderer.js +30 -11
  88. package/dist/renderer/knowledge-debugger-renderer.js +6 -1
  89. package/dist/renderer/observation-inbox-renderer.js +2 -2
  90. package/dist/renderer/trajectory-live.d.ts +1 -0
  91. package/dist/renderer/trajectory-live.js +5 -2
  92. package/dist/shared/trace-source-kind.js +1 -0
  93. package/dist/types/executor.d.ts +13 -2
  94. package/dist/types/judge.d.ts +2 -2
  95. package/dist/types/observability.d.ts +1 -1
  96. package/dist/types/trace.d.ts +1 -1
  97. package/package.json +25 -5
  98. package/dist/executors/claude-protocol.d.ts +0 -28
  99. package/dist/executors/claude-sdk-trace.d.ts +0 -9
  100. package/dist/executors/codex-protocol.d.ts +0 -24
  101. package/dist/executors/gemini.d.ts +0 -2
  102. package/dist/executors/gemini.js +0 -156
  103. package/dist/executors/runtime-fingerprint.d.ts +0 -6
  104. package/dist/executors/shared.d.ts +0 -226
  105. /package/dist/executors/{script-command.d.ts → script/command.d.ts} +0 -0
  106. /package/dist/executors/{script-command.js → script/command.js} +0 -0
@@ -13,7 +13,7 @@ import { buildVariantConfig, resolveExecutionStrategy } from './execution-strate
13
13
  import { getJudgePromptHash } from '../grading/judge.js';
14
14
  import { getDiagnosticPromptHash, resolveDiagnosticTarget, } from '../grading/diagnostic.js';
15
15
  import { bootstrapMeanCI, bootstrapPairedDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES, } from './bootstrap.js';
16
- import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
16
+ import { resolveExecutorRuntimeFingerprint, } from '../executors/core/runtime-fingerprint.js';
17
17
  import { ownRecordValue, setOwnRecordValue, } from '../shared/record-count.js';
18
18
  import { writeJsonFileAtomic } from '../shared/atomic-json.js';
19
19
  import { hashSample } from './sample-fingerprint.js';
@@ -39,15 +39,25 @@ export { hashSample } from './sample-fingerprint.js';
39
39
  export function getCliVersion() {
40
40
  return PKG.version;
41
41
  }
42
- export function getGitInfo() {
43
- // stdio 静默 stderr:在非 git 目录(如 omk init 出来的 demo)里 rev-parse 会打印
44
- // `fatal: not a git repository` 到终端。catch 已把失败兜成 null(报告省略 git 信息),
45
- // 这条 fatal 对用户是纯噪声,吞掉它。与 skill-loader 的 GIT_PROBE_STDIO 同口径。
42
+ export function getGitInfo(cwd = process.cwd()) {
43
+ // porcelain v2 的 branch header 一次返回 commit / branch,后续状态行同时表达 dirty。
44
+ // 相比 rev-parse ×2 + status,少启动两个同步 Git 子进程;报告字段语义不变。
45
+ // stdio 静默 stderr:非 git 目录仍返回 null,不把 fatal 噪声泄漏给用户。
46
46
  const gitProbeStdio = ['ignore', 'pipe', 'ignore'];
47
47
  try {
48
- const commit = execFileSync('git', ['rev-parse', 'HEAD'], { encoding: 'utf-8', stdio: gitProbeStdio }).trim();
49
- const branch = execFileSync('git', ['rev-parse', '--abbrev-ref', 'HEAD'], { encoding: 'utf-8', stdio: gitProbeStdio }).trim();
50
- const dirty = execFileSync('git', ['status', '--porcelain'], { encoding: 'utf-8', stdio: gitProbeStdio }).trim().length > 0;
48
+ const output = execFileSync('git', ['status', '--porcelain=v2', '--branch'], {
49
+ cwd,
50
+ encoding: 'utf-8',
51
+ stdio: gitProbeStdio,
52
+ });
53
+ const lines = output.split('\n');
54
+ const oid = lines.find((line) => line.startsWith('# branch.oid '))?.slice('# branch.oid '.length).trim();
55
+ const head = lines.find((line) => line.startsWith('# branch.head '))?.slice('# branch.head '.length).trim();
56
+ if (!oid || oid === '(initial)' || !head)
57
+ return null;
58
+ const commit = oid;
59
+ const branch = head === '(detached)' ? 'HEAD' : head;
60
+ const dirty = lines.some((line) => line.length > 0 && !line.startsWith('# '));
51
61
  return { commit, commitShort: commit.slice(0, 7), branch, dirty };
52
62
  }
53
63
  catch {
@@ -64,15 +74,15 @@ function commonRuntime(runtimes) {
64
74
  function representativeRuntime(runtimes) {
65
75
  return Object.values(runtimes)[0];
66
76
  }
67
- export function buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request, }) {
77
+ export function buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request, executor, }) {
68
78
  const runtimes = {};
69
79
  for (const task of tasks) {
70
80
  if (ownRecordValue(runtimes, task.variant))
71
81
  continue;
72
82
  const executionPlan = resolveExecutionStrategy(task, model, request?.timeoutMs, false);
73
- setOwnRecordValue(runtimes, task.variant, getExecutorRuntimeFingerprint(executorName, model, {
83
+ setOwnRecordValue(runtimes, task.variant, resolveExecutorRuntimeFingerprint(executorName, model, {
74
84
  skillDir: executionPlan.input.skillDir,
75
- }));
85
+ }, executor));
76
86
  }
77
87
  for (const variant of variants) {
78
88
  if (ownRecordValue(runtimes, variant))
@@ -86,13 +96,13 @@ export function buildExecutorRuntimesByVariant({ variants, model, executorName,
86
96
  : artifact?.locator
87
97
  ? dirname(artifact.locator)
88
98
  : request?.skillDir);
89
- setOwnRecordValue(runtimes, variant, getExecutorRuntimeFingerprint(executorName, model, {
99
+ setOwnRecordValue(runtimes, variant, resolveExecutorRuntimeFingerprint(executorName, model, {
90
100
  skillDir: fallbackSkillDir,
91
- }));
101
+ }, executor));
92
102
  }
93
103
  return runtimes;
94
104
  }
95
- export function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, samplesBaseDir, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, }) {
105
+ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, samplesBaseDir, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, executor, judgeExecutors, }) {
96
106
  const summary = {};
97
107
  for (const variant of variants) {
98
108
  const entries = Object.values(results)
@@ -172,10 +182,18 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
172
182
  const sampleHashes = Object.fromEntries(samples.map((sample) => [sample.sample_id, hashSample(sample, samplesBaseDir)]));
173
183
  const judgeRepeat = request?.judgeRepeat && request.judgeRepeat > 1 ? request.judgeRepeat : undefined;
174
184
  const runtimeOptions = { skillDir: request?.skillDir };
175
- const executorRuntimes = buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request });
185
+ const executorRuntimes = buildExecutorRuntimesByVariant({
186
+ variants,
187
+ model,
188
+ executorName,
189
+ tasks,
190
+ artifacts,
191
+ request,
192
+ executor,
193
+ });
176
194
  const executorRuntime = commonRuntime(executorRuntimes)
177
195
  ?? representativeRuntime(executorRuntimes)
178
- ?? getExecutorRuntimeFingerprint(executorName, model, runtimeOptions);
196
+ ?? resolveExecutorRuntimeFingerprint(executorName, model, runtimeOptions, executor);
179
197
  // request.judgeModels is the authoritative source (always non-empty in new schema).
180
198
  // Fallback synthesizes a 1-entry from positional judgeModel/executorName for any
181
199
  // legacy caller not yet migrated to the array. noJudge ⇒ runtime undefined per entry.
@@ -183,7 +201,9 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
183
201
  const judgeModelsMeta = requestJudges.map((jc) => ({
184
202
  executor: jc.executor,
185
203
  model: jc.model,
186
- ...(noJudge ? {} : { runtime: getExecutorRuntimeFingerprint(jc.executor, jc.model, runtimeOptions) }),
204
+ ...(noJudge ? {} : {
205
+ runtime: resolveExecutorRuntimeFingerprint(jc.executor, jc.model, runtimeOptions, judgeExecutors?.[jc.executor]),
206
+ }),
187
207
  }));
188
208
  const diagnosticEnabled = request?.noDiagnostic !== true;
189
209
  const diagnosticTarget = resolveDiagnosticTarget(requestJudges, executorName, model);
@@ -192,7 +212,8 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
192
212
  enabled: true,
193
213
  executor: diagnosticTarget.executor,
194
214
  model: diagnosticTarget.model,
195
- runtime: getExecutorRuntimeFingerprint(diagnosticTarget.executor, diagnosticTarget.model),
215
+ runtime: resolveExecutorRuntimeFingerprint(diagnosticTarget.executor, diagnosticTarget.model, {}, judgeExecutors?.[diagnosticTarget.executor]
216
+ ?? (diagnosticTarget.executor === executorName ? executor : undefined)),
196
217
  promptHash: getDiagnosticPromptHash(),
197
218
  }
198
219
  : { enabled: false };
@@ -1,5 +1,5 @@
1
1
  import type { Report } from '../types/index.js';
2
- import { type ExecutorVendor } from '../executors/shared.js';
2
+ import { type ExecutorVendor } from '../executors/core/registry.js';
3
3
  /**
4
4
  * 评委独立性分析(单一来源,verdict caveat 与 analysis 诊断共用)。
5
5
  *
@@ -1,4 +1,4 @@
1
- import { executorVendor } from '../executors/shared.js';
1
+ import { executorVendor } from '../executors/core/registry.js';
2
2
  export function analyzeJudgeIndependence(report) {
3
3
  const judges = report.meta?.judgeModels ?? [];
4
4
  const judgeVendors = judges.map((j) => executorVendor(j.executor));
@@ -71,6 +71,12 @@ function isRuntimeFingerprint(value) {
71
71
  || !['reported', 'not-reported', 'unknown'].includes(String(value.capabilities.costUSD))
72
72
  || !['native', 'best-effort', 'none', 'unknown'].includes(String(value.capabilities.trace))
73
73
  || !['full', 'full-no-partial', 'cwd-only', 'none', 'unknown'].includes(String(value.capabilities.skillIsolation))
74
+ || (value.auditability !== undefined
75
+ && (!isRecord(value.auditability)
76
+ || !['complete', 'partial'].includes(String(value.auditability.status))
77
+ || (value.auditability.reasons !== undefined
78
+ && (!Array.isArray(value.auditability.reasons)
79
+ || !value.auditability.reasons.every((reason) => typeof reason === 'string')))))
74
80
  || (value.sdk !== undefined && !isRuntimePackage(value.sdk)))
75
81
  return false;
76
82
  if (value.binary === undefined)
@@ -17,6 +17,7 @@ export interface ResumeCompatibilityInput {
17
17
  samplesBaseDir?: string;
18
18
  tasks: Task[];
19
19
  artifacts: Artifact[];
20
+ executorOverrides?: Readonly<Record<string, import('../types/index.js').ExecutorFn>>;
20
21
  }
21
22
  export interface ResumeCompatibilityResult {
22
23
  compatible: boolean;
@@ -3,7 +3,7 @@ import { buildExecutorRuntimesByVariant, EVALUATION_REPORT_SCHEMA_VERSION, getCl
3
3
  import { hashSample } from './sample-fingerprint.js';
4
4
  import { getJudgePromptHash } from '../grading/judge.js';
5
5
  import { getDiagnosticPromptHash, resolveDiagnosticTarget, } from '../grading/diagnostic.js';
6
- import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
6
+ import { resolveExecutorRuntimeFingerprint } from '../executors/core/runtime-fingerprint.js';
7
7
  function canonicalStringify(value) {
8
8
  if (value === undefined)
9
9
  return 'undefined';
@@ -73,6 +73,7 @@ export function checkResumeCompatibility(report, current) {
73
73
  skillDir: current.skillDir,
74
74
  timeoutMs: current.timeoutMs,
75
75
  },
76
+ executor: current.executorOverrides?.[current.executorName],
76
77
  });
77
78
  check('meta.executorRuntimes', runtimeFingerprints(report.meta.executorRuntimes), runtimeFingerprints(expectedExecutorRuntimes));
78
79
  const actualJudges = report.meta.judgeModels.map((judge) => ({
@@ -85,9 +86,9 @@ export function checkResumeCompatibility(report, current) {
85
86
  model: judge.model,
86
87
  ...(!current.noJudge
87
88
  ? {
88
- runtime: getExecutorRuntimeFingerprint(judge.executor, judge.model, {
89
+ runtime: resolveExecutorRuntimeFingerprint(judge.executor, judge.model, {
89
90
  skillDir: current.skillDir,
90
- }).fingerprint,
91
+ }, current.executorOverrides?.[judge.executor]).fingerprint,
91
92
  }
92
93
  : {}),
93
94
  }));
@@ -102,7 +103,7 @@ export function checkResumeCompatibility(report, current) {
102
103
  enabled: true,
103
104
  executor: diagnosticTarget.executor,
104
105
  model: diagnosticTarget.model,
105
- runtime: getExecutorRuntimeFingerprint(diagnosticTarget.executor, diagnosticTarget.model).fingerprint,
106
+ runtime: resolveExecutorRuntimeFingerprint(diagnosticTarget.executor, diagnosticTarget.model, {}, current.executorOverrides?.[diagnosticTarget.executor]).fingerprint,
106
107
  promptHash: getDiagnosticPromptHash(),
107
108
  }
108
109
  : { enabled: false };
@@ -1,7 +1,7 @@
1
1
  import { dirname } from 'node:path';
2
2
  import { DEFAULT_OUTPUT_DIR, EVALUATION_REPORT_SCHEMA_VERSION, generateRunId, getCliVersion, getGitInfo, persistReport, } from '../eval-core/evaluation-reporting.js';
3
3
  import { buildEvaluationRequest, createEvaluationRun, createSucceededJob, finalizeEvaluationRun } from '../eval-core/evaluation-job.js';
4
- import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
4
+ import { getExecutorRuntimeFingerprint } from '../executors/core/runtime-fingerprint.js';
5
5
  import { createFileJobStore } from '../server/job-store.js';
6
6
  import { DEFAULT_JOBS_DIR } from '../eval-core/default-dirs.js';
7
7
  import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
@@ -16,6 +16,7 @@
16
16
  import { existsSync, readdirSync } from 'node:fs';
17
17
  import { homedir } from 'node:os';
18
18
  import { join, relative } from 'node:path';
19
+ import { executorFamily } from '../../executors/core/registry.js';
19
20
  import { tEvalWorkflowMessage } from '../messages.js';
20
21
  export function buildPowerWarnings(sampleCount, repeat, lang = 'zh') {
21
22
  const warnings = [];
@@ -43,11 +44,12 @@ export function buildIsolationWarnings(artifacts, strictBaseline, options) {
43
44
  if (!hasUnisolatedBaseline)
44
45
  return [];
45
46
  const executorName = options.executorName;
47
+ const family = executorFamily(executorName);
46
48
  const home = options.homeDir ?? homedir();
47
49
  const cwd = options.cwd ?? process.cwd();
48
- const roots = executorName === 'claude' || executorName === 'claude-sdk'
50
+ const roots = family === 'claude'
49
51
  ? [join(home, '.claude', 'skills'), join(cwd, '.claude', 'skills')]
50
- : executorName === 'codex' || executorName === 'codex-sdk'
52
+ : family === 'codex'
51
53
  ? [
52
54
  join(home, '.agents', 'skills'),
53
55
  join(home, '.codex', 'skills'),
@@ -43,6 +43,8 @@ export interface EvaluationPipelineOptions {
43
43
  judgeExecutorName: string;
44
44
  executor: ExecutorFn;
45
45
  judgeExecutor: ExecutorFn;
46
+ /** Embedding-host executors that must not be reconstructed from a CLI name. */
47
+ executorOverrides?: Readonly<Record<string, ExecutorFn>>;
46
48
  outputDir?: string | null;
47
49
  project?: string;
48
50
  owner?: string;
@@ -91,7 +93,7 @@ export interface EvaluationPipelineOptions {
91
93
  noDiagnostic?: boolean;
92
94
  }
93
95
  type VariantResult = import('../types/index.js').VariantResult;
94
- export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
96
+ export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, executorOverrides, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
95
97
  report: Report;
96
98
  filePath: string | null;
97
99
  }>;
@@ -28,11 +28,11 @@ import { finalizeSuccessfulRun, initializeEvaluationRunState, persistFailedJob,
28
28
  import { finalizeEvaluationReport } from './evaluation-pipeline/report-finalize.js';
29
29
  import { emitIsolationWarnings, emitPowerWarnings } from './evaluation-pipeline/preflight-warnings.js';
30
30
  import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
31
- import { assertSamplesCompatibleWithExecutor } from '../executors/capabilities.js';
31
+ import { assertSamplesCompatibleWithExecutor } from '../executors/core/capabilities.js';
32
32
  // 兼容 re-export:测试与 run-evaluation.ts 动态 import 仍打 evaluation-pipeline.js
33
33
  export { buildPowerWarnings, buildIsolationWarnings } from './evaluation-pipeline/preflight-warnings.js';
34
34
  export { _computeTestSetHashForTest } from './evaluation-pipeline/test-set-hash.js';
35
- export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, concurrency = 1, timeoutMs, noCache = false, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, verbose = false, retry = 0, existingResults,
35
+ export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, executorOverrides, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, concurrency = 1, timeoutMs, noCache = false, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, verbose = false, retry = 0, existingResults,
36
36
  // requires 现在由 runEvaluation 上游传给 doctor 处理; eval-pipeline 不再用
37
37
  // 但保留接口字段,避免破 programmatic API (类型层面接收, 内部忽略)
38
38
  requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
@@ -87,7 +87,9 @@ requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRep
87
87
  resolvedJudgeExecutors = Object.create(null);
88
88
  for (const jc of resolvedJudgeModels) {
89
89
  if (!ownRecordValue(resolvedJudgeExecutors, jc.executor)) {
90
- setOwnRecordValue(resolvedJudgeExecutors, jc.executor, createExecutor(jc.executor));
90
+ setOwnRecordValue(resolvedJudgeExecutors, jc.executor, executorOverrides?.[jc.executor]
91
+ ?? (jc.executor === executorName ? executor : undefined)
92
+ ?? createExecutor(jc.executor));
91
93
  }
92
94
  }
93
95
  }
@@ -156,6 +158,8 @@ requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRep
156
158
  run,
157
159
  job,
158
160
  layeredStats,
161
+ executor,
162
+ judgeExecutors: resolvedJudgeExecutors,
159
163
  }),
160
164
  results,
161
165
  artifacts,
@@ -1,4 +1,4 @@
1
- import type { Artifact, BatchEvaluationReport, EvaluationReport, JobStore, ProgressCallback, Report, VarianceData, VariantSpec } from '../types/index.js';
1
+ import type { Artifact, BatchEvaluationReport, EvaluationReport, ExecutorFn, JobStore, ProgressCallback, Report, VarianceData, VariantSpec } from '../types/index.js';
2
2
  export interface SkillProgressInfo {
3
3
  phase: string;
4
4
  skill: string;
@@ -17,6 +17,8 @@ interface CommonEvaluationOptions {
17
17
  timeoutMs?: number;
18
18
  /** Core API is runtime-neutral: CLI/default resolution happens before this boundary. */
19
19
  executorName: string;
20
+ /** Same-process executors supplied by an embedding host, keyed by executor name. */
21
+ executorOverrides?: Readonly<Record<string, ExecutorFn>>;
20
22
  jobStore?: JobStore | null;
21
23
  persistJob?: boolean;
22
24
  onProgress?: ProgressCallback | null;
@@ -122,7 +124,7 @@ export interface DryRunReport extends DryRunBase {
122
124
  samplesPath: string;
123
125
  tasks: DryRunTask[];
124
126
  }
125
- export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
127
+ export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, executorOverrides, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
126
128
  report: Report | DryRunReport;
127
129
  filePath: string | null;
128
130
  }>;
@@ -10,7 +10,7 @@ import { checkResumeCompatibility } from '../eval-core/resume-compatibility.js';
10
10
  import { findSaturationPoint } from '../analysis/saturation.js';
11
11
  import { bootstrapMeanCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
12
12
  import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
13
- export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
13
+ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName, executorOverrides, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
14
14
  // Unified judgeModels → derive single-judge fields for downstream pipeline / grading
15
15
  // (which still operate on string `judgeModel` + `judgeExecutorName` fields per call).
16
16
  const effectiveJudgeModels = judgeModels && judgeModels.length > 0
@@ -123,6 +123,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
123
123
  samplesBaseDir,
124
124
  tasks,
125
125
  artifacts: resolvedArtifacts,
126
+ executorOverrides,
126
127
  });
127
128
  if (compatibility.compatible) {
128
129
  const sourceBySample = new Map(existing.results.map((entry) => [entry.sample_id, entry.variants]));
@@ -159,8 +160,10 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
159
160
  const effectiveSkipConnectivity = existingResults !== undefined
160
161
  ? true
161
162
  : skipConnectivity;
162
- const executor = createExecutor(executorName);
163
- const judgeExecutor = createExecutor(judgeExecutorName || executorName);
163
+ const executor = executorOverrides?.[executorName] ?? createExecutor(executorName);
164
+ const effectiveJudgeExecutorName = judgeExecutorName || executorName;
165
+ const judgeExecutor = executorOverrides?.[effectiveJudgeExecutorName]
166
+ ?? createExecutor(effectiveJudgeExecutorName);
164
167
  return executeEvaluationPipeline({
165
168
  samplesPath,
166
169
  samplesBaseDir,
@@ -176,6 +179,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
176
179
  judgeExecutorName: judgeExecutorName || executorName,
177
180
  executor,
178
181
  judgeExecutor,
182
+ executorOverrides,
179
183
  outputDir,
180
184
  project,
181
185
  owner,
@@ -1,2 +1,2 @@
1
- import type { ExecResult, ExecutorInput } from '../types/index.js';
1
+ import type { ExecResult, ExecutorInput } from '../../types/index.js';
2
2
  export declare function anthropicApiExecutor({ model, system, prompt, timeoutMs }: ExecutorInput): Promise<ExecResult>;
@@ -1,5 +1,7 @@
1
- import { asErrorLike, DEFAULT_TIMEOUT_MS, errorMessage, readJsonResponse, responseBodyPreview, } from './shared.js';
2
- import { optionalTokenCount } from '../shared/token-usage.js';
1
+ import { optionalTokenCount } from '../../shared/token-usage.js';
2
+ import { DEFAULT_TIMEOUT_MS } from '../core/limits.js';
3
+ import { readJsonResponse, responseBodyPreview } from '../core/http.js';
4
+ import { asErrorLike, errorMessage } from '../core/runtime.js';
3
5
  export async function anthropicApiExecutor({ model, system, prompt, timeoutMs = DEFAULT_TIMEOUT_MS }) {
4
6
  const apiKey = process.env.ANTHROPIC_API_KEY;
5
7
  if (!apiKey)
@@ -1,2 +1,2 @@
1
- import type { ExecResult, ExecutorInput } from '../types/index.js';
1
+ import type { ExecResult, ExecutorInput } from '../../../types/index.js';
2
2
  export declare function claudeCliExecutor({ model, system, prompt, cwd, skillDir, timeoutMs, allowedSkills, mocks, mocksBaseDir, mocksStrict, lean, effort }: ExecutorInput): Promise<ExecResult>;
@@ -1,6 +1,8 @@
1
- import { buildExecEnv, DEFAULT_TIMEOUT_MS, errorMessage, interruptedExecResult, MAX_BUFFER, spawnWithSigintPropagation, timeoutExecResult, } from './shared.js';
2
- import { materializeForCliConfigDir } from '../eval-core/mocks-runtime.js';
3
- import { buildClaudeResult, parseClaudeStreamJson } from './claude-protocol.js';
1
+ import { materializeForCliConfigDir } from '../../../eval-core/mocks-runtime.js';
2
+ import { buildClaudeResult, parseClaudeStreamJson } from './protocol.js';
3
+ import { DEFAULT_TIMEOUT_MS, MAX_BUFFER } from '../../core/limits.js';
4
+ import { buildExecEnv, errorMessage, interruptedExecResult, timeoutExecResult, } from '../../core/runtime.js';
5
+ import { spawnWithSigintPropagation } from '../../core/subprocess.js';
4
6
  // claude CLI 用 `--disable-slash-commands` (文档:"Disable all skills") +
5
7
  // `--disallowedTools Skill` 实现与 SDK 等价的完全隔离。非空 skill 白名单不再支持
6
8
  // (它无法真正隔离,见 claude-sdk.ts buildSdkIsolationOptions),所以 [name1, ...] 必须 throw。
@@ -0,0 +1,87 @@
1
+ import type { ExecResult } from '../../../types/index.js';
2
+ interface ClaudeTokenUsage {
3
+ input_tokens?: number;
4
+ output_tokens?: number;
5
+ cache_read_input_tokens?: number;
6
+ cache_creation_input_tokens?: number;
7
+ }
8
+ export interface ClaudeSdkQueryOptions {
9
+ model?: string;
10
+ systemPrompt?: string;
11
+ cwd: string;
12
+ permissionMode: 'bypassPermissions';
13
+ allowDangerouslySkipPermissions: true;
14
+ abortController: AbortController;
15
+ env: NodeJS.ProcessEnv;
16
+ }
17
+ export interface ClaudeSdkQueryInput {
18
+ prompt: string;
19
+ options: ClaudeSdkQueryOptions;
20
+ }
21
+ export interface ClaudeMessage {
22
+ type: string;
23
+ message?: {
24
+ role?: string;
25
+ content?: Array<{
26
+ type: string;
27
+ text?: string;
28
+ id?: string;
29
+ name?: string;
30
+ input?: unknown;
31
+ }>;
32
+ };
33
+ tool_use_id?: string;
34
+ content?: string | Array<{
35
+ type: string;
36
+ text?: string;
37
+ }>;
38
+ is_error?: boolean;
39
+ }
40
+ export interface ClaudeResultMessage extends ClaudeMessage {
41
+ type: 'result';
42
+ result?: string;
43
+ usage?: ClaudeTokenUsage;
44
+ total_cost_usd?: number;
45
+ duration_api_ms?: number;
46
+ duration_ms?: number;
47
+ num_turns?: number;
48
+ stop_reason?: string | null;
49
+ modelUsage?: Record<string, {
50
+ inputTokens?: number;
51
+ outputTokens?: number;
52
+ cacheReadInputTokens?: number;
53
+ cacheCreationInputTokens?: number;
54
+ }>;
55
+ subtype?: string;
56
+ errors?: string[];
57
+ }
58
+ export interface ClaudeSdkModule {
59
+ query: (opts: ClaudeSdkQueryInput) => AsyncIterable<ClaudeMessage>;
60
+ }
61
+ export interface ClaudeMeasurements {
62
+ durationMs: number;
63
+ durationApiMs: number;
64
+ inputTokens: number;
65
+ outputTokens: number;
66
+ cacheReadTokens: number;
67
+ cacheCreationTokens: number;
68
+ costUSD: number;
69
+ numTurns: number;
70
+ }
71
+ export declare function normalizeClaudeMeasurements(result: ClaudeResultMessage): ClaudeMeasurements | {
72
+ error: string;
73
+ };
74
+ export interface ClaudeStreamParseResult {
75
+ messages: ClaudeMessage[];
76
+ malformedLineCount: number;
77
+ }
78
+ export declare function parseClaudeStreamJson(stdout: string): ClaudeStreamParseResult;
79
+ export declare function buildClaudeResult(options: {
80
+ messages: ClaudeMessage[];
81
+ wallClockDurationMs: number;
82
+ source: 'claude stream-json' | 'claude-sdk';
83
+ malformedLineCount?: number;
84
+ forcedError?: string;
85
+ messageTimestamps?: number[];
86
+ }): ExecResult;
87
+ export {};
@@ -1,6 +1,6 @@
1
- import { checkedSumTokenCounts, nonNegativeMetric, optionalTokenCount, } from '../shared/token-usage.js';
2
- import { extractAgentTrace, isClaudeSdkResultMessage } from './claude-sdk-trace.js';
3
- export function normalizeClaudeSdkMeasurements(result) {
1
+ import { checkedSumTokenCounts, nonNegativeMetric, optionalTokenCount, } from '../../../shared/token-usage.js';
2
+ import { extractClaudeTrace, isClaudeResultMessage } from './trace.js';
3
+ export function normalizeClaudeMeasurements(result) {
4
4
  const durationMs = optionalTokenCount(result.duration_ms);
5
5
  const durationApiMs = optionalTokenCount(result.duration_api_ms);
6
6
  const numTurns = optionalTokenCount(result.num_turns);
@@ -96,8 +96,8 @@ export function parseClaudeStreamJson(stdout) {
96
96
  }
97
97
  export function buildClaudeResult(options) {
98
98
  const { messages, wallClockDurationMs, source, malformedLineCount = 0, forcedError, messageTimestamps, } = options;
99
- const resultMessages = messages.filter(isClaudeSdkResultMessage);
100
- const trace = extractAgentTrace(messages, messageTimestamps);
99
+ const resultMessages = messages.filter(isClaudeResultMessage);
100
+ const trace = extractClaudeTrace(messages, messageTimestamps);
101
101
  const traceFields = {
102
102
  fullNumTurns: trace.fullNumTurns,
103
103
  numSubAgents: trace.numSubAgents,
@@ -133,7 +133,7 @@ export function buildClaudeResult(options) {
133
133
  };
134
134
  }
135
135
  const result = resultMessages[0];
136
- const measurements = normalizeClaudeSdkMeasurements(result);
136
+ const measurements = normalizeClaudeMeasurements(result);
137
137
  if ('error' in measurements) {
138
138
  errors.push(measurements.error);
139
139
  return {
@@ -1,5 +1,5 @@
1
- import type { ExecResult, ExecutorInput } from '../types/index.js';
2
- export { normalizeClaudeSdkMeasurements } from './claude-protocol.js';
1
+ import type { ExecResult, ExecutorInput } from '../../../types/index.js';
2
+ export { normalizeClaudeMeasurements } from './protocol.js';
3
3
  /**
4
4
  * Map ExecutorInput.allowedSkills to SDK query options for skill isolation.
5
5
  * undefined → {} (SDK default: full ~/.claude/skills/ discovery)
@@ -1,11 +1,14 @@
1
1
  import { join } from 'node:path';
2
2
  import { mkdirSync, writeFileSync } from 'node:fs';
3
3
  import { tmpdir } from 'node:os';
4
- import { asErrorLike, buildExecEnv, DEFAULT_TIMEOUT_MS, errorMessage, interruptedExecResult, registerSigintSubscriber, timeoutExecResult, } from './shared.js';
5
- import { buildSdkHookCallback } from '../eval-core/mocks-runtime.js';
6
- import { buildClaudeResult } from './claude-protocol.js';
7
- export { normalizeClaudeSdkMeasurements } from './claude-protocol.js';
4
+ import { DEFAULT_TIMEOUT_MS } from '../../core/limits.js';
5
+ import { asErrorLike, buildExecEnv, errorMessage, interruptedExecResult, timeoutExecResult, } from '../../core/runtime.js';
6
+ import { registerSigintSubscriber } from '../../core/subprocess.js';
7
+ import { buildSdkHookCallback } from '../../../eval-core/mocks-runtime.js';
8
+ import { buildClaudeResult } from './protocol.js';
9
+ export { normalizeClaudeMeasurements } from './protocol.js';
8
10
  let sdkQuery = null;
11
+ const CLAUDE_AGENT_SDK_PACKAGE = '@anthropic-ai/claude-agent-sdk';
9
12
  /**
10
13
  * Map ExecutorInput.allowedSkills to SDK query options for skill isolation.
11
14
  * undefined → {} (SDK default: full ~/.claude/skills/ discovery)
@@ -28,7 +31,7 @@ export function buildSdkIsolationOptions(allowedSkills) {
28
31
  }
29
32
  async function getSdkQuery() {
30
33
  if (!sdkQuery) {
31
- const sdk = await import('@anthropic-ai/claude-agent-sdk');
34
+ const sdk = await import(CLAUDE_AGENT_SDK_PACKAGE);
32
35
  sdkQuery = sdk.query;
33
36
  }
34
37
  return sdkQuery;
@@ -0,0 +1,9 @@
1
+ import type { ToolCallInfo, TurnInfo } from '../../../types/index.js';
2
+ import type { ClaudeMessage } from './protocol.js';
3
+ export declare function isClaudeResultMessage(message: ClaudeMessage): boolean;
4
+ export declare function extractClaudeTrace(messages: ClaudeMessage[], timestamps?: number[]): {
5
+ turns: TurnInfo[];
6
+ toolCalls: ToolCallInfo[];
7
+ fullNumTurns: number;
8
+ numSubAgents: number;
9
+ };
@@ -1,10 +1,10 @@
1
- import { safeSliceForJson } from '../util/safe-slice.js';
2
- import { isToolResultFailureText } from '../observability/text-signals.js';
3
- import { normalizeToolIdentity } from '../shared/tool-identity.js';
4
- export function isClaudeSdkResultMessage(message) {
1
+ import { safeSliceForJson } from '../../../util/safe-slice.js';
2
+ import { isToolResultFailureText } from '../../../observability/text-signals.js';
3
+ import { normalizeToolIdentity } from '../../../shared/tool-identity.js';
4
+ export function isClaudeResultMessage(message) {
5
5
  return message.type === 'result';
6
6
  }
7
- export function extractAgentTrace(messages, timestamps) {
7
+ export function extractClaudeTrace(messages, timestamps) {
8
8
  const turns = [];
9
9
  const toolCalls = [];
10
10
  const pendingToolUse = new Map();
@@ -1,8 +1,6 @@
1
- import type { ExecutorFn, ExecutorInput, Sample } from '../types/index.js';
2
- export type SampleMockSupport = 'native-hooks' | 'delegated-script' | 'unsupported';
3
- export interface ExecutorCapabilities {
4
- sampleMocks: SampleMockSupport;
5
- }
1
+ import type { ExecutorFn, ExecutorInput, Sample } from '../../types/index.js';
2
+ import { type ExecutorCapabilities } from './registry.js';
3
+ export type { ExecutorCapabilities, SampleMockSupport } from './registry.js';
6
4
  /**
7
5
  * Custom script executors receive the OMK_MOCK_* protocol environment and own
8
6
  * the final adapter. Built-ins are explicit so unsupported runtimes can never
@@ -1,20 +1,13 @@
1
- const BUILTIN_CAPABILITIES = {
2
- claude: { sampleMocks: 'native-hooks' },
3
- 'claude-sdk': { sampleMocks: 'native-hooks' },
4
- codex: { sampleMocks: 'unsupported' },
5
- 'codex-sdk': { sampleMocks: 'unsupported' },
6
- gemini: { sampleMocks: 'unsupported' },
7
- 'anthropic-api': { sampleMocks: 'unsupported' },
8
- 'openai-api': { sampleMocks: 'unsupported' },
9
- };
1
+ import { getExecutorDescriptor, } from './registry.js';
10
2
  /**
11
3
  * Custom script executors receive the OMK_MOCK_* protocol environment and own
12
4
  * the final adapter. Built-ins are explicit so unsupported runtimes can never
13
5
  * silently turn mock assertions into model failures.
14
6
  */
15
7
  export function getExecutorCapabilities(executorName) {
16
- return BUILTIN_CAPABILITIES[executorName]
17
- ?? { sampleMocks: 'delegated-script' };
8
+ return {
9
+ sampleMocks: getExecutorDescriptor(executorName)?.sampleMocks ?? 'delegated-script',
10
+ };
18
11
  }
19
12
  export function executorSupportsSampleMocks(executorName) {
20
13
  return getExecutorCapabilities(executorName).sampleMocks !== 'unsupported';
@@ -0,0 +1,6 @@
1
+ export interface JsonResponseBody<T> {
2
+ data: T | null;
3
+ rawBody: string;
4
+ }
5
+ export declare function readJsonResponse<T>(response: Response): Promise<JsonResponseBody<T>>;
6
+ export declare function responseBodyPreview(rawBody: string, maxLength?: number): string;