oh-my-knowledge 0.52.3 → 0.53.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -1
- package/README.zh.md +13 -1
- package/dist/assets/agent-skills/omk/SKILL.md +4 -0
- package/dist/assets/agent-skills/omk/references/commands.md +1 -1
- package/dist/authoring/evolver.js +1 -1
- package/dist/authoring/generator.js +1 -1
- package/dist/cli/commands/eval/index.js +7 -6
- package/dist/cli/commands/install.js +5 -1
- package/dist/cli/lib/generation-failure-hint.js +6 -8
- package/dist/cli/lib/runtime-defaults.d.ts +2 -0
- package/dist/cli/lib/runtime-defaults.js +7 -4
- package/dist/dsh-plugin/cordis.patch.yml +3 -0
- package/dist/dsh-plugin/host-executor.d.ts +93 -0
- package/dist/dsh-plugin/host-executor.js +232 -0
- package/dist/dsh-plugin/index.d.ts +28 -0
- package/dist/dsh-plugin/index.js +156 -0
- package/dist/dsh-plugin/protocol.d.ts +21 -0
- package/dist/dsh-plugin/protocol.js +229 -0
- package/dist/eval-core/comparability.js +3 -0
- package/dist/eval-core/evaluation-execution.js +5 -13
- package/dist/eval-core/evaluation-reporting.d.ts +7 -4
- package/dist/eval-core/evaluation-reporting.js +39 -18
- package/dist/eval-core/judge-independence.d.ts +1 -1
- package/dist/eval-core/judge-independence.js +1 -1
- package/dist/eval-core/report-document.js +6 -0
- package/dist/eval-core/resume-compatibility.d.ts +1 -0
- package/dist/eval-core/resume-compatibility.js +5 -4
- package/dist/eval-workflows/batch-evaluation-workflow.js +1 -1
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +4 -2
- package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
- package/dist/eval-workflows/evaluation-pipeline.js +7 -3
- package/dist/eval-workflows/run-evaluation.d.ts +4 -2
- package/dist/eval-workflows/run-evaluation.js +7 -3
- package/dist/executors/{anthropic-api.d.ts → anthropic/api.d.ts} +1 -1
- package/dist/executors/{anthropic-api.js → anthropic/api.js} +4 -2
- package/dist/executors/{claude-cli.d.ts → anthropic/claude/cli.d.ts} +1 -1
- package/dist/executors/{claude-cli.js → anthropic/claude/cli.js} +5 -3
- package/dist/executors/anthropic/claude/protocol.d.ts +87 -0
- package/dist/executors/{claude-protocol.js → anthropic/claude/protocol.js} +6 -6
- package/dist/executors/{claude-sdk.d.ts → anthropic/claude/sdk.d.ts} +2 -2
- package/dist/executors/{claude-sdk.js → anthropic/claude/sdk.js} +8 -5
- package/dist/executors/anthropic/claude/trace.d.ts +9 -0
- package/dist/executors/{claude-sdk-trace.js → anthropic/claude/trace.js} +5 -5
- package/dist/executors/{capabilities.d.ts → core/capabilities.d.ts} +3 -5
- package/dist/executors/{capabilities.js → core/capabilities.js} +4 -11
- package/dist/executors/core/http.d.ts +6 -0
- package/dist/executors/core/http.js +19 -0
- package/dist/executors/core/limits.d.ts +2 -0
- package/dist/executors/core/limits.js +2 -0
- package/dist/executors/core/optional-dependencies.d.ts +7 -0
- package/dist/executors/core/optional-dependencies.js +35 -0
- package/dist/executors/core/registry.d.ts +145 -0
- package/dist/executors/core/registry.js +127 -0
- package/dist/executors/core/runtime-fingerprint.d.ts +13 -0
- package/dist/executors/{runtime-fingerprint.js → core/runtime-fingerprint.js} +119 -59
- package/dist/executors/core/runtime.d.ts +12 -0
- package/dist/executors/core/runtime.js +61 -0
- package/dist/executors/core/subprocess.d.ts +44 -0
- package/dist/executors/{shared.js → core/subprocess.js} +20 -156
- package/dist/executors/index.d.ts +4 -4
- package/dist/executors/index.js +26 -15
- package/dist/executors/{openai-api.d.ts → openai/api.d.ts} +1 -1
- package/dist/executors/{openai-api.js → openai/api.js} +4 -2
- package/dist/executors/{codex-cli.d.ts → openai/codex/cli.d.ts} +3 -3
- package/dist/executors/{codex-cli.js → openai/codex/cli.js} +5 -3
- package/dist/executors/openai/codex/protocol.d.ts +72 -0
- package/dist/executors/{codex-protocol.js → openai/codex/protocol.js} +33 -2
- package/dist/executors/{codex-sdk.d.ts → openai/codex/sdk.d.ts} +18 -4
- package/dist/executors/{codex-sdk.js → openai/codex/sdk.js} +7 -4
- package/dist/executors/{codex-cli-trace.d.ts → openai/codex/trace.d.ts} +2 -2
- package/dist/executors/{codex-cli-trace.js → openai/codex/trace.js} +4 -4
- package/dist/executors/{script.d.ts → script/index.d.ts} +1 -1
- package/dist/executors/{script.js → script/index.js} +6 -4
- package/dist/grading/judge.d.ts +1 -1
- package/dist/grading/judge.js +1 -1
- package/dist/renderer/html-renderer.js +30 -11
- package/dist/types/executor.d.ts +13 -2
- package/dist/types/judge.d.ts +2 -2
- package/package.json +21 -5
- package/dist/executors/claude-protocol.d.ts +0 -28
- package/dist/executors/claude-sdk-trace.d.ts +0 -9
- package/dist/executors/codex-protocol.d.ts +0 -24
- package/dist/executors/gemini.d.ts +0 -2
- package/dist/executors/gemini.js +0 -156
- package/dist/executors/runtime-fingerprint.d.ts +0 -6
- package/dist/executors/shared.d.ts +0 -226
- /package/dist/executors/{script-command.d.ts → script/command.d.ts} +0 -0
- /package/dist/executors/{script-command.js → script/command.js} +0 -0
|
@@ -57,7 +57,20 @@ function renderDebiasModeTag(modes, lang) {
|
|
|
57
57
|
function pluralizeEn(count, singular, plural = `${singular}s`) {
|
|
58
58
|
return `${count} ${count === 1 ? singular : plural}`;
|
|
59
59
|
}
|
|
60
|
-
function
|
|
60
|
+
function executorDisplayName(executor, lang) {
|
|
61
|
+
if (executor === 'dsh-host') {
|
|
62
|
+
return lang === 'zh' ? 'DeepSeek Harness(宿主模式)' : 'DeepSeek Harness (host mode)';
|
|
63
|
+
}
|
|
64
|
+
return executor || 'unknown';
|
|
65
|
+
}
|
|
66
|
+
function executorModelDisplayName(executor, model, lang) {
|
|
67
|
+
const executorName = executorDisplayName(executor, lang);
|
|
68
|
+
const modelName = model || 'unknown';
|
|
69
|
+
if (executor === 'dsh-host')
|
|
70
|
+
return `${executorName} · ${modelName}`;
|
|
71
|
+
return `${executorName}:${modelName}`;
|
|
72
|
+
}
|
|
73
|
+
function gatherRuntimeScopes(meta, lang) {
|
|
61
74
|
const scopes = [];
|
|
62
75
|
if (meta.executorRuntimes && Object.keys(meta.executorRuntimes).length > 0) {
|
|
63
76
|
Object.entries(meta.executorRuntimes)
|
|
@@ -75,14 +88,19 @@ function gatherRuntimeScopes(meta) {
|
|
|
75
88
|
.slice()
|
|
76
89
|
.sort((a, b) => `${a.executor}:${a.model}`.localeCompare(`${b.executor}:${b.model}`))
|
|
77
90
|
.forEach((entry) => {
|
|
78
|
-
if (entry.runtime)
|
|
79
|
-
scopes.push({
|
|
91
|
+
if (entry.runtime) {
|
|
92
|
+
scopes.push({
|
|
93
|
+
role: 'judge',
|
|
94
|
+
scope: executorModelDisplayName(entry.executor, entry.model, lang),
|
|
95
|
+
runtime: entry.runtime,
|
|
96
|
+
});
|
|
97
|
+
}
|
|
80
98
|
});
|
|
81
99
|
}
|
|
82
100
|
if (meta.diagnostic?.enabled && meta.diagnostic.runtime) {
|
|
83
101
|
scopes.push({
|
|
84
102
|
role: 'diagnostic',
|
|
85
|
-
scope:
|
|
103
|
+
scope: executorModelDisplayName(meta.diagnostic.executor, meta.diagnostic.model, lang),
|
|
86
104
|
runtime: meta.diagnostic.runtime,
|
|
87
105
|
});
|
|
88
106
|
}
|
|
@@ -105,6 +123,7 @@ function runtimeTooltip(runtime) {
|
|
|
105
123
|
`cost=${runtime.capabilities.costUSD}`,
|
|
106
124
|
`trace=${runtime.capabilities.trace}`,
|
|
107
125
|
`skillIsolation=${runtime.capabilities.skillIsolation}`,
|
|
126
|
+
...(runtime.auditability ? [`auditability=${runtime.auditability.status}`] : []),
|
|
108
127
|
...(runtime.binary?.contentHash
|
|
109
128
|
? [`contentHash=${runtime.binary.contentHash}`]
|
|
110
129
|
: []),
|
|
@@ -115,7 +134,7 @@ function runtimeTooltip(runtime) {
|
|
|
115
134
|
// fingerprint 重复 3 遍,扫读成本高。新版按 (fingerprint, versionText)
|
|
116
135
|
// 分组,scope 合并到 tag 内 "适用于 ..." 后缀。
|
|
117
136
|
function renderRuntimeFingerprintTags(meta, lang) {
|
|
118
|
-
const scopes = gatherRuntimeScopes(meta);
|
|
137
|
+
const scopes = gatherRuntimeScopes(meta, lang);
|
|
119
138
|
if (scopes.length === 0)
|
|
120
139
|
return '';
|
|
121
140
|
const groups = new Map();
|
|
@@ -476,11 +495,11 @@ export function renderRunDetail(report, lang = DEFAULT_LANG, skillContext) {
|
|
|
476
495
|
if (list.length === 0)
|
|
477
496
|
return `<span class="meta-tag">${t('judge', lang)}: —</span>`;
|
|
478
497
|
if (list.length === 1)
|
|
479
|
-
return `<span class="meta-tag">${t('judge', lang)}: ${e(
|
|
480
|
-
return `<span class="meta-tag" title="${t('ensembleDesc', lang)}">${t('judgeModelsLabel', lang)}: ${list.map((j) => e(
|
|
498
|
+
return `<span class="meta-tag">${t('judge', lang)}: ${e(executorModelDisplayName(list[0].executor, list[0].model, lang))}</span>`;
|
|
499
|
+
return `<span class="meta-tag" title="${t('ensembleDesc', lang)}">${t('judgeModelsLabel', lang)}: ${list.map((j) => e(executorModelDisplayName(j.executor, j.model, lang))).join(' · ')}</span>`;
|
|
481
500
|
})()}
|
|
482
501
|
${m.judgeRepeat && m.judgeRepeat > 1 ? `<span class="meta-tag" title="${t('judgeStddevDesc', lang)}">${t('judgeRepeatLabel', lang)}: ${m.judgeRepeat}</span>` : ''}
|
|
483
|
-
<span class="meta-tag">${t('executor', lang)}: ${e(m.executor
|
|
502
|
+
<span class="meta-tag">${t('executor', lang)}: ${e(executorDisplayName(m.executor, lang))}</span>
|
|
484
503
|
${m.effort ? `<span class="meta-tag" title="${e(lang === 'zh' ? 'executor LLM 的扩展思考预算(--effort)。跨 effort 报告不可严格比较' : 'reasoning effort for executor LLM (--effort); reports across different efforts are not strictly comparable')}">effort: ${e(m.effort)}</span>` : ''}
|
|
485
504
|
<span class="meta-tag"${execCostReported ? '' : ` title="${e(lang === 'zh' ? 'executor 不报 USD 成本(如 codex CLI),无法估算' : 'executor does not report USD cost (e.g. codex CLI); not measurable')}"`}>${t('cost', lang)}: ${fmtCost(totalExecCost, execCostReported)}</span>
|
|
486
505
|
<span class="meta-tag"${totalCostReported ? '' : ` title="${e(costCompletenessTooltip(lang))}"`}>${totalCostLabel}: ${fmtKnownCost(m.totalCostUSD, totalCostReported)}</span>${processCostTag ? `
|
|
@@ -639,10 +658,10 @@ export function renderBatchEvaluationDetail(report, lang = DEFAULT_LANG) {
|
|
|
639
658
|
if (list.length === 0)
|
|
640
659
|
return `<span class="meta-tag">${t('judge', lang)}: —</span>`;
|
|
641
660
|
if (list.length === 1)
|
|
642
|
-
return `<span class="meta-tag">${t('judge', lang)}: ${e(
|
|
643
|
-
return `<span class="meta-tag" title="${t('ensembleDesc', lang)}">${t('judgeModelsLabel', lang)}: ${list.map((j) => e(
|
|
661
|
+
return `<span class="meta-tag">${t('judge', lang)}: ${e(executorModelDisplayName(list[0].executor, list[0].model, lang))}</span>`;
|
|
662
|
+
return `<span class="meta-tag" title="${t('ensembleDesc', lang)}">${t('judgeModelsLabel', lang)}: ${list.map((j) => e(executorModelDisplayName(j.executor, j.model, lang))).join(' · ')}</span>`;
|
|
644
663
|
})()}
|
|
645
|
-
<span class="meta-tag">${t('executor', lang)}: ${e(m.executor
|
|
664
|
+
<span class="meta-tag">${t('executor', lang)}: ${e(executorDisplayName(m.executor, lang))}</span>
|
|
646
665
|
<span class="meta-tag"${totalCostReported ? '' : ` title="${e(costCompletenessTooltip(lang))}"`}>${t('totalCost', lang)}: ${fmtKnownCost(m.totalCostUSD, totalCostReported)}</span>
|
|
647
666
|
</div>
|
|
648
667
|
${(() => {
|
package/dist/types/executor.d.ts
CHANGED
|
@@ -88,7 +88,7 @@ export interface ExecutorInput {
|
|
|
88
88
|
* - claude-cli:物化为临时 settings.json + on-disk hook 脚本,跑完清理
|
|
89
89
|
* - script(自定义脚本):同样物化临时 settings,通过 env(OMK_MOCK_SETTINGS_FILE /
|
|
90
90
|
* OMK_MOCK_MCP_CONFIG_FILE / OMK_MOCKS_FILE)暴露给脚本;脚本负责消费该协议
|
|
91
|
-
* - codex / codex-sdk /
|
|
91
|
+
* - codex / codex-sdk / *-api:不支持,executor capability gate 会拒绝,
|
|
92
92
|
* 绝不静默忽略后把 mock_hit 记成模型失败 */
|
|
93
93
|
mocks?: import('./eval.js').Mock[];
|
|
94
94
|
/** 解析 mock.return_file 的相对路径锚点(默认 sample 文件所在目录)。 */
|
|
@@ -119,7 +119,13 @@ export interface ExecutorInput {
|
|
|
119
119
|
*/
|
|
120
120
|
effort?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
|
|
121
121
|
}
|
|
122
|
-
export type
|
|
122
|
+
export type ExecutorRuntimeFingerprintResolver = (model: string, options?: {
|
|
123
|
+
skillDir?: string | null;
|
|
124
|
+
}) => ExecutorRuntimeFingerprint;
|
|
125
|
+
export type ExecutorFn = ((input: ExecutorInput) => Promise<ExecResult>) & {
|
|
126
|
+
/** Same-process hosts can report the runtime that actually owns execution. */
|
|
127
|
+
readonly runtimeFingerprint?: ExecutorRuntimeFingerprintResolver;
|
|
128
|
+
};
|
|
123
129
|
export interface ExecutorCache {
|
|
124
130
|
get(key: string): ExecResult | null;
|
|
125
131
|
set(key: string, value: ExecResult): void;
|
|
@@ -176,5 +182,10 @@ export interface ExecutorRuntimeFingerprint {
|
|
|
176
182
|
fingerprint: string;
|
|
177
183
|
binary?: ExecutorRuntimeBinary;
|
|
178
184
|
sdk?: ExecutorRuntimePackage;
|
|
185
|
+
/** Whether the recorded fields cover the complete effective runtime composition. */
|
|
186
|
+
auditability?: {
|
|
187
|
+
status: 'complete' | 'partial';
|
|
188
|
+
reasons?: string[];
|
|
189
|
+
};
|
|
179
190
|
capabilities: ExecutorRuntimeCapabilities;
|
|
180
191
|
}
|
package/dist/types/judge.d.ts
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import type { ExecutorRuntimeFingerprint } from './executor.js';
|
|
2
2
|
/** Single judge configuration: which executor to call and which model alias to pass. */
|
|
3
3
|
export interface JudgeConfig {
|
|
4
|
-
/** Executor name (claude /
|
|
4
|
+
/** Executor name (claude / codex / anthropic-api / openai-api / shell command). */
|
|
5
5
|
executor: string;
|
|
6
|
-
/** Model alias passed to the executor (e.g. "opus", "haiku", "gpt-4o"
|
|
6
|
+
/** Model alias passed to the executor (e.g. "opus", "haiku", "gpt-4o"). */
|
|
7
7
|
model: string;
|
|
8
8
|
}
|
|
9
9
|
/** Persisted judge entry on Report.meta.judgeModels: judge config + runtime fingerprint of
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "oh-my-knowledge",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.53.0",
|
|
4
4
|
"packageManager": "yarn@4.16.0",
|
|
5
5
|
"description": "OMK — Observe. Measure. Know. Evidence-backed knowledge changes for AI applications.",
|
|
6
6
|
"type": "module",
|
|
@@ -13,6 +13,11 @@
|
|
|
13
13
|
"files": [
|
|
14
14
|
"dist/"
|
|
15
15
|
],
|
|
16
|
+
"dsh": {
|
|
17
|
+
"bundle": {
|
|
18
|
+
"patch": "./dist/dsh-plugin/cordis.patch.yml"
|
|
19
|
+
}
|
|
20
|
+
},
|
|
16
21
|
"scripts": {
|
|
17
22
|
"clean": "rm -rf dist dist-scripts node_modules/.cache/omk",
|
|
18
23
|
"build": "run-s build:src build:scripts build:assets",
|
|
@@ -28,6 +33,7 @@
|
|
|
28
33
|
"lint": "eslint 'src/**/*.ts' 'test/**/*.ts' --cache --cache-location node_modules/.cache/eslint/ --max-warnings 0",
|
|
29
34
|
"lint-staged": "lint-staged",
|
|
30
35
|
"test": "vitest run",
|
|
36
|
+
"test:profile": "yarn build && node dist-scripts/test-profile.js",
|
|
31
37
|
"ci": "run-s lint typecheck build build:docs:check test",
|
|
32
38
|
"prepublishOnly": "run-s clean build",
|
|
33
39
|
"prepare": "husky"
|
|
@@ -98,12 +104,10 @@
|
|
|
98
104
|
"author": "lizhiyao",
|
|
99
105
|
"license": "MIT",
|
|
100
106
|
"dependencies": {
|
|
101
|
-
"@anthropic-ai/
|
|
102
|
-
"@anthropic-ai/sdk": "^0.117.1",
|
|
107
|
+
"@anthropic-ai/sdk": "^0.120.0",
|
|
103
108
|
"@inquirer/prompts": "^8.4.3",
|
|
104
109
|
"@modelcontextprotocol/sdk": "^1.30.0",
|
|
105
110
|
"@oclif/core": "^4",
|
|
106
|
-
"@openai/codex-sdk": "0.147.0",
|
|
107
111
|
"ajv": "^8.18.0",
|
|
108
112
|
"chart.js": "^4.5.1",
|
|
109
113
|
"es-module-lexer": "^2.0.0",
|
|
@@ -111,6 +115,18 @@
|
|
|
111
115
|
"simple-statistics": "^7.8.9",
|
|
112
116
|
"zod": "^4.4.3"
|
|
113
117
|
},
|
|
118
|
+
"peerDependencies": {
|
|
119
|
+
"@anthropic-ai/claude-agent-sdk": "^0.3.143",
|
|
120
|
+
"@openai/codex-sdk": "^0.149.0"
|
|
121
|
+
},
|
|
122
|
+
"peerDependenciesMeta": {
|
|
123
|
+
"@anthropic-ai/claude-agent-sdk": {
|
|
124
|
+
"optional": true
|
|
125
|
+
},
|
|
126
|
+
"@openai/codex-sdk": {
|
|
127
|
+
"optional": true
|
|
128
|
+
}
|
|
129
|
+
},
|
|
114
130
|
"publishConfig": {
|
|
115
131
|
"registry": "https://registry.npmjs.org"
|
|
116
132
|
},
|
|
@@ -124,6 +140,6 @@
|
|
|
124
140
|
"typescript": "^6.0.2",
|
|
125
141
|
"typescript-eslint": "^8.58.0",
|
|
126
142
|
"vitepress": "^1.6.4",
|
|
127
|
-
"vitest": "4.1.
|
|
143
|
+
"vitest": "4.1.11"
|
|
128
144
|
}
|
|
129
145
|
}
|
|
@@ -1,28 +0,0 @@
|
|
|
1
|
-
import type { ExecResult } from '../types/index.js';
|
|
2
|
-
import type { ClaudeSdkBaseMessage, ClaudeSdkResultMessage } from './shared.js';
|
|
3
|
-
export interface ClaudeSdkMeasurements {
|
|
4
|
-
durationMs: number;
|
|
5
|
-
durationApiMs: number;
|
|
6
|
-
inputTokens: number;
|
|
7
|
-
outputTokens: number;
|
|
8
|
-
cacheReadTokens: number;
|
|
9
|
-
cacheCreationTokens: number;
|
|
10
|
-
costUSD: number;
|
|
11
|
-
numTurns: number;
|
|
12
|
-
}
|
|
13
|
-
export declare function normalizeClaudeSdkMeasurements(result: ClaudeSdkResultMessage): ClaudeSdkMeasurements | {
|
|
14
|
-
error: string;
|
|
15
|
-
};
|
|
16
|
-
export interface ClaudeStreamParseResult {
|
|
17
|
-
messages: ClaudeSdkBaseMessage[];
|
|
18
|
-
malformedLineCount: number;
|
|
19
|
-
}
|
|
20
|
-
export declare function parseClaudeStreamJson(stdout: string): ClaudeStreamParseResult;
|
|
21
|
-
export declare function buildClaudeResult(options: {
|
|
22
|
-
messages: ClaudeSdkBaseMessage[];
|
|
23
|
-
wallClockDurationMs: number;
|
|
24
|
-
source: 'claude stream-json' | 'claude-sdk';
|
|
25
|
-
malformedLineCount?: number;
|
|
26
|
-
forcedError?: string;
|
|
27
|
-
messageTimestamps?: number[];
|
|
28
|
-
}): ExecResult;
|
|
@@ -1,9 +0,0 @@
|
|
|
1
|
-
import type { ToolCallInfo, TurnInfo } from '../types/index.js';
|
|
2
|
-
import type { ClaudeSdkBaseMessage } from './shared.js';
|
|
3
|
-
export declare function isClaudeSdkResultMessage(message: ClaudeSdkBaseMessage): boolean;
|
|
4
|
-
export declare function extractAgentTrace(messages: ClaudeSdkBaseMessage[], timestamps?: number[]): {
|
|
5
|
-
turns: TurnInfo[];
|
|
6
|
-
toolCalls: ToolCallInfo[];
|
|
7
|
-
fullNumTurns: number;
|
|
8
|
-
numSubAgents: number;
|
|
9
|
-
};
|
|
@@ -1,24 +0,0 @@
|
|
|
1
|
-
import type { ExecResult } from '../types/index.js';
|
|
2
|
-
import type { CodexEvent } from './shared.js';
|
|
3
|
-
export declare function extractCodexUsage(events: CodexEvent[]): {
|
|
4
|
-
input: number;
|
|
5
|
-
cached: number;
|
|
6
|
-
output: number;
|
|
7
|
-
};
|
|
8
|
-
/**
|
|
9
|
-
* Match @openai/codex-sdk's `finalResponse`: the latest completed
|
|
10
|
-
* agent_message is the answer. Earlier messages remain available in `turns`.
|
|
11
|
-
*/
|
|
12
|
-
export declare function extractCodexFinalOutput(events: CodexEvent[]): string;
|
|
13
|
-
export declare function extractCodexProtocolError(events: CodexEvent[]): string | undefined;
|
|
14
|
-
export declare function validateCodexProtocol(events: CodexEvent[]): string | undefined;
|
|
15
|
-
export declare function extractCodexStopReason(events: CodexEvent[]): string;
|
|
16
|
-
export declare function sumCodexElapsed(resultEvents: CodexEvent[], wallClock: number): number;
|
|
17
|
-
export interface BuildCodexResultOptions {
|
|
18
|
-
events: CodexEvent[];
|
|
19
|
-
wallClockDurationMs: number;
|
|
20
|
-
source: 'codex --json' | 'codex-sdk';
|
|
21
|
-
malformedLineCount?: number;
|
|
22
|
-
forcedError?: string;
|
|
23
|
-
}
|
|
24
|
-
export declare function buildCodexResult({ events, wallClockDurationMs, source, malformedLineCount, forcedError, }: BuildCodexResultOptions): ExecResult;
|
package/dist/executors/gemini.js
DELETED
|
@@ -1,156 +0,0 @@
|
|
|
1
|
-
import { DEFAULT_TIMEOUT_MS, errorMessage, interruptedExecResult, parseJson, spawnWithSigintPropagation, timeoutExecResult, } from './shared.js';
|
|
2
|
-
import { optionalTokenCount } from '../shared/token-usage.js';
|
|
3
|
-
export async function geminiExecutor({ model, system, prompt, timeoutMs = DEFAULT_TIMEOUT_MS }) {
|
|
4
|
-
const fullPrompt = system ? `${system}\n\n${prompt}` : prompt;
|
|
5
|
-
const start = Date.now();
|
|
6
|
-
try {
|
|
7
|
-
const args = [];
|
|
8
|
-
if (model)
|
|
9
|
-
args.push('--model', model);
|
|
10
|
-
const { child, done } = spawnWithSigintPropagation('gemini', args, {
|
|
11
|
-
env: { ...process.env },
|
|
12
|
-
timeoutMs,
|
|
13
|
-
});
|
|
14
|
-
child.stdin?.on('error', () => undefined);
|
|
15
|
-
child.stdin?.write(fullPrompt);
|
|
16
|
-
child.stdin?.end();
|
|
17
|
-
let output;
|
|
18
|
-
try {
|
|
19
|
-
const r = await done;
|
|
20
|
-
output = r.stdout;
|
|
21
|
-
}
|
|
22
|
-
catch (err) {
|
|
23
|
-
const details = err;
|
|
24
|
-
if (details.killedByTimeout)
|
|
25
|
-
return timeoutExecResult(timeoutMs, Date.now() - start);
|
|
26
|
-
if (details.killedBySignal)
|
|
27
|
-
return interruptedExecResult(Date.now() - start);
|
|
28
|
-
const durationMs = Date.now() - start;
|
|
29
|
-
return {
|
|
30
|
-
ok: false,
|
|
31
|
-
error: details.stderr?.trim() || details.message || `gemini exited with code ${details.code ?? '?'}`,
|
|
32
|
-
durationMs,
|
|
33
|
-
durationApiMs: 0,
|
|
34
|
-
inputTokens: 0,
|
|
35
|
-
outputTokens: 0,
|
|
36
|
-
cacheReadTokens: 0,
|
|
37
|
-
cacheCreationTokens: 0,
|
|
38
|
-
tokenUsageReportedByExecutor: false,
|
|
39
|
-
costUSD: 0,
|
|
40
|
-
costReportedByExecutor: false,
|
|
41
|
-
output: details.stdout?.trim() || null,
|
|
42
|
-
stopReason: 'error',
|
|
43
|
-
numTurns: 0,
|
|
44
|
-
};
|
|
45
|
-
}
|
|
46
|
-
const durationMs = Date.now() - start;
|
|
47
|
-
let text = output;
|
|
48
|
-
let inputTokens = 0;
|
|
49
|
-
let outputTokens = 0;
|
|
50
|
-
let tokenUsageReported = false;
|
|
51
|
-
try {
|
|
52
|
-
const parsed = parseJson(output);
|
|
53
|
-
if (parsed && typeof parsed === 'object' && !Array.isArray(parsed)) {
|
|
54
|
-
const record = parsed;
|
|
55
|
-
const declaresProtocol = Object.hasOwn(record, 'response')
|
|
56
|
-
|| Object.hasOwn(record, 'stats');
|
|
57
|
-
if (declaresProtocol) {
|
|
58
|
-
const data = parsed;
|
|
59
|
-
if (typeof data.response !== 'string') {
|
|
60
|
-
return invalidGeminiProtocolResult(durationMs, '"response" must be a string');
|
|
61
|
-
}
|
|
62
|
-
text = data.response;
|
|
63
|
-
if (data.stats !== undefined
|
|
64
|
-
&& (!data.stats
|
|
65
|
-
|| typeof data.stats !== 'object'
|
|
66
|
-
|| Array.isArray(data.stats))) {
|
|
67
|
-
return invalidGeminiProtocolResult(durationMs, '"stats" must be an object when present');
|
|
68
|
-
}
|
|
69
|
-
if (data.stats !== undefined) {
|
|
70
|
-
const parsedInput = optionalTokenCount(data.stats.inputTokens);
|
|
71
|
-
const parsedOutput = optionalTokenCount(data.stats.outputTokens);
|
|
72
|
-
if (parsedInput === undefined || parsedOutput === undefined) {
|
|
73
|
-
return invalidGeminiProtocolResult(durationMs, 'invalid token usage');
|
|
74
|
-
}
|
|
75
|
-
inputTokens = parsedInput;
|
|
76
|
-
outputTokens = parsedOutput;
|
|
77
|
-
tokenUsageReported = true;
|
|
78
|
-
}
|
|
79
|
-
}
|
|
80
|
-
}
|
|
81
|
-
}
|
|
82
|
-
catch {
|
|
83
|
-
text = output.trim();
|
|
84
|
-
}
|
|
85
|
-
if (!text.trim()) {
|
|
86
|
-
return {
|
|
87
|
-
ok: false,
|
|
88
|
-
error: 'gemini completed without model output',
|
|
89
|
-
durationMs,
|
|
90
|
-
durationApiMs: 0,
|
|
91
|
-
inputTokens,
|
|
92
|
-
outputTokens,
|
|
93
|
-
cacheReadTokens: 0,
|
|
94
|
-
cacheCreationTokens: 0,
|
|
95
|
-
...(!tokenUsageReported && { tokenUsageReportedByExecutor: false }),
|
|
96
|
-
costUSD: 0,
|
|
97
|
-
costReportedByExecutor: false,
|
|
98
|
-
output: null,
|
|
99
|
-
stopReason: 'error',
|
|
100
|
-
numTurns: 1,
|
|
101
|
-
};
|
|
102
|
-
}
|
|
103
|
-
return {
|
|
104
|
-
ok: true,
|
|
105
|
-
durationMs,
|
|
106
|
-
durationApiMs: 0,
|
|
107
|
-
inputTokens,
|
|
108
|
-
outputTokens,
|
|
109
|
-
cacheReadTokens: 0,
|
|
110
|
-
cacheCreationTokens: 0,
|
|
111
|
-
...(!tokenUsageReported && { tokenUsageReportedByExecutor: false }),
|
|
112
|
-
costUSD: 0,
|
|
113
|
-
costReportedByExecutor: false,
|
|
114
|
-
output: text,
|
|
115
|
-
stopReason: 'end',
|
|
116
|
-
numTurns: 1,
|
|
117
|
-
};
|
|
118
|
-
}
|
|
119
|
-
catch (err) {
|
|
120
|
-
const durationMs = Date.now() - start;
|
|
121
|
-
return {
|
|
122
|
-
ok: false,
|
|
123
|
-
error: errorMessage(err),
|
|
124
|
-
durationMs,
|
|
125
|
-
durationApiMs: 0,
|
|
126
|
-
inputTokens: 0,
|
|
127
|
-
outputTokens: 0,
|
|
128
|
-
cacheReadTokens: 0,
|
|
129
|
-
cacheCreationTokens: 0,
|
|
130
|
-
tokenUsageReportedByExecutor: false,
|
|
131
|
-
costUSD: 0,
|
|
132
|
-
costReportedByExecutor: false,
|
|
133
|
-
output: null,
|
|
134
|
-
stopReason: 'error',
|
|
135
|
-
numTurns: 0,
|
|
136
|
-
};
|
|
137
|
-
}
|
|
138
|
-
}
|
|
139
|
-
function invalidGeminiProtocolResult(durationMs, message) {
|
|
140
|
-
return {
|
|
141
|
-
ok: false,
|
|
142
|
-
error: `gemini returned malformed protocol JSON: ${message}`,
|
|
143
|
-
durationMs,
|
|
144
|
-
durationApiMs: 0,
|
|
145
|
-
inputTokens: 0,
|
|
146
|
-
outputTokens: 0,
|
|
147
|
-
cacheReadTokens: 0,
|
|
148
|
-
cacheCreationTokens: 0,
|
|
149
|
-
tokenUsageReportedByExecutor: false,
|
|
150
|
-
costUSD: 0,
|
|
151
|
-
costReportedByExecutor: false,
|
|
152
|
-
output: null,
|
|
153
|
-
stopReason: 'error',
|
|
154
|
-
numTurns: 1,
|
|
155
|
-
};
|
|
156
|
-
}
|
|
@@ -1,6 +0,0 @@
|
|
|
1
|
-
import type { ExecutorRuntimeFingerprint } from '../types/index.js';
|
|
2
|
-
export interface ExecutorRuntimeFingerprintOptions {
|
|
3
|
-
skillDir?: string | null;
|
|
4
|
-
env?: NodeJS.ProcessEnv;
|
|
5
|
-
}
|
|
6
|
-
export declare function getExecutorRuntimeFingerprint(executorName: string, model: string, options?: ExecutorRuntimeFingerprintOptions): ExecutorRuntimeFingerprint;
|
|
@@ -1,226 +0,0 @@
|
|
|
1
|
-
import { execFile, type ChildProcess } from 'node:child_process';
|
|
2
|
-
import type { ExecResult } from '../types/index.js';
|
|
3
|
-
export declare const execFileAsync: typeof execFile.__promisify__;
|
|
4
|
-
export declare const DEFAULT_MODEL = "sonnet";
|
|
5
|
-
export declare const JUDGE_MODEL = "haiku";
|
|
6
|
-
export declare const DEFAULT_TIMEOUT_MS = 600000;
|
|
7
|
-
export declare const MAX_BUFFER: number;
|
|
8
|
-
export type ExecutorVendor = 'anthropic' | 'openai' | 'google' | 'unknown';
|
|
9
|
-
export declare function executorVendor(executor: string): ExecutorVendor;
|
|
10
|
-
export interface TokenUsage {
|
|
11
|
-
input_tokens?: number;
|
|
12
|
-
output_tokens?: number;
|
|
13
|
-
cache_read_input_tokens?: number;
|
|
14
|
-
cache_creation_input_tokens?: number;
|
|
15
|
-
}
|
|
16
|
-
export interface ClaudeCliResponse {
|
|
17
|
-
is_error?: boolean;
|
|
18
|
-
duration_ms?: number;
|
|
19
|
-
duration_api_ms?: number;
|
|
20
|
-
usage?: TokenUsage;
|
|
21
|
-
total_cost_usd?: number;
|
|
22
|
-
result?: string;
|
|
23
|
-
stop_reason?: string | null;
|
|
24
|
-
num_turns?: number;
|
|
25
|
-
}
|
|
26
|
-
export interface OpenAiUsage {
|
|
27
|
-
prompt_tokens?: number;
|
|
28
|
-
completion_tokens?: number;
|
|
29
|
-
prompt_tokens_details?: {
|
|
30
|
-
cached_tokens?: number;
|
|
31
|
-
};
|
|
32
|
-
}
|
|
33
|
-
export interface OpenAiResponse {
|
|
34
|
-
usage?: OpenAiUsage;
|
|
35
|
-
choices?: Array<{
|
|
36
|
-
message?: {
|
|
37
|
-
content?: string | null;
|
|
38
|
-
refusal?: string | null;
|
|
39
|
-
};
|
|
40
|
-
finish_reason?: string;
|
|
41
|
-
}>;
|
|
42
|
-
error?: {
|
|
43
|
-
message?: string;
|
|
44
|
-
};
|
|
45
|
-
}
|
|
46
|
-
export interface GeminiResponse {
|
|
47
|
-
response?: string;
|
|
48
|
-
stats?: {
|
|
49
|
-
inputTokens?: number;
|
|
50
|
-
outputTokens?: number;
|
|
51
|
-
};
|
|
52
|
-
}
|
|
53
|
-
export interface AnthropicResponse {
|
|
54
|
-
usage?: TokenUsage;
|
|
55
|
-
content?: Array<{
|
|
56
|
-
type?: string;
|
|
57
|
-
text?: string;
|
|
58
|
-
}>;
|
|
59
|
-
stop_reason?: string;
|
|
60
|
-
error?: {
|
|
61
|
-
message?: string;
|
|
62
|
-
};
|
|
63
|
-
}
|
|
64
|
-
export interface ClaudeSdkQueryOptions {
|
|
65
|
-
model?: string;
|
|
66
|
-
systemPrompt?: string;
|
|
67
|
-
cwd: string;
|
|
68
|
-
permissionMode: 'bypassPermissions';
|
|
69
|
-
allowDangerouslySkipPermissions: true;
|
|
70
|
-
abortController: AbortController;
|
|
71
|
-
env: NodeJS.ProcessEnv;
|
|
72
|
-
}
|
|
73
|
-
export interface ClaudeSdkQueryInput {
|
|
74
|
-
prompt: string;
|
|
75
|
-
options: ClaudeSdkQueryOptions;
|
|
76
|
-
}
|
|
77
|
-
export interface ClaudeSdkBaseMessage {
|
|
78
|
-
type: string;
|
|
79
|
-
message?: {
|
|
80
|
-
role?: string;
|
|
81
|
-
content?: Array<{
|
|
82
|
-
type: string;
|
|
83
|
-
text?: string;
|
|
84
|
-
id?: string;
|
|
85
|
-
name?: string;
|
|
86
|
-
input?: unknown;
|
|
87
|
-
}>;
|
|
88
|
-
};
|
|
89
|
-
tool_use_id?: string;
|
|
90
|
-
content?: string | Array<{
|
|
91
|
-
type: string;
|
|
92
|
-
text?: string;
|
|
93
|
-
}>;
|
|
94
|
-
is_error?: boolean;
|
|
95
|
-
}
|
|
96
|
-
export interface ClaudeSdkResultMessage extends ClaudeSdkBaseMessage {
|
|
97
|
-
type: 'result';
|
|
98
|
-
result?: string;
|
|
99
|
-
usage?: TokenUsage;
|
|
100
|
-
total_cost_usd?: number;
|
|
101
|
-
duration_api_ms?: number;
|
|
102
|
-
duration_ms?: number;
|
|
103
|
-
num_turns?: number;
|
|
104
|
-
stop_reason?: string | null;
|
|
105
|
-
modelUsage?: Record<string, {
|
|
106
|
-
inputTokens?: number;
|
|
107
|
-
outputTokens?: number;
|
|
108
|
-
cacheReadInputTokens?: number;
|
|
109
|
-
cacheCreationInputTokens?: number;
|
|
110
|
-
}>;
|
|
111
|
-
subtype?: string;
|
|
112
|
-
errors?: string[];
|
|
113
|
-
}
|
|
114
|
-
export interface ClaudeSdkModule {
|
|
115
|
-
query: (opts: ClaudeSdkQueryInput) => AsyncIterable<ClaudeSdkBaseMessage>;
|
|
116
|
-
}
|
|
117
|
-
export interface CodexEvent {
|
|
118
|
-
type?: string;
|
|
119
|
-
turn_id?: string;
|
|
120
|
-
usage?: {
|
|
121
|
-
input_tokens?: number;
|
|
122
|
-
cached_input_tokens?: number;
|
|
123
|
-
output_tokens?: number;
|
|
124
|
-
reasoning_output_tokens?: number;
|
|
125
|
-
};
|
|
126
|
-
elapsed_ms?: number;
|
|
127
|
-
stop_reason?: string;
|
|
128
|
-
item?: {
|
|
129
|
-
id?: string;
|
|
130
|
-
type?: string;
|
|
131
|
-
text?: string;
|
|
132
|
-
command?: string;
|
|
133
|
-
aggregated_output?: string;
|
|
134
|
-
exit_code?: number | null;
|
|
135
|
-
status?: string;
|
|
136
|
-
path?: string;
|
|
137
|
-
content?: string;
|
|
138
|
-
query?: string;
|
|
139
|
-
results?: unknown[];
|
|
140
|
-
changes?: Array<{
|
|
141
|
-
path?: string;
|
|
142
|
-
changeKind?: string;
|
|
143
|
-
}>;
|
|
144
|
-
server?: string;
|
|
145
|
-
tool?: string;
|
|
146
|
-
name?: string;
|
|
147
|
-
arguments?: unknown;
|
|
148
|
-
result?: unknown;
|
|
149
|
-
message?: string;
|
|
150
|
-
error?: {
|
|
151
|
-
message?: string;
|
|
152
|
-
};
|
|
153
|
-
};
|
|
154
|
-
error?: {
|
|
155
|
-
message?: string;
|
|
156
|
-
};
|
|
157
|
-
message?: string;
|
|
158
|
-
ts?: number;
|
|
159
|
-
}
|
|
160
|
-
/**
|
|
161
|
-
* Translate Codex's external event shape into omk's internal protocol model.
|
|
162
|
-
* Codex currently calls file-change discriminators `kind`; omk reserves bare
|
|
163
|
-
* `kind` for ArtifactKind, so the raw field is qualified at the boundary.
|
|
164
|
-
*/
|
|
165
|
-
export declare function normalizeCodexProtocolEvent(value: unknown): CodexEvent | null;
|
|
166
|
-
export interface ExecutorErrorLike {
|
|
167
|
-
message?: string;
|
|
168
|
-
name?: string;
|
|
169
|
-
killed?: boolean;
|
|
170
|
-
stdout?: string;
|
|
171
|
-
}
|
|
172
|
-
export declare function asErrorLike(err: unknown): ExecutorErrorLike;
|
|
173
|
-
export declare function errorMessage(err: unknown, fallback?: string): string;
|
|
174
|
-
export declare function parseJson<T>(content: string): T;
|
|
175
|
-
export interface JsonResponseBody<T> {
|
|
176
|
-
data: T | null;
|
|
177
|
-
rawBody: string;
|
|
178
|
-
}
|
|
179
|
-
export declare function readJsonResponse<T>(response: Response): Promise<JsonResponseBody<T>>;
|
|
180
|
-
export declare function responseBodyPreview(rawBody: string, maxLength?: number): string;
|
|
181
|
-
export declare function buildExecEnv(skillDir?: string | null): NodeJS.ProcessEnv;
|
|
182
|
-
export declare function timeoutExecResult(timeoutMs: number, durationMs: number): ExecResult;
|
|
183
|
-
export declare function interruptedExecResult(durationMs: number): ExecResult;
|
|
184
|
-
/**
|
|
185
|
-
* Register an in-process runtime (for example an SDK-owned child) with the
|
|
186
|
-
* same SIGINT coordinator used by spawned executors.
|
|
187
|
-
*/
|
|
188
|
-
export declare function registerSigintSubscriber(subscriber: () => void): () => void;
|
|
189
|
-
export declare function __resetSigintRegistryForTest(): void;
|
|
190
|
-
export interface SpawnHelperResult {
|
|
191
|
-
stdout: string;
|
|
192
|
-
stderr: string;
|
|
193
|
-
code: number | null;
|
|
194
|
-
signal: NodeJS.Signals | null;
|
|
195
|
-
killedByTimeout: boolean;
|
|
196
|
-
killedBySignal: NodeJS.Signals | null;
|
|
197
|
-
}
|
|
198
|
-
export interface SpawnHelperError extends Error {
|
|
199
|
-
stdout?: string;
|
|
200
|
-
stderr?: string;
|
|
201
|
-
code?: number | null;
|
|
202
|
-
signal?: NodeJS.Signals | null;
|
|
203
|
-
killedByTimeout?: boolean;
|
|
204
|
-
killedBySignal?: NodeJS.Signals | null;
|
|
205
|
-
}
|
|
206
|
-
export interface SpawnHelperOptions {
|
|
207
|
-
cwd?: string;
|
|
208
|
-
env?: NodeJS.ProcessEnv;
|
|
209
|
-
/** kill child after this many ms; reject with killedByTimeout=true */
|
|
210
|
-
timeoutMs?: number;
|
|
211
|
-
/** per-stream stdout/stderr byte limit; reject when either stream exceeds it */
|
|
212
|
-
maxBuffer?: number;
|
|
213
|
-
/** external abort signal; abort() 走跟 SIGINT 同一 grace 路径 */
|
|
214
|
-
abortSignal?: AbortSignal;
|
|
215
|
-
}
|
|
216
|
-
/**
|
|
217
|
-
* spawn child + 注册到全局 SIGINT registry。返回 { child, done } 让 caller 自己
|
|
218
|
-
* 操作 stdin(写入 / 关闭),通过 done 等结果。
|
|
219
|
-
*
|
|
220
|
-
* 适用:claude / codex / gemini / script CLI 子进程。HTTP executor(*-api)用 fetch +
|
|
221
|
-
* AbortSignal.timeout,自带 abort,不走这个。claude-sdk in-process 也不走。
|
|
222
|
-
*/
|
|
223
|
-
export declare function spawnWithSigintPropagation(command: string, args: string[], options?: SpawnHelperOptions): {
|
|
224
|
-
child: ChildProcess;
|
|
225
|
-
done: Promise<SpawnHelperResult>;
|
|
226
|
-
};
|
|
File without changes
|
|
File without changes
|