oh-my-knowledge 0.52.3 → 0.54.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +21 -1
- package/README.zh.md +21 -1
- package/dist/assets/agent-skills/omk/SKILL.md +6 -0
- package/dist/assets/agent-skills/omk/references/commands.md +1 -1
- package/dist/authoring/evolver.js +1 -1
- package/dist/authoring/generator.js +1 -1
- package/dist/cli/commands/eval/index.js +7 -6
- package/dist/cli/commands/install.js +5 -1
- package/dist/cli/lib/generation-failure-hint.js +6 -8
- package/dist/cli/lib/runtime-defaults.d.ts +2 -0
- package/dist/cli/lib/runtime-defaults.js +7 -4
- package/dist/dsh-plugin/cordis.patch.yml +3 -0
- package/dist/dsh-plugin/host-executor.d.ts +93 -0
- package/dist/dsh-plugin/host-executor.js +232 -0
- package/dist/dsh-plugin/index.d.ts +30 -0
- package/dist/dsh-plugin/index.js +275 -0
- package/dist/dsh-plugin/observe.d.ts +47 -0
- package/dist/dsh-plugin/observe.js +359 -0
- package/dist/dsh-plugin/protocol.d.ts +21 -0
- package/dist/dsh-plugin/protocol.js +229 -0
- package/dist/dsh-plugin/trace-adapter.d.ts +48 -0
- package/dist/dsh-plugin/trace-adapter.js +590 -0
- package/dist/eval-core/comparability.js +3 -0
- package/dist/eval-core/evaluation-execution.js +5 -13
- package/dist/eval-core/evaluation-reporting.d.ts +7 -4
- package/dist/eval-core/evaluation-reporting.js +39 -18
- package/dist/eval-core/judge-independence.d.ts +1 -1
- package/dist/eval-core/judge-independence.js +1 -1
- package/dist/eval-core/report-document.js +6 -0
- package/dist/eval-core/resume-compatibility.d.ts +1 -0
- package/dist/eval-core/resume-compatibility.js +5 -4
- package/dist/eval-workflows/batch-evaluation-workflow.js +1 -1
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +4 -2
- package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
- package/dist/eval-workflows/evaluation-pipeline.js +7 -3
- package/dist/eval-workflows/run-evaluation.d.ts +4 -2
- package/dist/eval-workflows/run-evaluation.js +7 -3
- package/dist/executors/{anthropic-api.d.ts → anthropic/api.d.ts} +1 -1
- package/dist/executors/{anthropic-api.js → anthropic/api.js} +4 -2
- package/dist/executors/{claude-cli.d.ts → anthropic/claude/cli.d.ts} +1 -1
- package/dist/executors/{claude-cli.js → anthropic/claude/cli.js} +5 -3
- package/dist/executors/anthropic/claude/protocol.d.ts +87 -0
- package/dist/executors/{claude-protocol.js → anthropic/claude/protocol.js} +6 -6
- package/dist/executors/{claude-sdk.d.ts → anthropic/claude/sdk.d.ts} +2 -2
- package/dist/executors/{claude-sdk.js → anthropic/claude/sdk.js} +8 -5
- package/dist/executors/anthropic/claude/trace.d.ts +9 -0
- package/dist/executors/{claude-sdk-trace.js → anthropic/claude/trace.js} +5 -5
- package/dist/executors/{capabilities.d.ts → core/capabilities.d.ts} +3 -5
- package/dist/executors/{capabilities.js → core/capabilities.js} +4 -11
- package/dist/executors/core/http.d.ts +6 -0
- package/dist/executors/core/http.js +19 -0
- package/dist/executors/core/limits.d.ts +2 -0
- package/dist/executors/core/limits.js +2 -0
- package/dist/executors/core/optional-dependencies.d.ts +7 -0
- package/dist/executors/core/optional-dependencies.js +35 -0
- package/dist/executors/core/registry.d.ts +145 -0
- package/dist/executors/core/registry.js +127 -0
- package/dist/executors/core/runtime-fingerprint.d.ts +13 -0
- package/dist/executors/{runtime-fingerprint.js → core/runtime-fingerprint.js} +119 -59
- package/dist/executors/core/runtime.d.ts +12 -0
- package/dist/executors/core/runtime.js +61 -0
- package/dist/executors/core/subprocess.d.ts +44 -0
- package/dist/executors/{shared.js → core/subprocess.js} +20 -156
- package/dist/executors/index.d.ts +4 -4
- package/dist/executors/index.js +26 -15
- package/dist/executors/{openai-api.d.ts → openai/api.d.ts} +1 -1
- package/dist/executors/{openai-api.js → openai/api.js} +4 -2
- package/dist/executors/{codex-cli.d.ts → openai/codex/cli.d.ts} +3 -3
- package/dist/executors/{codex-cli.js → openai/codex/cli.js} +5 -3
- package/dist/executors/openai/codex/protocol.d.ts +72 -0
- package/dist/executors/{codex-protocol.js → openai/codex/protocol.js} +33 -2
- package/dist/executors/{codex-sdk.d.ts → openai/codex/sdk.d.ts} +18 -4
- package/dist/executors/{codex-sdk.js → openai/codex/sdk.js} +7 -4
- package/dist/executors/{codex-cli-trace.d.ts → openai/codex/trace.d.ts} +2 -2
- package/dist/executors/{codex-cli-trace.js → openai/codex/trace.js} +4 -4
- package/dist/executors/{script.d.ts → script/index.d.ts} +1 -1
- package/dist/executors/{script.js → script/index.js} +6 -4
- package/dist/grading/judge.d.ts +1 -1
- package/dist/grading/judge.js +1 -1
- package/dist/observability/conversation-catalog.js +1 -1
- package/dist/observability/experience.js +4 -0
- package/dist/observability/inbox.d.ts +4 -1
- package/dist/observability/inbox.js +7 -2
- package/dist/observability/trace-ir.d.ts +6 -3
- package/dist/observability/turn-index.js +4 -0
- package/dist/renderer/conversation-renderer.js +20 -6
- package/dist/renderer/html-renderer.js +30 -11
- package/dist/renderer/knowledge-debugger-renderer.js +6 -1
- package/dist/renderer/observation-inbox-renderer.js +2 -2
- package/dist/renderer/trajectory-live.d.ts +1 -0
- package/dist/renderer/trajectory-live.js +5 -2
- package/dist/shared/trace-source-kind.js +1 -0
- package/dist/types/executor.d.ts +13 -2
- package/dist/types/judge.d.ts +2 -2
- package/dist/types/observability.d.ts +1 -1
- package/dist/types/trace.d.ts +1 -1
- package/package.json +25 -5
- package/dist/executors/claude-protocol.d.ts +0 -28
- package/dist/executors/claude-sdk-trace.d.ts +0 -9
- package/dist/executors/codex-protocol.d.ts +0 -24
- package/dist/executors/gemini.d.ts +0 -2
- package/dist/executors/gemini.js +0 -156
- package/dist/executors/runtime-fingerprint.d.ts +0 -6
- package/dist/executors/shared.d.ts +0 -226
- /package/dist/executors/{script-command.d.ts → script/command.d.ts} +0 -0
- /package/dist/executors/{script-command.js → script/command.js} +0 -0
|
@@ -1,7 +1,9 @@
|
|
|
1
|
-
import { materializeForCliConfigDir } from '
|
|
2
|
-
import { executorResultValidationError, normalizeExecResultToolIdentities, } from '
|
|
3
|
-
import { resolveScriptCommand } from './
|
|
4
|
-
import { DEFAULT_TIMEOUT_MS
|
|
1
|
+
import { materializeForCliConfigDir } from '../../eval-core/mocks-runtime.js';
|
|
2
|
+
import { executorResultValidationError, normalizeExecResultToolIdentities, } from '../../shared/executor-result.js';
|
|
3
|
+
import { resolveScriptCommand } from './command.js';
|
|
4
|
+
import { DEFAULT_TIMEOUT_MS } from '../core/limits.js';
|
|
5
|
+
import { interruptedExecResult, timeoutExecResult, } from '../core/runtime.js';
|
|
6
|
+
import { spawnWithSigintPropagation } from '../core/subprocess.js';
|
|
5
7
|
// script executor 由用户自定义,omk 无法保证它实现 skill 隔离。
|
|
6
8
|
// 任何 allowedSkills(包括 [])下都 stderr 一次性 warn,不阻塞执行,
|
|
7
9
|
// 让用户知道 strict-baseline / 显式 allowedSkills 在 script executor 下静默无效。
|
package/dist/grading/judge.d.ts
CHANGED
|
@@ -53,7 +53,7 @@ export declare function judgeId(config: JudgeConfig): string;
|
|
|
53
53
|
export declare function computeJudgeAgreement(judgeScores: number[][]): JudgeAgreement;
|
|
54
54
|
/**
|
|
55
55
|
* Judge a single (output, rubric) pair with N judge models in parallel. Each judge
|
|
56
|
-
* may use a different executor (e.g. claude:opus + openai-api:gpt-4o
|
|
56
|
+
* may use a different executor (e.g. claude:opus + openai-api:gpt-4o). Each
|
|
57
57
|
* judge can also be repeated `judgeRepeat` times — final per-judge score is its mean.
|
|
58
58
|
*
|
|
59
59
|
* Returns: aggregate DimensionResult (score = mean across judges; this is the "consensus"
|
package/dist/grading/judge.js
CHANGED
|
@@ -342,7 +342,7 @@ export function computeJudgeAgreement(judgeScores) {
|
|
|
342
342
|
}
|
|
343
343
|
/**
|
|
344
344
|
* Judge a single (output, rubric) pair with N judge models in parallel. Each judge
|
|
345
|
-
* may use a different executor (e.g. claude:opus + openai-api:gpt-4o
|
|
345
|
+
* may use a different executor (e.g. claude:opus + openai-api:gpt-4o). Each
|
|
346
346
|
* judge can also be repeated `judgeRepeat` times — final per-judge score is its mean.
|
|
347
347
|
*
|
|
348
348
|
* Returns: aggregate DimensionResult (score = mean across judges; this is the "consensus"
|
|
@@ -542,7 +542,7 @@ function trajectoryRevisionFromStat(sourceStat, status) {
|
|
|
542
542
|
return `${sourceStat.size}:${sourceStat.mtimeMs}:${status}`;
|
|
543
543
|
}
|
|
544
544
|
function isTerminalTaskStatus(status) {
|
|
545
|
-
return status === 'completed' || status === 'aborted' || status === 'interrupted';
|
|
545
|
+
return status === 'completed' || status === 'failed' || status === 'aborted' || status === 'interrupted';
|
|
546
546
|
}
|
|
547
547
|
function secondsToMs(value) {
|
|
548
548
|
const seconds = numberValue(value);
|
|
@@ -1012,6 +1012,7 @@ function isExperienceTurnSummaryArray(value) {
|
|
|
1012
1012
|
&& isOptionalTimestamp(turn.startTimestamp)
|
|
1013
1013
|
&& isOptionalTimestamp(turn.endTimestamp)
|
|
1014
1014
|
&& (turn.status === 'completed'
|
|
1015
|
+
|| turn.status === 'failed'
|
|
1015
1016
|
|| turn.status === 'aborted'
|
|
1016
1017
|
|| turn.status === 'interrupted'
|
|
1017
1018
|
|| turn.status === 'open'
|
|
@@ -2621,6 +2622,9 @@ function timelineEventsFromTraceEvent(session, event, eventIndex) {
|
|
|
2621
2622
|
memoryMode: event.memoryMode,
|
|
2622
2623
|
historyMode: event.historyMode,
|
|
2623
2624
|
contextWindowId: event.contextWindowId,
|
|
2625
|
+
parentRunId: event.parentRunId,
|
|
2626
|
+
delegationDepth: event.delegationDepth,
|
|
2627
|
+
sourceOrigin: event.sourceOrigin,
|
|
2624
2628
|
availableTools: event.availableTools,
|
|
2625
2629
|
instructions: event.instructions,
|
|
2626
2630
|
goal: event.goal,
|
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import type { BuildObservationInboxReportOptions, ObservationEvidence, ObservationInboxItem, ObservationInboxReport, ObservationMessageRef, ObservationMessageWindow, ObservationSessionTimeRange, ObservationSeverityReasonCode, ObservationSignalSubtype, ObservationSignalType, ObservationSkillRollup, ObservationSourceKind } from '../types/index.js';
|
|
1
|
+
import type { BuildObservationInboxReportOptions, ObservationEvidence, ObservationInboxItem, ObservationInboxReport, ObservationMessageRef, ObservationMessageWindow, ObservationSessionTimeRange, ObservationSeverityReasonCode, ObservationSignalSubtype, ObservationSignalType, ObservationSkillRollup, ObservationSourceKind, TraceIngestionSummary } from '../types/index.js';
|
|
2
|
+
import { type TraceSession } from './trace-adapter.js';
|
|
2
3
|
import { type PersistedObservationExperienceReport } from './experience.js';
|
|
3
4
|
export type { BuildObservationInboxReportOptions, ObservationEvidence, ObservationInboxItem, ObservationInboxReport, ObservationMessageRef, ObservationMessageWindow, ObservationSessionTimeRange, ObservationSeverityReasonCode, ObservationSignalSubtype, ObservationSignalType, ObservationSkillRollup, ObservationSourceKind, };
|
|
4
5
|
export type PersistedObservationInboxReport = Omit<ObservationInboxReport, 'experience'> & {
|
|
@@ -19,6 +20,8 @@ export declare function severityReasonCodeFor(item: Pick<ObservationInboxItem, '
|
|
|
19
20
|
* `buildObserveDiagnosticsFromReport(report)` 写入 `report.diagnostics`。
|
|
20
21
|
*/
|
|
21
22
|
export declare function buildObservationInboxReport(tracePath: string, options?: BuildObservationInboxReportOptions): ObservationInboxReport;
|
|
23
|
+
/** Build the existing inbox artifact from an in-memory source-neutral corpus. */
|
|
24
|
+
export declare function buildObservationInboxReportFromTraceSessions(tracePath: string, sessions: TraceSession[], ingestion: TraceIngestionSummary, options?: BuildObservationInboxReportOptions): ObservationInboxReport;
|
|
22
25
|
export declare function aggregateInboxItems(items: ObservationInboxItem[]): ObservationInboxItem[];
|
|
23
26
|
export declare function saveObservationInboxReport(report: ObservationInboxReport, outDir?: string): string;
|
|
24
27
|
export declare function compactObservationInboxReport(report: ObservationInboxReport): PersistedObservationInboxReport;
|
|
@@ -5,7 +5,7 @@ import { OMK_HOME } from '../eval-core/default-dirs.js';
|
|
|
5
5
|
import { isReportFileName, randomRunToken, reportFilePath } from '../eval-core/artifact-file-names.js';
|
|
6
6
|
import { migrateLegacyReportFiles } from '../eval-core/report-file-migration.js';
|
|
7
7
|
import { extractGapSignalsFromTrace } from '../analysis/gap-analyzer.js';
|
|
8
|
-
import { loadTraceSessions, tracesToResultEntries, skillSegmentTimestampObserved, } from './trace-adapter.js';
|
|
8
|
+
import { loadTraceSessions, segmentTraceBySkill, tracesToResultEntries, skillSegmentTimestampObserved, } from './trace-adapter.js';
|
|
9
9
|
import { normalizeTraceTimestamp } from './trace-ir.js';
|
|
10
10
|
import { isSearchToolCall, toolCallQuery } from '../shared/tool-search.js';
|
|
11
11
|
import { isToolCallFailure, isToolCallSuccess } from '../shared/tool-call-status.js';
|
|
@@ -410,7 +410,12 @@ function skillSessionCountKey(segment) {
|
|
|
410
410
|
* `buildObserveDiagnosticsFromReport(report)` 写入 `report.diagnostics`。
|
|
411
411
|
*/
|
|
412
412
|
export function buildObservationInboxReport(tracePath, options = {}) {
|
|
413
|
-
const { sessions,
|
|
413
|
+
const { sessions, ingestion } = tracesToResultEntries(tracePath);
|
|
414
|
+
return buildObservationInboxReportFromTraceSessions(tracePath, sessions, ingestion, options);
|
|
415
|
+
}
|
|
416
|
+
/** Build the existing inbox artifact from an in-memory source-neutral corpus. */
|
|
417
|
+
export function buildObservationInboxReportFromTraceSessions(tracePath, sessions, ingestion, options = {}) {
|
|
418
|
+
const segments = sessions.flatMap(segmentTraceBySkill);
|
|
414
419
|
const skillSegments = segments.filter((segment) => segment.skillName !== 'general');
|
|
415
420
|
const generatedAt = new Date().toISOString();
|
|
416
421
|
const sessionTimeRanges = buildSessionTimeRanges(sessions);
|
|
@@ -60,8 +60,8 @@ export interface TraceUsageEvent extends TraceEventBase {
|
|
|
60
60
|
model?: string;
|
|
61
61
|
inputTokens: number;
|
|
62
62
|
outputTokens: number;
|
|
63
|
-
cacheReadTokens
|
|
64
|
-
cacheCreationTokens
|
|
63
|
+
cacheReadTokens?: number;
|
|
64
|
+
cacheCreationTokens?: number;
|
|
65
65
|
reasoningTokens?: number;
|
|
66
66
|
}
|
|
67
67
|
export interface TraceModelActivityEvent extends TraceEventBase {
|
|
@@ -74,7 +74,7 @@ export interface TraceModelActivityEvent extends TraceEventBase {
|
|
|
74
74
|
}
|
|
75
75
|
export interface TraceLifecycleEvent extends TraceEventBase {
|
|
76
76
|
eventKind: 'lifecycle';
|
|
77
|
-
phase: 'session_started' | 'session_ended' | 'turn_started' | 'turn_completed' | 'turn_aborted' | 'turn_interrupted';
|
|
77
|
+
phase: 'session_started' | 'session_ended' | 'turn_started' | 'turn_completed' | 'turn_failed' | 'turn_aborted' | 'turn_interrupted' | 'turn_ended_unknown' | 'step_started' | 'step_completed';
|
|
78
78
|
reason?: string;
|
|
79
79
|
durationMs?: number;
|
|
80
80
|
}
|
|
@@ -105,6 +105,9 @@ export interface TraceRuntimeContextEvent extends TraceEventBase {
|
|
|
105
105
|
memoryMode?: string;
|
|
106
106
|
historyMode?: string;
|
|
107
107
|
contextWindowId?: string;
|
|
108
|
+
parentRunId?: string;
|
|
109
|
+
delegationDepth?: number;
|
|
110
|
+
sourceOrigin?: string;
|
|
108
111
|
availableTools?: string[];
|
|
109
112
|
instructions?: string;
|
|
110
113
|
goal?: string;
|
|
@@ -1,8 +1,10 @@
|
|
|
1
1
|
import { createHash } from 'node:crypto';
|
|
2
2
|
const TERMINAL_LIFECYCLE_LABELS = new Set([
|
|
3
3
|
'turn_completed',
|
|
4
|
+
'turn_failed',
|
|
4
5
|
'turn_aborted',
|
|
5
6
|
'turn_interrupted',
|
|
7
|
+
'turn_ended_unknown',
|
|
6
8
|
]);
|
|
7
9
|
/** Reconstruct every observable task without consulting Skill attribution. */
|
|
8
10
|
export function reconstructExperienceTurns(timeline) {
|
|
@@ -124,6 +126,8 @@ function turnStatus(events) {
|
|
|
124
126
|
const labels = new Set(events.map((event) => event.label));
|
|
125
127
|
if (labels.has('turn_completed'))
|
|
126
128
|
return 'completed';
|
|
129
|
+
if (labels.has('turn_failed'))
|
|
130
|
+
return 'failed';
|
|
127
131
|
if (labels.has('turn_aborted'))
|
|
128
132
|
return 'aborted';
|
|
129
133
|
if (labels.has('turn_interrupted'))
|
|
@@ -54,8 +54,8 @@ export function renderConversationIndexPage(model, lang = DEFAULT_LANG) {
|
|
|
54
54
|
const rows = conversations.map((conversation) => renderConversationRow(conversation, lang)).join('');
|
|
55
55
|
const content = model.conversations.length > 0
|
|
56
56
|
? `<div class="conversation-list" role="list">${rows}</div>`
|
|
57
|
-
: `<div class="empty-state"><strong>${zh ? '
|
|
58
|
-
return layout(zh ? '
|
|
57
|
+
: `<div class="empty-state"><strong>${zh ? '还没有可浏览的对话' : 'No conversations yet'}</strong><span>${zh ? '从受支持的 runtime 摄取任务轨迹后,会话会显示在这里。' : 'Sessions appear here after a supported runtime supplies task trajectories.'}</span></div>`;
|
|
58
|
+
return layout(zh ? '对话' : 'Conversations', `
|
|
59
59
|
<main class="conversation-page conversation-index-app" data-activity-revision="${e(activity.revision)}">
|
|
60
60
|
<header class="conversation-app-head">
|
|
61
61
|
<a class="conversation-app-brand" href="/${langQuery}">
|
|
@@ -67,7 +67,7 @@ export function renderConversationIndexPage(model, lang = DEFAULT_LANG) {
|
|
|
67
67
|
</nav>
|
|
68
68
|
</header>
|
|
69
69
|
<div class="conversation-app-body">
|
|
70
|
-
<section class="conversation-browser" aria-label="${zh ? '
|
|
70
|
+
<section class="conversation-browser" aria-label="${zh ? '对话' : 'Conversations'}">
|
|
71
71
|
<header class="conversation-toolbar">
|
|
72
72
|
<div class="conversation-browser-title">
|
|
73
73
|
<h1>${zh ? '对话' : 'Conversations'}</h1>
|
|
@@ -117,7 +117,7 @@ export function renderConversationDetailPage(conversation, lang = DEFAULT_LANG)
|
|
|
117
117
|
<header class="conversation-page-head conversation-detail-head">
|
|
118
118
|
<div>
|
|
119
119
|
<a class="back-link" href="/conversations${langSuffix}">${zh ? '返回对话总览' : 'Back to conversations'}</a>
|
|
120
|
-
<p class="conversation-eyebrow"
|
|
120
|
+
<p class="conversation-eyebrow">${e(sourceLabel(conversation))} · ${e(shortThreadId(conversation.sourceThreadId))}</p>
|
|
121
121
|
<h1>${renderSafeInlineMarkdown(conversation.title)}</h1>
|
|
122
122
|
<div class="detail-meta">
|
|
123
123
|
${conversation.model ? `<span>${e(conversation.model)}</span>` : ''}
|
|
@@ -157,8 +157,8 @@ function renderConversationRow(conversation, lang) {
|
|
|
157
157
|
? '<span class="index-pending">—</span>'
|
|
158
158
|
: `<span><b>${conversation.turnCount}</b>${zh ? '任务' : 'tasks'}</span>`;
|
|
159
159
|
const activity = openTask
|
|
160
|
-
? `<strong class="running-label"><i aria-hidden="true"></i>${zh ? '进行中' : 'Running'}</strong><span>${e(updated)} · ${e(conversation.model ??
|
|
161
|
-
: `<strong>${e(updated)}</strong><span>${e(conversation.model ??
|
|
160
|
+
? `<strong class="running-label"><i aria-hidden="true"></i>${zh ? '进行中' : 'Running'}</strong><span>${e(updated)} · ${e(conversation.model ?? sourceLabel(conversation))}</span>`
|
|
161
|
+
: `<strong>${e(updated)}</strong><span>${e(conversation.model ?? sourceLabel(conversation))}</span>`;
|
|
162
162
|
const liveTaskLink = openTaskHref
|
|
163
163
|
? `<a class="live-task-link" href="${e(openTaskHref)}" aria-label="${zh ? '查看进行中任务的实时轨迹' : 'View the live trajectory for the running task'}"><i aria-hidden="true"></i>${zh ? '查看实时轨迹' : 'View live'}</a>`
|
|
164
164
|
: '';
|
|
@@ -345,10 +345,24 @@ function compactPath(value) {
|
|
|
345
345
|
function shortThreadId(value) {
|
|
346
346
|
return value.length > 18 ? `${value.slice(0, 8)}…${value.slice(-6)}` : value;
|
|
347
347
|
}
|
|
348
|
+
function sourceLabel(conversation) {
|
|
349
|
+
if (conversation.sourceKind === 'dsh')
|
|
350
|
+
return 'DeepSeek Harness';
|
|
351
|
+
if (conversation.sourceKind === 'codex')
|
|
352
|
+
return 'Codex';
|
|
353
|
+
if (conversation.sourceKind === 'claude')
|
|
354
|
+
return 'Claude';
|
|
355
|
+
if (conversation.sourceKind === 'openclaw')
|
|
356
|
+
return 'OpenClaw';
|
|
357
|
+
if (conversation.sourceKind === 'markdown_log')
|
|
358
|
+
return 'Markdown';
|
|
359
|
+
return 'Runtime';
|
|
360
|
+
}
|
|
348
361
|
function statusLabel(status, lang) {
|
|
349
362
|
const zh = lang === 'zh';
|
|
350
363
|
const labels = {
|
|
351
364
|
completed: ['已完成', 'Completed'],
|
|
365
|
+
failed: ['已失败', 'Failed'],
|
|
352
366
|
aborted: ['已中止', 'Aborted'],
|
|
353
367
|
interrupted: ['已打断', 'Interrupted'],
|
|
354
368
|
open: ['进行中', 'Open'],
|
|
@@ -57,7 +57,20 @@ function renderDebiasModeTag(modes, lang) {
|
|
|
57
57
|
function pluralizeEn(count, singular, plural = `${singular}s`) {
|
|
58
58
|
return `${count} ${count === 1 ? singular : plural}`;
|
|
59
59
|
}
|
|
60
|
-
function
|
|
60
|
+
function executorDisplayName(executor, lang) {
|
|
61
|
+
if (executor === 'dsh-host') {
|
|
62
|
+
return lang === 'zh' ? 'DeepSeek Harness(宿主模式)' : 'DeepSeek Harness (host mode)';
|
|
63
|
+
}
|
|
64
|
+
return executor || 'unknown';
|
|
65
|
+
}
|
|
66
|
+
function executorModelDisplayName(executor, model, lang) {
|
|
67
|
+
const executorName = executorDisplayName(executor, lang);
|
|
68
|
+
const modelName = model || 'unknown';
|
|
69
|
+
if (executor === 'dsh-host')
|
|
70
|
+
return `${executorName} · ${modelName}`;
|
|
71
|
+
return `${executorName}:${modelName}`;
|
|
72
|
+
}
|
|
73
|
+
function gatherRuntimeScopes(meta, lang) {
|
|
61
74
|
const scopes = [];
|
|
62
75
|
if (meta.executorRuntimes && Object.keys(meta.executorRuntimes).length > 0) {
|
|
63
76
|
Object.entries(meta.executorRuntimes)
|
|
@@ -75,14 +88,19 @@ function gatherRuntimeScopes(meta) {
|
|
|
75
88
|
.slice()
|
|
76
89
|
.sort((a, b) => `${a.executor}:${a.model}`.localeCompare(`${b.executor}:${b.model}`))
|
|
77
90
|
.forEach((entry) => {
|
|
78
|
-
if (entry.runtime)
|
|
79
|
-
scopes.push({
|
|
91
|
+
if (entry.runtime) {
|
|
92
|
+
scopes.push({
|
|
93
|
+
role: 'judge',
|
|
94
|
+
scope: executorModelDisplayName(entry.executor, entry.model, lang),
|
|
95
|
+
runtime: entry.runtime,
|
|
96
|
+
});
|
|
97
|
+
}
|
|
80
98
|
});
|
|
81
99
|
}
|
|
82
100
|
if (meta.diagnostic?.enabled && meta.diagnostic.runtime) {
|
|
83
101
|
scopes.push({
|
|
84
102
|
role: 'diagnostic',
|
|
85
|
-
scope:
|
|
103
|
+
scope: executorModelDisplayName(meta.diagnostic.executor, meta.diagnostic.model, lang),
|
|
86
104
|
runtime: meta.diagnostic.runtime,
|
|
87
105
|
});
|
|
88
106
|
}
|
|
@@ -105,6 +123,7 @@ function runtimeTooltip(runtime) {
|
|
|
105
123
|
`cost=${runtime.capabilities.costUSD}`,
|
|
106
124
|
`trace=${runtime.capabilities.trace}`,
|
|
107
125
|
`skillIsolation=${runtime.capabilities.skillIsolation}`,
|
|
126
|
+
...(runtime.auditability ? [`auditability=${runtime.auditability.status}`] : []),
|
|
108
127
|
...(runtime.binary?.contentHash
|
|
109
128
|
? [`contentHash=${runtime.binary.contentHash}`]
|
|
110
129
|
: []),
|
|
@@ -115,7 +134,7 @@ function runtimeTooltip(runtime) {
|
|
|
115
134
|
// fingerprint 重复 3 遍,扫读成本高。新版按 (fingerprint, versionText)
|
|
116
135
|
// 分组,scope 合并到 tag 内 "适用于 ..." 后缀。
|
|
117
136
|
function renderRuntimeFingerprintTags(meta, lang) {
|
|
118
|
-
const scopes = gatherRuntimeScopes(meta);
|
|
137
|
+
const scopes = gatherRuntimeScopes(meta, lang);
|
|
119
138
|
if (scopes.length === 0)
|
|
120
139
|
return '';
|
|
121
140
|
const groups = new Map();
|
|
@@ -476,11 +495,11 @@ export function renderRunDetail(report, lang = DEFAULT_LANG, skillContext) {
|
|
|
476
495
|
if (list.length === 0)
|
|
477
496
|
return `<span class="meta-tag">${t('judge', lang)}: —</span>`;
|
|
478
497
|
if (list.length === 1)
|
|
479
|
-
return `<span class="meta-tag">${t('judge', lang)}: ${e(
|
|
480
|
-
return `<span class="meta-tag" title="${t('ensembleDesc', lang)}">${t('judgeModelsLabel', lang)}: ${list.map((j) => e(
|
|
498
|
+
return `<span class="meta-tag">${t('judge', lang)}: ${e(executorModelDisplayName(list[0].executor, list[0].model, lang))}</span>`;
|
|
499
|
+
return `<span class="meta-tag" title="${t('ensembleDesc', lang)}">${t('judgeModelsLabel', lang)}: ${list.map((j) => e(executorModelDisplayName(j.executor, j.model, lang))).join(' · ')}</span>`;
|
|
481
500
|
})()}
|
|
482
501
|
${m.judgeRepeat && m.judgeRepeat > 1 ? `<span class="meta-tag" title="${t('judgeStddevDesc', lang)}">${t('judgeRepeatLabel', lang)}: ${m.judgeRepeat}</span>` : ''}
|
|
483
|
-
<span class="meta-tag">${t('executor', lang)}: ${e(m.executor
|
|
502
|
+
<span class="meta-tag">${t('executor', lang)}: ${e(executorDisplayName(m.executor, lang))}</span>
|
|
484
503
|
${m.effort ? `<span class="meta-tag" title="${e(lang === 'zh' ? 'executor LLM 的扩展思考预算(--effort)。跨 effort 报告不可严格比较' : 'reasoning effort for executor LLM (--effort); reports across different efforts are not strictly comparable')}">effort: ${e(m.effort)}</span>` : ''}
|
|
485
504
|
<span class="meta-tag"${execCostReported ? '' : ` title="${e(lang === 'zh' ? 'executor 不报 USD 成本(如 codex CLI),无法估算' : 'executor does not report USD cost (e.g. codex CLI); not measurable')}"`}>${t('cost', lang)}: ${fmtCost(totalExecCost, execCostReported)}</span>
|
|
486
505
|
<span class="meta-tag"${totalCostReported ? '' : ` title="${e(costCompletenessTooltip(lang))}"`}>${totalCostLabel}: ${fmtKnownCost(m.totalCostUSD, totalCostReported)}</span>${processCostTag ? `
|
|
@@ -639,10 +658,10 @@ export function renderBatchEvaluationDetail(report, lang = DEFAULT_LANG) {
|
|
|
639
658
|
if (list.length === 0)
|
|
640
659
|
return `<span class="meta-tag">${t('judge', lang)}: —</span>`;
|
|
641
660
|
if (list.length === 1)
|
|
642
|
-
return `<span class="meta-tag">${t('judge', lang)}: ${e(
|
|
643
|
-
return `<span class="meta-tag" title="${t('ensembleDesc', lang)}">${t('judgeModelsLabel', lang)}: ${list.map((j) => e(
|
|
661
|
+
return `<span class="meta-tag">${t('judge', lang)}: ${e(executorModelDisplayName(list[0].executor, list[0].model, lang))}</span>`;
|
|
662
|
+
return `<span class="meta-tag" title="${t('ensembleDesc', lang)}">${t('judgeModelsLabel', lang)}: ${list.map((j) => e(executorModelDisplayName(j.executor, j.model, lang))).join(' · ')}</span>`;
|
|
644
663
|
})()}
|
|
645
|
-
<span class="meta-tag">${t('executor', lang)}: ${e(m.executor
|
|
664
|
+
<span class="meta-tag">${t('executor', lang)}: ${e(executorDisplayName(m.executor, lang))}</span>
|
|
646
665
|
<span class="meta-tag"${totalCostReported ? '' : ` title="${e(costCompletenessTooltip(lang))}"`}>${t('totalCost', lang)}: ${fmtKnownCost(m.totalCostUSD, totalCostReported)}</span>
|
|
647
666
|
</div>
|
|
648
667
|
${(() => {
|
|
@@ -1101,6 +1101,7 @@ export function renderKnowledgeDebuggerPage(model, lang = DEFAULT_LANG, options
|
|
|
1101
1101
|
syncing: zh ? '同步中' : 'Syncing',
|
|
1102
1102
|
reconnecting: zh ? '重连中' : 'Reconnecting',
|
|
1103
1103
|
failed: zh ? '实时更新失败' : 'Live update failed',
|
|
1104
|
+
taskFailed: zh ? '任务已失败' : 'Task failed',
|
|
1104
1105
|
following: zh ? '跟随中' : 'Following',
|
|
1105
1106
|
resume: zh ? '跟随最新' : 'Follow latest',
|
|
1106
1107
|
pending: zh ? '查看更新' : 'View update',
|
|
@@ -1456,8 +1457,12 @@ function lifecycleEventLabel(label, lang) {
|
|
|
1456
1457
|
session_ended: { zh: '会话结束', en: 'Session ended' },
|
|
1457
1458
|
turn_started: { zh: '本轮开始', en: 'Turn started' },
|
|
1458
1459
|
turn_completed: { zh: '本轮完成', en: 'Turn completed' },
|
|
1460
|
+
turn_failed: { zh: '本轮失败', en: 'Turn failed' },
|
|
1459
1461
|
turn_aborted: { zh: '本轮中止', en: 'Turn aborted' },
|
|
1460
1462
|
turn_interrupted: { zh: '本轮被打断', en: 'Turn interrupted' },
|
|
1463
|
+
turn_ended_unknown: { zh: '本轮结束状态未知', en: 'Turn ended with unknown status' },
|
|
1464
|
+
step_started: { zh: '步骤开始', en: 'Step started' },
|
|
1465
|
+
step_completed: { zh: '步骤完成', en: 'Step completed' },
|
|
1461
1466
|
};
|
|
1462
1467
|
return labels[label ?? '']?.[lang] ?? (label || (lang === 'zh' ? '运行状态变化' : 'Lifecycle event'));
|
|
1463
1468
|
}
|
|
@@ -1466,7 +1471,7 @@ function lifecycleMilestoneTone(label) {
|
|
|
1466
1471
|
return 'start';
|
|
1467
1472
|
if (label === 'session_ended' || label === 'turn_completed')
|
|
1468
1473
|
return 'end';
|
|
1469
|
-
if (label === 'turn_aborted' || label === 'turn_interrupted')
|
|
1474
|
+
if (label === 'turn_failed' || label === 'turn_aborted' || label === 'turn_interrupted')
|
|
1470
1475
|
return 'warning';
|
|
1471
1476
|
return 'neutral';
|
|
1472
1477
|
}
|
|
@@ -188,8 +188,8 @@ export function renderObservationInboxPage(model, lang = DEFAULT_LANG) {
|
|
|
188
188
|
</div>`;
|
|
189
189
|
};
|
|
190
190
|
const renderSourceBadge = (item) => {
|
|
191
|
-
const label = item.sourceKind === 'openclaw' ? 'OpenClaw' : item.sourceKind === 'codex' ? 'Codex' : item.sourceKind === 'markdown_log' ? 'Markdown log' : item.sourceKind === 'claude' ? 'Claude' : 'Unknown';
|
|
192
|
-
const color = item.sourceKind === 'openclaw' ? '#7c3aed' : item.sourceKind === 'codex' ? '#1677ff' : item.sourceKind === 'markdown_log' ? 'var(--green)' : item.sourceKind === 'claude' ? 'var(--accent)' : 'var(--text-muted)';
|
|
191
|
+
const label = item.sourceKind === 'dsh' ? 'DeepSeek Harness' : item.sourceKind === 'openclaw' ? 'OpenClaw' : item.sourceKind === 'codex' ? 'Codex' : item.sourceKind === 'markdown_log' ? 'Markdown log' : item.sourceKind === 'claude' ? 'Claude' : 'Unknown';
|
|
192
|
+
const color = item.sourceKind === 'dsh' ? '#0f766e' : item.sourceKind === 'openclaw' ? '#7c3aed' : item.sourceKind === 'codex' ? '#1677ff' : item.sourceKind === 'markdown_log' ? 'var(--green)' : item.sourceKind === 'claude' ? 'var(--accent)' : 'var(--text-muted)';
|
|
193
193
|
return `<span title="调用日志来源:${e(label)}" style="display:inline-flex;margin-top:4px;padding:2px 6px;border-radius:999px;background:var(--bg-muted);color:${color};font-size:11px;font-weight:650">${e(label)}</span>`;
|
|
194
194
|
};
|
|
195
195
|
const confidenceHeaderHelp = '判断把握:OMK 对“这条 过程发现 是否需要处理/是否高风险/需关注”的规则判断有多确定。归属把握:OMK 把这条 过程发现 归到当前 skill 名下有多确定,例如明确调用 skill 通常更高。';
|
|
@@ -48,7 +48,9 @@ export function createTrajectoryLiveController(options) {
|
|
|
48
48
|
const state = terminalState ?? (!followLatest && pendingRevision
|
|
49
49
|
? 'pending'
|
|
50
50
|
: (followLatest ? 'following' : 'paused'));
|
|
51
|
-
const terminalLabel = terminalState
|
|
51
|
+
const terminalLabel = terminalState
|
|
52
|
+
? terminalState === 'failed' ? labels.taskFailed : labels[terminalState]
|
|
53
|
+
: undefined;
|
|
52
54
|
followButton.dataset.state = state;
|
|
53
55
|
followButton.dataset.following = String(followLatest);
|
|
54
56
|
followButton.setAttribute('aria-pressed', String(followLatest));
|
|
@@ -212,6 +214,7 @@ export function createTrajectoryLiveController(options) {
|
|
|
212
214
|
try {
|
|
213
215
|
const update = JSON.parse(event.data);
|
|
214
216
|
const explicitTerminal = update.status === 'completed'
|
|
217
|
+
|| update.status === 'failed'
|
|
215
218
|
|| update.status === 'aborted'
|
|
216
219
|
|| update.status === 'interrupted';
|
|
217
220
|
const streamTerminal = update.liveObservable === false || explicitTerminal;
|
|
@@ -226,7 +229,7 @@ export function createTrajectoryLiveController(options) {
|
|
|
226
229
|
if (!update.revision || update.revision === shell.dataset.liveRevision) {
|
|
227
230
|
const quiet = update.status === 'unknown';
|
|
228
231
|
const connectionLabel = terminalState
|
|
229
|
-
? labels[terminalState]
|
|
232
|
+
? terminalState === 'failed' ? labels.taskFailed : labels[terminalState]
|
|
230
233
|
: quiet ? labels.reconnecting : labels.live;
|
|
231
234
|
setConnectionState(terminalState ?? (quiet ? 'reconnecting' : 'live'), connectionLabel);
|
|
232
235
|
updateFollowControl();
|
package/dist/types/executor.d.ts
CHANGED
|
@@ -88,7 +88,7 @@ export interface ExecutorInput {
|
|
|
88
88
|
* - claude-cli:物化为临时 settings.json + on-disk hook 脚本,跑完清理
|
|
89
89
|
* - script(自定义脚本):同样物化临时 settings,通过 env(OMK_MOCK_SETTINGS_FILE /
|
|
90
90
|
* OMK_MOCK_MCP_CONFIG_FILE / OMK_MOCKS_FILE)暴露给脚本;脚本负责消费该协议
|
|
91
|
-
* - codex / codex-sdk /
|
|
91
|
+
* - codex / codex-sdk / *-api:不支持,executor capability gate 会拒绝,
|
|
92
92
|
* 绝不静默忽略后把 mock_hit 记成模型失败 */
|
|
93
93
|
mocks?: import('./eval.js').Mock[];
|
|
94
94
|
/** 解析 mock.return_file 的相对路径锚点(默认 sample 文件所在目录)。 */
|
|
@@ -119,7 +119,13 @@ export interface ExecutorInput {
|
|
|
119
119
|
*/
|
|
120
120
|
effort?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
|
|
121
121
|
}
|
|
122
|
-
export type
|
|
122
|
+
export type ExecutorRuntimeFingerprintResolver = (model: string, options?: {
|
|
123
|
+
skillDir?: string | null;
|
|
124
|
+
}) => ExecutorRuntimeFingerprint;
|
|
125
|
+
export type ExecutorFn = ((input: ExecutorInput) => Promise<ExecResult>) & {
|
|
126
|
+
/** Same-process hosts can report the runtime that actually owns execution. */
|
|
127
|
+
readonly runtimeFingerprint?: ExecutorRuntimeFingerprintResolver;
|
|
128
|
+
};
|
|
123
129
|
export interface ExecutorCache {
|
|
124
130
|
get(key: string): ExecResult | null;
|
|
125
131
|
set(key: string, value: ExecResult): void;
|
|
@@ -176,5 +182,10 @@ export interface ExecutorRuntimeFingerprint {
|
|
|
176
182
|
fingerprint: string;
|
|
177
183
|
binary?: ExecutorRuntimeBinary;
|
|
178
184
|
sdk?: ExecutorRuntimePackage;
|
|
185
|
+
/** Whether the recorded fields cover the complete effective runtime composition. */
|
|
186
|
+
auditability?: {
|
|
187
|
+
status: 'complete' | 'partial';
|
|
188
|
+
reasons?: string[];
|
|
189
|
+
};
|
|
179
190
|
capabilities: ExecutorRuntimeCapabilities;
|
|
180
191
|
}
|
package/dist/types/judge.d.ts
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import type { ExecutorRuntimeFingerprint } from './executor.js';
|
|
2
2
|
/** Single judge configuration: which executor to call and which model alias to pass. */
|
|
3
3
|
export interface JudgeConfig {
|
|
4
|
-
/** Executor name (claude /
|
|
4
|
+
/** Executor name (claude / codex / anthropic-api / openai-api / shell command). */
|
|
5
5
|
executor: string;
|
|
6
|
-
/** Model alias passed to the executor (e.g. "opus", "haiku", "gpt-4o"
|
|
6
|
+
/** Model alias passed to the executor (e.g. "opus", "haiku", "gpt-4o"). */
|
|
7
7
|
model: string;
|
|
8
8
|
}
|
|
9
9
|
/** Persisted judge entry on Report.meta.judgeModels: judge config + runtime fingerprint of
|
|
@@ -381,7 +381,7 @@ export type DebugKnowledgeAccessKind = 'injected' | 'read' | 'returned';
|
|
|
381
381
|
export type TaskReplayStepKind = 'user_request' | 'user_message' | 'user_correction' | 'runtime_context' | 'skill_context' | 'tool_exchange' | 'unmatched_tool_result' | 'assistant_message' | 'model_activity' | 'lifecycle' | 'observation' | 'system_event';
|
|
382
382
|
export type TaskReplayIntegrityCode = 'task_boundary_unavailable' | 'timeline_truncated' | 'malformed_records' | 'ignored_values' | 'unknown_events' | 'unmatched_tool_calls' | 'unmatched_tool_results' | 'missing_timestamps';
|
|
383
383
|
export type TaskWindowBasis = 'turn_id' | 'turn_lifecycle' | 'user_message' | 'unresolved';
|
|
384
|
-
export type ExperienceTurnStatus = 'completed' | 'aborted' | 'interrupted' | 'open' | 'unknown';
|
|
384
|
+
export type ExperienceTurnStatus = 'completed' | 'failed' | 'aborted' | 'interrupted' | 'open' | 'unknown';
|
|
385
385
|
/**
|
|
386
386
|
* One user-visible task inside a source thread. `turnId` is the stable,
|
|
387
387
|
* source-neutral identity used by Studio routes. `sourceTurnId` preserves the
|
package/dist/types/trace.d.ts
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
/** Stable source identity shared by execution evidence and Trace IR. */
|
|
2
|
-
export type TraceSourceKind = 'claude' | 'codex' | 'openclaw' | 'markdown_log' | 'unknown';
|
|
2
|
+
export type TraceSourceKind = 'claude' | 'codex' | 'dsh' | 'openclaw' | 'markdown_log' | 'unknown';
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "oh-my-knowledge",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.54.0",
|
|
4
4
|
"packageManager": "yarn@4.16.0",
|
|
5
5
|
"description": "OMK — Observe. Measure. Know. Evidence-backed knowledge changes for AI applications.",
|
|
6
6
|
"type": "module",
|
|
@@ -13,6 +13,11 @@
|
|
|
13
13
|
"files": [
|
|
14
14
|
"dist/"
|
|
15
15
|
],
|
|
16
|
+
"dsh": {
|
|
17
|
+
"bundle": {
|
|
18
|
+
"patch": "./dist/dsh-plugin/cordis.patch.yml"
|
|
19
|
+
}
|
|
20
|
+
},
|
|
16
21
|
"scripts": {
|
|
17
22
|
"clean": "rm -rf dist dist-scripts node_modules/.cache/omk",
|
|
18
23
|
"build": "run-s build:src build:scripts build:assets",
|
|
@@ -28,6 +33,7 @@
|
|
|
28
33
|
"lint": "eslint 'src/**/*.ts' 'test/**/*.ts' --cache --cache-location node_modules/.cache/eslint/ --max-warnings 0",
|
|
29
34
|
"lint-staged": "lint-staged",
|
|
30
35
|
"test": "vitest run",
|
|
36
|
+
"test:profile": "yarn build && node dist-scripts/test-profile.js",
|
|
31
37
|
"ci": "run-s lint typecheck build build:docs:check test",
|
|
32
38
|
"prepublishOnly": "run-s clean build",
|
|
33
39
|
"prepare": "husky"
|
|
@@ -79,6 +85,10 @@
|
|
|
79
85
|
"release-evidence",
|
|
80
86
|
"llm-observability",
|
|
81
87
|
"prompt-engineering",
|
|
88
|
+
"deepseek-harness",
|
|
89
|
+
"dsh-plugin",
|
|
90
|
+
"dsh",
|
|
91
|
+
"cordis",
|
|
82
92
|
"codex",
|
|
83
93
|
"codex-cli",
|
|
84
94
|
"openai",
|
|
@@ -98,12 +108,10 @@
|
|
|
98
108
|
"author": "lizhiyao",
|
|
99
109
|
"license": "MIT",
|
|
100
110
|
"dependencies": {
|
|
101
|
-
"@anthropic-ai/
|
|
102
|
-
"@anthropic-ai/sdk": "^0.117.1",
|
|
111
|
+
"@anthropic-ai/sdk": "^0.120.0",
|
|
103
112
|
"@inquirer/prompts": "^8.4.3",
|
|
104
113
|
"@modelcontextprotocol/sdk": "^1.30.0",
|
|
105
114
|
"@oclif/core": "^4",
|
|
106
|
-
"@openai/codex-sdk": "0.147.0",
|
|
107
115
|
"ajv": "^8.18.0",
|
|
108
116
|
"chart.js": "^4.5.1",
|
|
109
117
|
"es-module-lexer": "^2.0.0",
|
|
@@ -111,6 +119,18 @@
|
|
|
111
119
|
"simple-statistics": "^7.8.9",
|
|
112
120
|
"zod": "^4.4.3"
|
|
113
121
|
},
|
|
122
|
+
"peerDependencies": {
|
|
123
|
+
"@anthropic-ai/claude-agent-sdk": "^0.3.143",
|
|
124
|
+
"@openai/codex-sdk": "^0.149.0"
|
|
125
|
+
},
|
|
126
|
+
"peerDependenciesMeta": {
|
|
127
|
+
"@anthropic-ai/claude-agent-sdk": {
|
|
128
|
+
"optional": true
|
|
129
|
+
},
|
|
130
|
+
"@openai/codex-sdk": {
|
|
131
|
+
"optional": true
|
|
132
|
+
}
|
|
133
|
+
},
|
|
114
134
|
"publishConfig": {
|
|
115
135
|
"registry": "https://registry.npmjs.org"
|
|
116
136
|
},
|
|
@@ -124,6 +144,6 @@
|
|
|
124
144
|
"typescript": "^6.0.2",
|
|
125
145
|
"typescript-eslint": "^8.58.0",
|
|
126
146
|
"vitepress": "^1.6.4",
|
|
127
|
-
"vitest": "4.1.
|
|
147
|
+
"vitest": "4.1.11"
|
|
128
148
|
}
|
|
129
149
|
}
|
|
@@ -1,28 +0,0 @@
|
|
|
1
|
-
import type { ExecResult } from '../types/index.js';
|
|
2
|
-
import type { ClaudeSdkBaseMessage, ClaudeSdkResultMessage } from './shared.js';
|
|
3
|
-
export interface ClaudeSdkMeasurements {
|
|
4
|
-
durationMs: number;
|
|
5
|
-
durationApiMs: number;
|
|
6
|
-
inputTokens: number;
|
|
7
|
-
outputTokens: number;
|
|
8
|
-
cacheReadTokens: number;
|
|
9
|
-
cacheCreationTokens: number;
|
|
10
|
-
costUSD: number;
|
|
11
|
-
numTurns: number;
|
|
12
|
-
}
|
|
13
|
-
export declare function normalizeClaudeSdkMeasurements(result: ClaudeSdkResultMessage): ClaudeSdkMeasurements | {
|
|
14
|
-
error: string;
|
|
15
|
-
};
|
|
16
|
-
export interface ClaudeStreamParseResult {
|
|
17
|
-
messages: ClaudeSdkBaseMessage[];
|
|
18
|
-
malformedLineCount: number;
|
|
19
|
-
}
|
|
20
|
-
export declare function parseClaudeStreamJson(stdout: string): ClaudeStreamParseResult;
|
|
21
|
-
export declare function buildClaudeResult(options: {
|
|
22
|
-
messages: ClaudeSdkBaseMessage[];
|
|
23
|
-
wallClockDurationMs: number;
|
|
24
|
-
source: 'claude stream-json' | 'claude-sdk';
|
|
25
|
-
malformedLineCount?: number;
|
|
26
|
-
forcedError?: string;
|
|
27
|
-
messageTimestamps?: number[];
|
|
28
|
-
}): ExecResult;
|
|
@@ -1,9 +0,0 @@
|
|
|
1
|
-
import type { ToolCallInfo, TurnInfo } from '../types/index.js';
|
|
2
|
-
import type { ClaudeSdkBaseMessage } from './shared.js';
|
|
3
|
-
export declare function isClaudeSdkResultMessage(message: ClaudeSdkBaseMessage): boolean;
|
|
4
|
-
export declare function extractAgentTrace(messages: ClaudeSdkBaseMessage[], timestamps?: number[]): {
|
|
5
|
-
turns: TurnInfo[];
|
|
6
|
-
toolCalls: ToolCallInfo[];
|
|
7
|
-
fullNumTurns: number;
|
|
8
|
-
numSubAgents: number;
|
|
9
|
-
};
|
|
@@ -1,24 +0,0 @@
|
|
|
1
|
-
import type { ExecResult } from '../types/index.js';
|
|
2
|
-
import type { CodexEvent } from './shared.js';
|
|
3
|
-
export declare function extractCodexUsage(events: CodexEvent[]): {
|
|
4
|
-
input: number;
|
|
5
|
-
cached: number;
|
|
6
|
-
output: number;
|
|
7
|
-
};
|
|
8
|
-
/**
|
|
9
|
-
* Match @openai/codex-sdk's `finalResponse`: the latest completed
|
|
10
|
-
* agent_message is the answer. Earlier messages remain available in `turns`.
|
|
11
|
-
*/
|
|
12
|
-
export declare function extractCodexFinalOutput(events: CodexEvent[]): string;
|
|
13
|
-
export declare function extractCodexProtocolError(events: CodexEvent[]): string | undefined;
|
|
14
|
-
export declare function validateCodexProtocol(events: CodexEvent[]): string | undefined;
|
|
15
|
-
export declare function extractCodexStopReason(events: CodexEvent[]): string;
|
|
16
|
-
export declare function sumCodexElapsed(resultEvents: CodexEvent[], wallClock: number): number;
|
|
17
|
-
export interface BuildCodexResultOptions {
|
|
18
|
-
events: CodexEvent[];
|
|
19
|
-
wallClockDurationMs: number;
|
|
20
|
-
source: 'codex --json' | 'codex-sdk';
|
|
21
|
-
malformedLineCount?: number;
|
|
22
|
-
forcedError?: string;
|
|
23
|
-
}
|
|
24
|
-
export declare function buildCodexResult({ events, wallClockDurationMs, source, malformedLineCount, forcedError, }: BuildCodexResultOptions): ExecResult;
|