oh-my-knowledge 0.52.3 → 0.54.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/README.md +21 -1
  2. package/README.zh.md +21 -1
  3. package/dist/assets/agent-skills/omk/SKILL.md +6 -0
  4. package/dist/assets/agent-skills/omk/references/commands.md +1 -1
  5. package/dist/authoring/evolver.js +1 -1
  6. package/dist/authoring/generator.js +1 -1
  7. package/dist/cli/commands/eval/index.js +7 -6
  8. package/dist/cli/commands/install.js +5 -1
  9. package/dist/cli/lib/generation-failure-hint.js +6 -8
  10. package/dist/cli/lib/runtime-defaults.d.ts +2 -0
  11. package/dist/cli/lib/runtime-defaults.js +7 -4
  12. package/dist/dsh-plugin/cordis.patch.yml +3 -0
  13. package/dist/dsh-plugin/host-executor.d.ts +93 -0
  14. package/dist/dsh-plugin/host-executor.js +232 -0
  15. package/dist/dsh-plugin/index.d.ts +30 -0
  16. package/dist/dsh-plugin/index.js +275 -0
  17. package/dist/dsh-plugin/observe.d.ts +47 -0
  18. package/dist/dsh-plugin/observe.js +359 -0
  19. package/dist/dsh-plugin/protocol.d.ts +21 -0
  20. package/dist/dsh-plugin/protocol.js +229 -0
  21. package/dist/dsh-plugin/trace-adapter.d.ts +48 -0
  22. package/dist/dsh-plugin/trace-adapter.js +590 -0
  23. package/dist/eval-core/comparability.js +3 -0
  24. package/dist/eval-core/evaluation-execution.js +5 -13
  25. package/dist/eval-core/evaluation-reporting.d.ts +7 -4
  26. package/dist/eval-core/evaluation-reporting.js +39 -18
  27. package/dist/eval-core/judge-independence.d.ts +1 -1
  28. package/dist/eval-core/judge-independence.js +1 -1
  29. package/dist/eval-core/report-document.js +6 -0
  30. package/dist/eval-core/resume-compatibility.d.ts +1 -0
  31. package/dist/eval-core/resume-compatibility.js +5 -4
  32. package/dist/eval-workflows/batch-evaluation-workflow.js +1 -1
  33. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +4 -2
  34. package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
  35. package/dist/eval-workflows/evaluation-pipeline.js +7 -3
  36. package/dist/eval-workflows/run-evaluation.d.ts +4 -2
  37. package/dist/eval-workflows/run-evaluation.js +7 -3
  38. package/dist/executors/{anthropic-api.d.ts → anthropic/api.d.ts} +1 -1
  39. package/dist/executors/{anthropic-api.js → anthropic/api.js} +4 -2
  40. package/dist/executors/{claude-cli.d.ts → anthropic/claude/cli.d.ts} +1 -1
  41. package/dist/executors/{claude-cli.js → anthropic/claude/cli.js} +5 -3
  42. package/dist/executors/anthropic/claude/protocol.d.ts +87 -0
  43. package/dist/executors/{claude-protocol.js → anthropic/claude/protocol.js} +6 -6
  44. package/dist/executors/{claude-sdk.d.ts → anthropic/claude/sdk.d.ts} +2 -2
  45. package/dist/executors/{claude-sdk.js → anthropic/claude/sdk.js} +8 -5
  46. package/dist/executors/anthropic/claude/trace.d.ts +9 -0
  47. package/dist/executors/{claude-sdk-trace.js → anthropic/claude/trace.js} +5 -5
  48. package/dist/executors/{capabilities.d.ts → core/capabilities.d.ts} +3 -5
  49. package/dist/executors/{capabilities.js → core/capabilities.js} +4 -11
  50. package/dist/executors/core/http.d.ts +6 -0
  51. package/dist/executors/core/http.js +19 -0
  52. package/dist/executors/core/limits.d.ts +2 -0
  53. package/dist/executors/core/limits.js +2 -0
  54. package/dist/executors/core/optional-dependencies.d.ts +7 -0
  55. package/dist/executors/core/optional-dependencies.js +35 -0
  56. package/dist/executors/core/registry.d.ts +145 -0
  57. package/dist/executors/core/registry.js +127 -0
  58. package/dist/executors/core/runtime-fingerprint.d.ts +13 -0
  59. package/dist/executors/{runtime-fingerprint.js → core/runtime-fingerprint.js} +119 -59
  60. package/dist/executors/core/runtime.d.ts +12 -0
  61. package/dist/executors/core/runtime.js +61 -0
  62. package/dist/executors/core/subprocess.d.ts +44 -0
  63. package/dist/executors/{shared.js → core/subprocess.js} +20 -156
  64. package/dist/executors/index.d.ts +4 -4
  65. package/dist/executors/index.js +26 -15
  66. package/dist/executors/{openai-api.d.ts → openai/api.d.ts} +1 -1
  67. package/dist/executors/{openai-api.js → openai/api.js} +4 -2
  68. package/dist/executors/{codex-cli.d.ts → openai/codex/cli.d.ts} +3 -3
  69. package/dist/executors/{codex-cli.js → openai/codex/cli.js} +5 -3
  70. package/dist/executors/openai/codex/protocol.d.ts +72 -0
  71. package/dist/executors/{codex-protocol.js → openai/codex/protocol.js} +33 -2
  72. package/dist/executors/{codex-sdk.d.ts → openai/codex/sdk.d.ts} +18 -4
  73. package/dist/executors/{codex-sdk.js → openai/codex/sdk.js} +7 -4
  74. package/dist/executors/{codex-cli-trace.d.ts → openai/codex/trace.d.ts} +2 -2
  75. package/dist/executors/{codex-cli-trace.js → openai/codex/trace.js} +4 -4
  76. package/dist/executors/{script.d.ts → script/index.d.ts} +1 -1
  77. package/dist/executors/{script.js → script/index.js} +6 -4
  78. package/dist/grading/judge.d.ts +1 -1
  79. package/dist/grading/judge.js +1 -1
  80. package/dist/observability/conversation-catalog.js +1 -1
  81. package/dist/observability/experience.js +4 -0
  82. package/dist/observability/inbox.d.ts +4 -1
  83. package/dist/observability/inbox.js +7 -2
  84. package/dist/observability/trace-ir.d.ts +6 -3
  85. package/dist/observability/turn-index.js +4 -0
  86. package/dist/renderer/conversation-renderer.js +20 -6
  87. package/dist/renderer/html-renderer.js +30 -11
  88. package/dist/renderer/knowledge-debugger-renderer.js +6 -1
  89. package/dist/renderer/observation-inbox-renderer.js +2 -2
  90. package/dist/renderer/trajectory-live.d.ts +1 -0
  91. package/dist/renderer/trajectory-live.js +5 -2
  92. package/dist/shared/trace-source-kind.js +1 -0
  93. package/dist/types/executor.d.ts +13 -2
  94. package/dist/types/judge.d.ts +2 -2
  95. package/dist/types/observability.d.ts +1 -1
  96. package/dist/types/trace.d.ts +1 -1
  97. package/package.json +25 -5
  98. package/dist/executors/claude-protocol.d.ts +0 -28
  99. package/dist/executors/claude-sdk-trace.d.ts +0 -9
  100. package/dist/executors/codex-protocol.d.ts +0 -24
  101. package/dist/executors/gemini.d.ts +0 -2
  102. package/dist/executors/gemini.js +0 -156
  103. package/dist/executors/runtime-fingerprint.d.ts +0 -6
  104. package/dist/executors/shared.d.ts +0 -226
  105. /package/dist/executors/{script-command.d.ts → script/command.d.ts} +0 -0
  106. /package/dist/executors/{script-command.js → script/command.js} +0 -0
@@ -1,7 +1,9 @@
1
- import { materializeForCliConfigDir } from '../eval-core/mocks-runtime.js';
2
- import { executorResultValidationError, normalizeExecResultToolIdentities, } from '../shared/executor-result.js';
3
- import { resolveScriptCommand } from './script-command.js';
4
- import { DEFAULT_TIMEOUT_MS, interruptedExecResult, spawnWithSigintPropagation, timeoutExecResult, } from './shared.js';
1
+ import { materializeForCliConfigDir } from '../../eval-core/mocks-runtime.js';
2
+ import { executorResultValidationError, normalizeExecResultToolIdentities, } from '../../shared/executor-result.js';
3
+ import { resolveScriptCommand } from './command.js';
4
+ import { DEFAULT_TIMEOUT_MS } from '../core/limits.js';
5
+ import { interruptedExecResult, timeoutExecResult, } from '../core/runtime.js';
6
+ import { spawnWithSigintPropagation } from '../core/subprocess.js';
5
7
  // script executor 由用户自定义,omk 无法保证它实现 skill 隔离。
6
8
  // 任何 allowedSkills(包括 [])下都 stderr 一次性 warn,不阻塞执行,
7
9
  // 让用户知道 strict-baseline / 显式 allowedSkills 在 script executor 下静默无效。
@@ -53,7 +53,7 @@ export declare function judgeId(config: JudgeConfig): string;
53
53
  export declare function computeJudgeAgreement(judgeScores: number[][]): JudgeAgreement;
54
54
  /**
55
55
  * Judge a single (output, rubric) pair with N judge models in parallel. Each judge
56
- * may use a different executor (e.g. claude:opus + openai-api:gpt-4o + gemini:pro). Each
56
+ * may use a different executor (e.g. claude:opus + openai-api:gpt-4o). Each
57
57
  * judge can also be repeated `judgeRepeat` times — final per-judge score is its mean.
58
58
  *
59
59
  * Returns: aggregate DimensionResult (score = mean across judges; this is the "consensus"
@@ -342,7 +342,7 @@ export function computeJudgeAgreement(judgeScores) {
342
342
  }
343
343
  /**
344
344
  * Judge a single (output, rubric) pair with N judge models in parallel. Each judge
345
- * may use a different executor (e.g. claude:opus + openai-api:gpt-4o + gemini:pro). Each
345
+ * may use a different executor (e.g. claude:opus + openai-api:gpt-4o). Each
346
346
  * judge can also be repeated `judgeRepeat` times — final per-judge score is its mean.
347
347
  *
348
348
  * Returns: aggregate DimensionResult (score = mean across judges; this is the "consensus"
@@ -542,7 +542,7 @@ function trajectoryRevisionFromStat(sourceStat, status) {
542
542
  return `${sourceStat.size}:${sourceStat.mtimeMs}:${status}`;
543
543
  }
544
544
  function isTerminalTaskStatus(status) {
545
- return status === 'completed' || status === 'aborted' || status === 'interrupted';
545
+ return status === 'completed' || status === 'failed' || status === 'aborted' || status === 'interrupted';
546
546
  }
547
547
  function secondsToMs(value) {
548
548
  const seconds = numberValue(value);
@@ -1012,6 +1012,7 @@ function isExperienceTurnSummaryArray(value) {
1012
1012
  && isOptionalTimestamp(turn.startTimestamp)
1013
1013
  && isOptionalTimestamp(turn.endTimestamp)
1014
1014
  && (turn.status === 'completed'
1015
+ || turn.status === 'failed'
1015
1016
  || turn.status === 'aborted'
1016
1017
  || turn.status === 'interrupted'
1017
1018
  || turn.status === 'open'
@@ -2621,6 +2622,9 @@ function timelineEventsFromTraceEvent(session, event, eventIndex) {
2621
2622
  memoryMode: event.memoryMode,
2622
2623
  historyMode: event.historyMode,
2623
2624
  contextWindowId: event.contextWindowId,
2625
+ parentRunId: event.parentRunId,
2626
+ delegationDepth: event.delegationDepth,
2627
+ sourceOrigin: event.sourceOrigin,
2624
2628
  availableTools: event.availableTools,
2625
2629
  instructions: event.instructions,
2626
2630
  goal: event.goal,
@@ -1,4 +1,5 @@
1
- import type { BuildObservationInboxReportOptions, ObservationEvidence, ObservationInboxItem, ObservationInboxReport, ObservationMessageRef, ObservationMessageWindow, ObservationSessionTimeRange, ObservationSeverityReasonCode, ObservationSignalSubtype, ObservationSignalType, ObservationSkillRollup, ObservationSourceKind } from '../types/index.js';
1
+ import type { BuildObservationInboxReportOptions, ObservationEvidence, ObservationInboxItem, ObservationInboxReport, ObservationMessageRef, ObservationMessageWindow, ObservationSessionTimeRange, ObservationSeverityReasonCode, ObservationSignalSubtype, ObservationSignalType, ObservationSkillRollup, ObservationSourceKind, TraceIngestionSummary } from '../types/index.js';
2
+ import { type TraceSession } from './trace-adapter.js';
2
3
  import { type PersistedObservationExperienceReport } from './experience.js';
3
4
  export type { BuildObservationInboxReportOptions, ObservationEvidence, ObservationInboxItem, ObservationInboxReport, ObservationMessageRef, ObservationMessageWindow, ObservationSessionTimeRange, ObservationSeverityReasonCode, ObservationSignalSubtype, ObservationSignalType, ObservationSkillRollup, ObservationSourceKind, };
4
5
  export type PersistedObservationInboxReport = Omit<ObservationInboxReport, 'experience'> & {
@@ -19,6 +20,8 @@ export declare function severityReasonCodeFor(item: Pick<ObservationInboxItem, '
19
20
  * `buildObserveDiagnosticsFromReport(report)` 写入 `report.diagnostics`。
20
21
  */
21
22
  export declare function buildObservationInboxReport(tracePath: string, options?: BuildObservationInboxReportOptions): ObservationInboxReport;
23
+ /** Build the existing inbox artifact from an in-memory source-neutral corpus. */
24
+ export declare function buildObservationInboxReportFromTraceSessions(tracePath: string, sessions: TraceSession[], ingestion: TraceIngestionSummary, options?: BuildObservationInboxReportOptions): ObservationInboxReport;
22
25
  export declare function aggregateInboxItems(items: ObservationInboxItem[]): ObservationInboxItem[];
23
26
  export declare function saveObservationInboxReport(report: ObservationInboxReport, outDir?: string): string;
24
27
  export declare function compactObservationInboxReport(report: ObservationInboxReport): PersistedObservationInboxReport;
@@ -5,7 +5,7 @@ import { OMK_HOME } from '../eval-core/default-dirs.js';
5
5
  import { isReportFileName, randomRunToken, reportFilePath } from '../eval-core/artifact-file-names.js';
6
6
  import { migrateLegacyReportFiles } from '../eval-core/report-file-migration.js';
7
7
  import { extractGapSignalsFromTrace } from '../analysis/gap-analyzer.js';
8
- import { loadTraceSessions, tracesToResultEntries, skillSegmentTimestampObserved, } from './trace-adapter.js';
8
+ import { loadTraceSessions, segmentTraceBySkill, tracesToResultEntries, skillSegmentTimestampObserved, } from './trace-adapter.js';
9
9
  import { normalizeTraceTimestamp } from './trace-ir.js';
10
10
  import { isSearchToolCall, toolCallQuery } from '../shared/tool-search.js';
11
11
  import { isToolCallFailure, isToolCallSuccess } from '../shared/tool-call-status.js';
@@ -410,7 +410,12 @@ function skillSessionCountKey(segment) {
410
410
  * `buildObserveDiagnosticsFromReport(report)` 写入 `report.diagnostics`。
411
411
  */
412
412
  export function buildObservationInboxReport(tracePath, options = {}) {
413
- const { sessions, segments, ingestion } = tracesToResultEntries(tracePath);
413
+ const { sessions, ingestion } = tracesToResultEntries(tracePath);
414
+ return buildObservationInboxReportFromTraceSessions(tracePath, sessions, ingestion, options);
415
+ }
416
+ /** Build the existing inbox artifact from an in-memory source-neutral corpus. */
417
+ export function buildObservationInboxReportFromTraceSessions(tracePath, sessions, ingestion, options = {}) {
418
+ const segments = sessions.flatMap(segmentTraceBySkill);
414
419
  const skillSegments = segments.filter((segment) => segment.skillName !== 'general');
415
420
  const generatedAt = new Date().toISOString();
416
421
  const sessionTimeRanges = buildSessionTimeRanges(sessions);
@@ -60,8 +60,8 @@ export interface TraceUsageEvent extends TraceEventBase {
60
60
  model?: string;
61
61
  inputTokens: number;
62
62
  outputTokens: number;
63
- cacheReadTokens: number;
64
- cacheCreationTokens: number;
63
+ cacheReadTokens?: number;
64
+ cacheCreationTokens?: number;
65
65
  reasoningTokens?: number;
66
66
  }
67
67
  export interface TraceModelActivityEvent extends TraceEventBase {
@@ -74,7 +74,7 @@ export interface TraceModelActivityEvent extends TraceEventBase {
74
74
  }
75
75
  export interface TraceLifecycleEvent extends TraceEventBase {
76
76
  eventKind: 'lifecycle';
77
- phase: 'session_started' | 'session_ended' | 'turn_started' | 'turn_completed' | 'turn_aborted' | 'turn_interrupted';
77
+ phase: 'session_started' | 'session_ended' | 'turn_started' | 'turn_completed' | 'turn_failed' | 'turn_aborted' | 'turn_interrupted' | 'turn_ended_unknown' | 'step_started' | 'step_completed';
78
78
  reason?: string;
79
79
  durationMs?: number;
80
80
  }
@@ -105,6 +105,9 @@ export interface TraceRuntimeContextEvent extends TraceEventBase {
105
105
  memoryMode?: string;
106
106
  historyMode?: string;
107
107
  contextWindowId?: string;
108
+ parentRunId?: string;
109
+ delegationDepth?: number;
110
+ sourceOrigin?: string;
108
111
  availableTools?: string[];
109
112
  instructions?: string;
110
113
  goal?: string;
@@ -1,8 +1,10 @@
1
1
  import { createHash } from 'node:crypto';
2
2
  const TERMINAL_LIFECYCLE_LABELS = new Set([
3
3
  'turn_completed',
4
+ 'turn_failed',
4
5
  'turn_aborted',
5
6
  'turn_interrupted',
7
+ 'turn_ended_unknown',
6
8
  ]);
7
9
  /** Reconstruct every observable task without consulting Skill attribution. */
8
10
  export function reconstructExperienceTurns(timeline) {
@@ -124,6 +126,8 @@ function turnStatus(events) {
124
126
  const labels = new Set(events.map((event) => event.label));
125
127
  if (labels.has('turn_completed'))
126
128
  return 'completed';
129
+ if (labels.has('turn_failed'))
130
+ return 'failed';
127
131
  if (labels.has('turn_aborted'))
128
132
  return 'aborted';
129
133
  if (labels.has('turn_interrupted'))
@@ -54,8 +54,8 @@ export function renderConversationIndexPage(model, lang = DEFAULT_LANG) {
54
54
  const rows = conversations.map((conversation) => renderConversationRow(conversation, lang)).join('');
55
55
  const content = model.conversations.length > 0
56
56
  ? `<div class="conversation-list" role="list">${rows}</div>`
57
- : `<div class="empty-state"><strong>${zh ? '还没有可浏览的 Codex 对话' : 'No Codex conversations yet'}</strong><span>${zh ? 'Studio 会直接读取 Codex 的本机会话索引,不需要先运行 observe。' : 'Studio reads the local Codex conversation index directly; observe is not required.'}</span></div>`;
58
- return layout(zh ? 'Codex 对话' : 'Codex conversations', `
57
+ : `<div class="empty-state"><strong>${zh ? '还没有可浏览的对话' : 'No conversations yet'}</strong><span>${zh ? '从受支持的 runtime 摄取任务轨迹后,会话会显示在这里。' : 'Sessions appear here after a supported runtime supplies task trajectories.'}</span></div>`;
58
+ return layout(zh ? '对话' : 'Conversations', `
59
59
  <main class="conversation-page conversation-index-app" data-activity-revision="${e(activity.revision)}">
60
60
  <header class="conversation-app-head">
61
61
  <a class="conversation-app-brand" href="/${langQuery}">
@@ -67,7 +67,7 @@ export function renderConversationIndexPage(model, lang = DEFAULT_LANG) {
67
67
  </nav>
68
68
  </header>
69
69
  <div class="conversation-app-body">
70
- <section class="conversation-browser" aria-label="${zh ? 'Codex 对话' : 'Codex conversations'}">
70
+ <section class="conversation-browser" aria-label="${zh ? '对话' : 'Conversations'}">
71
71
  <header class="conversation-toolbar">
72
72
  <div class="conversation-browser-title">
73
73
  <h1>${zh ? '对话' : 'Conversations'}</h1>
@@ -117,7 +117,7 @@ export function renderConversationDetailPage(conversation, lang = DEFAULT_LANG)
117
117
  <header class="conversation-page-head conversation-detail-head">
118
118
  <div>
119
119
  <a class="back-link" href="/conversations${langSuffix}">${zh ? '返回对话总览' : 'Back to conversations'}</a>
120
- <p class="conversation-eyebrow">CODEX · ${e(shortThreadId(conversation.sourceThreadId))}</p>
120
+ <p class="conversation-eyebrow">${e(sourceLabel(conversation))} · ${e(shortThreadId(conversation.sourceThreadId))}</p>
121
121
  <h1>${renderSafeInlineMarkdown(conversation.title)}</h1>
122
122
  <div class="detail-meta">
123
123
  ${conversation.model ? `<span>${e(conversation.model)}</span>` : ''}
@@ -157,8 +157,8 @@ function renderConversationRow(conversation, lang) {
157
157
  ? '<span class="index-pending">—</span>'
158
158
  : `<span><b>${conversation.turnCount}</b>${zh ? '任务' : 'tasks'}</span>`;
159
159
  const activity = openTask
160
- ? `<strong class="running-label"><i aria-hidden="true"></i>${zh ? '进行中' : 'Running'}</strong><span>${e(updated)} · ${e(conversation.model ?? 'Codex')}</span>`
161
- : `<strong>${e(updated)}</strong><span>${e(conversation.model ?? 'Codex')}</span>`;
160
+ ? `<strong class="running-label"><i aria-hidden="true"></i>${zh ? '进行中' : 'Running'}</strong><span>${e(updated)} · ${e(conversation.model ?? sourceLabel(conversation))}</span>`
161
+ : `<strong>${e(updated)}</strong><span>${e(conversation.model ?? sourceLabel(conversation))}</span>`;
162
162
  const liveTaskLink = openTaskHref
163
163
  ? `<a class="live-task-link" href="${e(openTaskHref)}" aria-label="${zh ? '查看进行中任务的实时轨迹' : 'View the live trajectory for the running task'}"><i aria-hidden="true"></i>${zh ? '查看实时轨迹' : 'View live'}</a>`
164
164
  : '';
@@ -345,10 +345,24 @@ function compactPath(value) {
345
345
  function shortThreadId(value) {
346
346
  return value.length > 18 ? `${value.slice(0, 8)}…${value.slice(-6)}` : value;
347
347
  }
348
+ function sourceLabel(conversation) {
349
+ if (conversation.sourceKind === 'dsh')
350
+ return 'DeepSeek Harness';
351
+ if (conversation.sourceKind === 'codex')
352
+ return 'Codex';
353
+ if (conversation.sourceKind === 'claude')
354
+ return 'Claude';
355
+ if (conversation.sourceKind === 'openclaw')
356
+ return 'OpenClaw';
357
+ if (conversation.sourceKind === 'markdown_log')
358
+ return 'Markdown';
359
+ return 'Runtime';
360
+ }
348
361
  function statusLabel(status, lang) {
349
362
  const zh = lang === 'zh';
350
363
  const labels = {
351
364
  completed: ['已完成', 'Completed'],
365
+ failed: ['已失败', 'Failed'],
352
366
  aborted: ['已中止', 'Aborted'],
353
367
  interrupted: ['已打断', 'Interrupted'],
354
368
  open: ['进行中', 'Open'],
@@ -57,7 +57,20 @@ function renderDebiasModeTag(modes, lang) {
57
57
  function pluralizeEn(count, singular, plural = `${singular}s`) {
58
58
  return `${count} ${count === 1 ? singular : plural}`;
59
59
  }
60
- function gatherRuntimeScopes(meta) {
60
+ function executorDisplayName(executor, lang) {
61
+ if (executor === 'dsh-host') {
62
+ return lang === 'zh' ? 'DeepSeek Harness(宿主模式)' : 'DeepSeek Harness (host mode)';
63
+ }
64
+ return executor || 'unknown';
65
+ }
66
+ function executorModelDisplayName(executor, model, lang) {
67
+ const executorName = executorDisplayName(executor, lang);
68
+ const modelName = model || 'unknown';
69
+ if (executor === 'dsh-host')
70
+ return `${executorName} · ${modelName}`;
71
+ return `${executorName}:${modelName}`;
72
+ }
73
+ function gatherRuntimeScopes(meta, lang) {
61
74
  const scopes = [];
62
75
  if (meta.executorRuntimes && Object.keys(meta.executorRuntimes).length > 0) {
63
76
  Object.entries(meta.executorRuntimes)
@@ -75,14 +88,19 @@ function gatherRuntimeScopes(meta) {
75
88
  .slice()
76
89
  .sort((a, b) => `${a.executor}:${a.model}`.localeCompare(`${b.executor}:${b.model}`))
77
90
  .forEach((entry) => {
78
- if (entry.runtime)
79
- scopes.push({ role: 'judge', scope: `${entry.executor}:${entry.model}`, runtime: entry.runtime });
91
+ if (entry.runtime) {
92
+ scopes.push({
93
+ role: 'judge',
94
+ scope: executorModelDisplayName(entry.executor, entry.model, lang),
95
+ runtime: entry.runtime,
96
+ });
97
+ }
80
98
  });
81
99
  }
82
100
  if (meta.diagnostic?.enabled && meta.diagnostic.runtime) {
83
101
  scopes.push({
84
102
  role: 'diagnostic',
85
- scope: `${meta.diagnostic.executor}:${meta.diagnostic.model}`,
103
+ scope: executorModelDisplayName(meta.diagnostic.executor, meta.diagnostic.model, lang),
86
104
  runtime: meta.diagnostic.runtime,
87
105
  });
88
106
  }
@@ -105,6 +123,7 @@ function runtimeTooltip(runtime) {
105
123
  `cost=${runtime.capabilities.costUSD}`,
106
124
  `trace=${runtime.capabilities.trace}`,
107
125
  `skillIsolation=${runtime.capabilities.skillIsolation}`,
126
+ ...(runtime.auditability ? [`auditability=${runtime.auditability.status}`] : []),
108
127
  ...(runtime.binary?.contentHash
109
128
  ? [`contentHash=${runtime.binary.contentHash}`]
110
129
  : []),
@@ -115,7 +134,7 @@ function runtimeTooltip(runtime) {
115
134
  // fingerprint 重复 3 遍,扫读成本高。新版按 (fingerprint, versionText)
116
135
  // 分组,scope 合并到 tag 内 "适用于 ..." 后缀。
117
136
  function renderRuntimeFingerprintTags(meta, lang) {
118
- const scopes = gatherRuntimeScopes(meta);
137
+ const scopes = gatherRuntimeScopes(meta, lang);
119
138
  if (scopes.length === 0)
120
139
  return '';
121
140
  const groups = new Map();
@@ -476,11 +495,11 @@ export function renderRunDetail(report, lang = DEFAULT_LANG, skillContext) {
476
495
  if (list.length === 0)
477
496
  return `<span class="meta-tag">${t('judge', lang)}: —</span>`;
478
497
  if (list.length === 1)
479
- return `<span class="meta-tag">${t('judge', lang)}: ${e(`${list[0].executor}:${list[0].model}`)}</span>`;
480
- return `<span class="meta-tag" title="${t('ensembleDesc', lang)}">${t('judgeModelsLabel', lang)}: ${list.map((j) => e(`${j.executor}:${j.model}`)).join(' · ')}</span>`;
498
+ return `<span class="meta-tag">${t('judge', lang)}: ${e(executorModelDisplayName(list[0].executor, list[0].model, lang))}</span>`;
499
+ return `<span class="meta-tag" title="${t('ensembleDesc', lang)}">${t('judgeModelsLabel', lang)}: ${list.map((j) => e(executorModelDisplayName(j.executor, j.model, lang))).join(' · ')}</span>`;
481
500
  })()}
482
501
  ${m.judgeRepeat && m.judgeRepeat > 1 ? `<span class="meta-tag" title="${t('judgeStddevDesc', lang)}">${t('judgeRepeatLabel', lang)}: ${m.judgeRepeat}</span>` : ''}
483
- <span class="meta-tag">${t('executor', lang)}: ${e(m.executor || 'unknown')}</span>
502
+ <span class="meta-tag">${t('executor', lang)}: ${e(executorDisplayName(m.executor, lang))}</span>
484
503
  ${m.effort ? `<span class="meta-tag" title="${e(lang === 'zh' ? 'executor LLM 的扩展思考预算(--effort)。跨 effort 报告不可严格比较' : 'reasoning effort for executor LLM (--effort); reports across different efforts are not strictly comparable')}">effort: ${e(m.effort)}</span>` : ''}
485
504
  <span class="meta-tag"${execCostReported ? '' : ` title="${e(lang === 'zh' ? 'executor 不报 USD 成本(如 codex CLI),无法估算' : 'executor does not report USD cost (e.g. codex CLI); not measurable')}"`}>${t('cost', lang)}: ${fmtCost(totalExecCost, execCostReported)}</span>
486
505
  <span class="meta-tag"${totalCostReported ? '' : ` title="${e(costCompletenessTooltip(lang))}"`}>${totalCostLabel}: ${fmtKnownCost(m.totalCostUSD, totalCostReported)}</span>${processCostTag ? `
@@ -639,10 +658,10 @@ export function renderBatchEvaluationDetail(report, lang = DEFAULT_LANG) {
639
658
  if (list.length === 0)
640
659
  return `<span class="meta-tag">${t('judge', lang)}: —</span>`;
641
660
  if (list.length === 1)
642
- return `<span class="meta-tag">${t('judge', lang)}: ${e(`${list[0].executor}:${list[0].model}`)}</span>`;
643
- return `<span class="meta-tag" title="${t('ensembleDesc', lang)}">${t('judgeModelsLabel', lang)}: ${list.map((j) => e(`${j.executor}:${j.model}`)).join(' · ')}</span>`;
661
+ return `<span class="meta-tag">${t('judge', lang)}: ${e(executorModelDisplayName(list[0].executor, list[0].model, lang))}</span>`;
662
+ return `<span class="meta-tag" title="${t('ensembleDesc', lang)}">${t('judgeModelsLabel', lang)}: ${list.map((j) => e(executorModelDisplayName(j.executor, j.model, lang))).join(' · ')}</span>`;
644
663
  })()}
645
- <span class="meta-tag">${t('executor', lang)}: ${e(m.executor || 'unknown')}</span>
664
+ <span class="meta-tag">${t('executor', lang)}: ${e(executorDisplayName(m.executor, lang))}</span>
646
665
  <span class="meta-tag"${totalCostReported ? '' : ` title="${e(costCompletenessTooltip(lang))}"`}>${t('totalCost', lang)}: ${fmtKnownCost(m.totalCostUSD, totalCostReported)}</span>
647
666
  </div>
648
667
  ${(() => {
@@ -1101,6 +1101,7 @@ export function renderKnowledgeDebuggerPage(model, lang = DEFAULT_LANG, options
1101
1101
  syncing: zh ? '同步中' : 'Syncing',
1102
1102
  reconnecting: zh ? '重连中' : 'Reconnecting',
1103
1103
  failed: zh ? '实时更新失败' : 'Live update failed',
1104
+ taskFailed: zh ? '任务已失败' : 'Task failed',
1104
1105
  following: zh ? '跟随中' : 'Following',
1105
1106
  resume: zh ? '跟随最新' : 'Follow latest',
1106
1107
  pending: zh ? '查看更新' : 'View update',
@@ -1456,8 +1457,12 @@ function lifecycleEventLabel(label, lang) {
1456
1457
  session_ended: { zh: '会话结束', en: 'Session ended' },
1457
1458
  turn_started: { zh: '本轮开始', en: 'Turn started' },
1458
1459
  turn_completed: { zh: '本轮完成', en: 'Turn completed' },
1460
+ turn_failed: { zh: '本轮失败', en: 'Turn failed' },
1459
1461
  turn_aborted: { zh: '本轮中止', en: 'Turn aborted' },
1460
1462
  turn_interrupted: { zh: '本轮被打断', en: 'Turn interrupted' },
1463
+ turn_ended_unknown: { zh: '本轮结束状态未知', en: 'Turn ended with unknown status' },
1464
+ step_started: { zh: '步骤开始', en: 'Step started' },
1465
+ step_completed: { zh: '步骤完成', en: 'Step completed' },
1461
1466
  };
1462
1467
  return labels[label ?? '']?.[lang] ?? (label || (lang === 'zh' ? '运行状态变化' : 'Lifecycle event'));
1463
1468
  }
@@ -1466,7 +1471,7 @@ function lifecycleMilestoneTone(label) {
1466
1471
  return 'start';
1467
1472
  if (label === 'session_ended' || label === 'turn_completed')
1468
1473
  return 'end';
1469
- if (label === 'turn_aborted' || label === 'turn_interrupted')
1474
+ if (label === 'turn_failed' || label === 'turn_aborted' || label === 'turn_interrupted')
1470
1475
  return 'warning';
1471
1476
  return 'neutral';
1472
1477
  }
@@ -188,8 +188,8 @@ export function renderObservationInboxPage(model, lang = DEFAULT_LANG) {
188
188
  </div>`;
189
189
  };
190
190
  const renderSourceBadge = (item) => {
191
- const label = item.sourceKind === 'openclaw' ? 'OpenClaw' : item.sourceKind === 'codex' ? 'Codex' : item.sourceKind === 'markdown_log' ? 'Markdown log' : item.sourceKind === 'claude' ? 'Claude' : 'Unknown';
192
- const color = item.sourceKind === 'openclaw' ? '#7c3aed' : item.sourceKind === 'codex' ? '#1677ff' : item.sourceKind === 'markdown_log' ? 'var(--green)' : item.sourceKind === 'claude' ? 'var(--accent)' : 'var(--text-muted)';
191
+ const label = item.sourceKind === 'dsh' ? 'DeepSeek Harness' : item.sourceKind === 'openclaw' ? 'OpenClaw' : item.sourceKind === 'codex' ? 'Codex' : item.sourceKind === 'markdown_log' ? 'Markdown log' : item.sourceKind === 'claude' ? 'Claude' : 'Unknown';
192
+ const color = item.sourceKind === 'dsh' ? '#0f766e' : item.sourceKind === 'openclaw' ? '#7c3aed' : item.sourceKind === 'codex' ? '#1677ff' : item.sourceKind === 'markdown_log' ? 'var(--green)' : item.sourceKind === 'claude' ? 'var(--accent)' : 'var(--text-muted)';
193
193
  return `<span title="调用日志来源:${e(label)}" style="display:inline-flex;margin-top:4px;padding:2px 6px;border-radius:999px;background:var(--bg-muted);color:${color};font-size:11px;font-weight:650">${e(label)}</span>`;
194
194
  };
195
195
  const confidenceHeaderHelp = '判断把握:OMK 对“这条 过程发现 是否需要处理/是否高风险/需关注”的规则判断有多确定。归属把握:OMK 把这条 过程发现 归到当前 skill 名下有多确定,例如明确调用 skill 通常更高。';
@@ -4,6 +4,7 @@ export interface TrajectoryLiveLabels {
4
4
  syncing: string;
5
5
  reconnecting: string;
6
6
  failed: string;
7
+ taskFailed: string;
7
8
  following: string;
8
9
  resume: string;
9
10
  pending: string;
@@ -48,7 +48,9 @@ export function createTrajectoryLiveController(options) {
48
48
  const state = terminalState ?? (!followLatest && pendingRevision
49
49
  ? 'pending'
50
50
  : (followLatest ? 'following' : 'paused'));
51
- const terminalLabel = terminalState ? labels[terminalState] : undefined;
51
+ const terminalLabel = terminalState
52
+ ? terminalState === 'failed' ? labels.taskFailed : labels[terminalState]
53
+ : undefined;
52
54
  followButton.dataset.state = state;
53
55
  followButton.dataset.following = String(followLatest);
54
56
  followButton.setAttribute('aria-pressed', String(followLatest));
@@ -212,6 +214,7 @@ export function createTrajectoryLiveController(options) {
212
214
  try {
213
215
  const update = JSON.parse(event.data);
214
216
  const explicitTerminal = update.status === 'completed'
217
+ || update.status === 'failed'
215
218
  || update.status === 'aborted'
216
219
  || update.status === 'interrupted';
217
220
  const streamTerminal = update.liveObservable === false || explicitTerminal;
@@ -226,7 +229,7 @@ export function createTrajectoryLiveController(options) {
226
229
  if (!update.revision || update.revision === shell.dataset.liveRevision) {
227
230
  const quiet = update.status === 'unknown';
228
231
  const connectionLabel = terminalState
229
- ? labels[terminalState]
232
+ ? terminalState === 'failed' ? labels.taskFailed : labels[terminalState]
230
233
  : quiet ? labels.reconnecting : labels.live;
231
234
  setConnectionState(terminalState ?? (quiet ? 'reconnecting' : 'live'), connectionLabel);
232
235
  updateFollowControl();
@@ -1,6 +1,7 @@
1
1
  const TRACE_SOURCE_KINDS = new Set([
2
2
  'claude',
3
3
  'codex',
4
+ 'dsh',
4
5
  'openclaw',
5
6
  'markdown_log',
6
7
  'unknown',
@@ -88,7 +88,7 @@ export interface ExecutorInput {
88
88
  * - claude-cli:物化为临时 settings.json + on-disk hook 脚本,跑完清理
89
89
  * - script(自定义脚本):同样物化临时 settings,通过 env(OMK_MOCK_SETTINGS_FILE /
90
90
  * OMK_MOCK_MCP_CONFIG_FILE / OMK_MOCKS_FILE)暴露给脚本;脚本负责消费该协议
91
- * - codex / codex-sdk / gemini / *-api:不支持,executor capability gate 会拒绝,
91
+ * - codex / codex-sdk / *-api:不支持,executor capability gate 会拒绝,
92
92
  * 绝不静默忽略后把 mock_hit 记成模型失败 */
93
93
  mocks?: import('./eval.js').Mock[];
94
94
  /** 解析 mock.return_file 的相对路径锚点(默认 sample 文件所在目录)。 */
@@ -119,7 +119,13 @@ export interface ExecutorInput {
119
119
  */
120
120
  effort?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
121
121
  }
122
- export type ExecutorFn = (input: ExecutorInput) => Promise<ExecResult>;
122
+ export type ExecutorRuntimeFingerprintResolver = (model: string, options?: {
123
+ skillDir?: string | null;
124
+ }) => ExecutorRuntimeFingerprint;
125
+ export type ExecutorFn = ((input: ExecutorInput) => Promise<ExecResult>) & {
126
+ /** Same-process hosts can report the runtime that actually owns execution. */
127
+ readonly runtimeFingerprint?: ExecutorRuntimeFingerprintResolver;
128
+ };
123
129
  export interface ExecutorCache {
124
130
  get(key: string): ExecResult | null;
125
131
  set(key: string, value: ExecResult): void;
@@ -176,5 +182,10 @@ export interface ExecutorRuntimeFingerprint {
176
182
  fingerprint: string;
177
183
  binary?: ExecutorRuntimeBinary;
178
184
  sdk?: ExecutorRuntimePackage;
185
+ /** Whether the recorded fields cover the complete effective runtime composition. */
186
+ auditability?: {
187
+ status: 'complete' | 'partial';
188
+ reasons?: string[];
189
+ };
179
190
  capabilities: ExecutorRuntimeCapabilities;
180
191
  }
@@ -1,9 +1,9 @@
1
1
  import type { ExecutorRuntimeFingerprint } from './executor.js';
2
2
  /** Single judge configuration: which executor to call and which model alias to pass. */
3
3
  export interface JudgeConfig {
4
- /** Executor name (claude / openai / gemini / anthropic-api / openai-api / shell command). */
4
+ /** Executor name (claude / codex / anthropic-api / openai-api / shell command). */
5
5
  executor: string;
6
- /** Model alias passed to the executor (e.g. "opus", "haiku", "gpt-4o", "gemini-2.0-pro"). */
6
+ /** Model alias passed to the executor (e.g. "opus", "haiku", "gpt-4o"). */
7
7
  model: string;
8
8
  }
9
9
  /** Persisted judge entry on Report.meta.judgeModels: judge config + runtime fingerprint of
@@ -381,7 +381,7 @@ export type DebugKnowledgeAccessKind = 'injected' | 'read' | 'returned';
381
381
  export type TaskReplayStepKind = 'user_request' | 'user_message' | 'user_correction' | 'runtime_context' | 'skill_context' | 'tool_exchange' | 'unmatched_tool_result' | 'assistant_message' | 'model_activity' | 'lifecycle' | 'observation' | 'system_event';
382
382
  export type TaskReplayIntegrityCode = 'task_boundary_unavailable' | 'timeline_truncated' | 'malformed_records' | 'ignored_values' | 'unknown_events' | 'unmatched_tool_calls' | 'unmatched_tool_results' | 'missing_timestamps';
383
383
  export type TaskWindowBasis = 'turn_id' | 'turn_lifecycle' | 'user_message' | 'unresolved';
384
- export type ExperienceTurnStatus = 'completed' | 'aborted' | 'interrupted' | 'open' | 'unknown';
384
+ export type ExperienceTurnStatus = 'completed' | 'failed' | 'aborted' | 'interrupted' | 'open' | 'unknown';
385
385
  /**
386
386
  * One user-visible task inside a source thread. `turnId` is the stable,
387
387
  * source-neutral identity used by Studio routes. `sourceTurnId` preserves the
@@ -1,2 +1,2 @@
1
1
  /** Stable source identity shared by execution evidence and Trace IR. */
2
- export type TraceSourceKind = 'claude' | 'codex' | 'openclaw' | 'markdown_log' | 'unknown';
2
+ export type TraceSourceKind = 'claude' | 'codex' | 'dsh' | 'openclaw' | 'markdown_log' | 'unknown';
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "oh-my-knowledge",
3
- "version": "0.52.3",
3
+ "version": "0.54.0",
4
4
  "packageManager": "yarn@4.16.0",
5
5
  "description": "OMK — Observe. Measure. Know. Evidence-backed knowledge changes for AI applications.",
6
6
  "type": "module",
@@ -13,6 +13,11 @@
13
13
  "files": [
14
14
  "dist/"
15
15
  ],
16
+ "dsh": {
17
+ "bundle": {
18
+ "patch": "./dist/dsh-plugin/cordis.patch.yml"
19
+ }
20
+ },
16
21
  "scripts": {
17
22
  "clean": "rm -rf dist dist-scripts node_modules/.cache/omk",
18
23
  "build": "run-s build:src build:scripts build:assets",
@@ -28,6 +33,7 @@
28
33
  "lint": "eslint 'src/**/*.ts' 'test/**/*.ts' --cache --cache-location node_modules/.cache/eslint/ --max-warnings 0",
29
34
  "lint-staged": "lint-staged",
30
35
  "test": "vitest run",
36
+ "test:profile": "yarn build && node dist-scripts/test-profile.js",
31
37
  "ci": "run-s lint typecheck build build:docs:check test",
32
38
  "prepublishOnly": "run-s clean build",
33
39
  "prepare": "husky"
@@ -79,6 +85,10 @@
79
85
  "release-evidence",
80
86
  "llm-observability",
81
87
  "prompt-engineering",
88
+ "deepseek-harness",
89
+ "dsh-plugin",
90
+ "dsh",
91
+ "cordis",
82
92
  "codex",
83
93
  "codex-cli",
84
94
  "openai",
@@ -98,12 +108,10 @@
98
108
  "author": "lizhiyao",
99
109
  "license": "MIT",
100
110
  "dependencies": {
101
- "@anthropic-ai/claude-agent-sdk": "^0.3.143",
102
- "@anthropic-ai/sdk": "^0.117.1",
111
+ "@anthropic-ai/sdk": "^0.120.0",
103
112
  "@inquirer/prompts": "^8.4.3",
104
113
  "@modelcontextprotocol/sdk": "^1.30.0",
105
114
  "@oclif/core": "^4",
106
- "@openai/codex-sdk": "0.147.0",
107
115
  "ajv": "^8.18.0",
108
116
  "chart.js": "^4.5.1",
109
117
  "es-module-lexer": "^2.0.0",
@@ -111,6 +119,18 @@
111
119
  "simple-statistics": "^7.8.9",
112
120
  "zod": "^4.4.3"
113
121
  },
122
+ "peerDependencies": {
123
+ "@anthropic-ai/claude-agent-sdk": "^0.3.143",
124
+ "@openai/codex-sdk": "^0.149.0"
125
+ },
126
+ "peerDependenciesMeta": {
127
+ "@anthropic-ai/claude-agent-sdk": {
128
+ "optional": true
129
+ },
130
+ "@openai/codex-sdk": {
131
+ "optional": true
132
+ }
133
+ },
114
134
  "publishConfig": {
115
135
  "registry": "https://registry.npmjs.org"
116
136
  },
@@ -124,6 +144,6 @@
124
144
  "typescript": "^6.0.2",
125
145
  "typescript-eslint": "^8.58.0",
126
146
  "vitepress": "^1.6.4",
127
- "vitest": "4.1.10"
147
+ "vitest": "4.1.11"
128
148
  }
129
149
  }
@@ -1,28 +0,0 @@
1
- import type { ExecResult } from '../types/index.js';
2
- import type { ClaudeSdkBaseMessage, ClaudeSdkResultMessage } from './shared.js';
3
- export interface ClaudeSdkMeasurements {
4
- durationMs: number;
5
- durationApiMs: number;
6
- inputTokens: number;
7
- outputTokens: number;
8
- cacheReadTokens: number;
9
- cacheCreationTokens: number;
10
- costUSD: number;
11
- numTurns: number;
12
- }
13
- export declare function normalizeClaudeSdkMeasurements(result: ClaudeSdkResultMessage): ClaudeSdkMeasurements | {
14
- error: string;
15
- };
16
- export interface ClaudeStreamParseResult {
17
- messages: ClaudeSdkBaseMessage[];
18
- malformedLineCount: number;
19
- }
20
- export declare function parseClaudeStreamJson(stdout: string): ClaudeStreamParseResult;
21
- export declare function buildClaudeResult(options: {
22
- messages: ClaudeSdkBaseMessage[];
23
- wallClockDurationMs: number;
24
- source: 'claude stream-json' | 'claude-sdk';
25
- malformedLineCount?: number;
26
- forcedError?: string;
27
- messageTimestamps?: number[];
28
- }): ExecResult;
@@ -1,9 +0,0 @@
1
- import type { ToolCallInfo, TurnInfo } from '../types/index.js';
2
- import type { ClaudeSdkBaseMessage } from './shared.js';
3
- export declare function isClaudeSdkResultMessage(message: ClaudeSdkBaseMessage): boolean;
4
- export declare function extractAgentTrace(messages: ClaudeSdkBaseMessage[], timestamps?: number[]): {
5
- turns: TurnInfo[];
6
- toolCalls: ToolCallInfo[];
7
- fullNumTurns: number;
8
- numSubAgents: number;
9
- };
@@ -1,24 +0,0 @@
1
- import type { ExecResult } from '../types/index.js';
2
- import type { CodexEvent } from './shared.js';
3
- export declare function extractCodexUsage(events: CodexEvent[]): {
4
- input: number;
5
- cached: number;
6
- output: number;
7
- };
8
- /**
9
- * Match @openai/codex-sdk's `finalResponse`: the latest completed
10
- * agent_message is the answer. Earlier messages remain available in `turns`.
11
- */
12
- export declare function extractCodexFinalOutput(events: CodexEvent[]): string;
13
- export declare function extractCodexProtocolError(events: CodexEvent[]): string | undefined;
14
- export declare function validateCodexProtocol(events: CodexEvent[]): string | undefined;
15
- export declare function extractCodexStopReason(events: CodexEvent[]): string;
16
- export declare function sumCodexElapsed(resultEvents: CodexEvent[], wallClock: number): number;
17
- export interface BuildCodexResultOptions {
18
- events: CodexEvent[];
19
- wallClockDurationMs: number;
20
- source: 'codex --json' | 'codex-sdk';
21
- malformedLineCount?: number;
22
- forcedError?: string;
23
- }
24
- export declare function buildCodexResult({ events, wallClockDurationMs, source, malformedLineCount, forcedError, }: BuildCodexResultOptions): ExecResult;
@@ -1,2 +0,0 @@
1
- import type { ExecResult, ExecutorInput } from '../types/index.js';
2
- export declare function geminiExecutor({ model, system, prompt, timeoutMs }: ExecutorInput): Promise<ExecResult>;