oh-my-knowledge 0.48.0 → 0.49.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (243) hide show
  1. package/README.md +50 -18
  2. package/README.zh.md +55 -23
  3. package/dist/analysis/coverage-analyzer.d.ts +1 -0
  4. package/dist/analysis/coverage-analyzer.js +125 -62
  5. package/dist/analysis/failure-clusterer.js +2 -1
  6. package/dist/analysis/gap-analyzer.d.ts +2 -2
  7. package/dist/analysis/gap-analyzer.js +13 -3
  8. package/dist/analysis/hedging-classifier.d.ts +2 -2
  9. package/dist/analysis/hedging-classifier.js +3 -4
  10. package/dist/analysis/report-diagnostics.js +9 -7
  11. package/dist/analysis/sample-diagnostics.js +6 -6
  12. package/dist/artifact-graph/doctor.js +15 -7
  13. package/dist/assets/agent-skills/omk/SKILL.md +27 -7
  14. package/dist/assets/agent-skills/omk/references/commands.md +18 -17
  15. package/dist/authoring/evolver.d.ts +10 -6
  16. package/dist/authoring/evolver.js +496 -83
  17. package/dist/authoring/generator.d.ts +3 -3
  18. package/dist/authoring/generator.js +5 -10
  19. package/dist/authoring/sample-fixer.d.ts +8 -6
  20. package/dist/authoring/sample-fixer.js +76 -5
  21. package/dist/cli/commands/doctor.js +31 -14
  22. package/dist/cli/commands/eval/index.d.ts +3 -0
  23. package/dist/cli/commands/eval/index.js +163 -17
  24. package/dist/cli/commands/evolve.d.ts +4 -4
  25. package/dist/cli/commands/evolve.js +27 -13
  26. package/dist/cli/commands/init.js +16 -3
  27. package/dist/cli/commands/observe/inbox.js +28 -21
  28. package/dist/cli/commands/observe/index.js +20 -11
  29. package/dist/cli/commands/observe/ingest.d.ts +3 -0
  30. package/dist/cli/commands/observe/ingest.js +30 -2
  31. package/dist/cli/commands/sample.d.ts +6 -3
  32. package/dist/cli/commands/sample.js +72 -68
  33. package/dist/cli/lib/codex-model-hint.d.ts +9 -0
  34. package/dist/cli/lib/codex-model-hint.js +45 -0
  35. package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
  36. package/dist/cli/lib/generation-failure-hint.js +61 -0
  37. package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
  38. package/dist/cli/lib/i18n-dict/common.js +4 -0
  39. package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
  40. package/dist/cli/lib/i18n-dict/gen.js +38 -6
  41. package/dist/cli/lib/i18n-dict/help.js +6 -6
  42. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  43. package/dist/cli/lib/i18n-dict/init.js +13 -9
  44. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  45. package/dist/cli/lib/i18n-dict/run.js +34 -2
  46. package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
  47. package/dist/cli/lib/llm-failure-classifier.js +8 -0
  48. package/dist/cli/lib/parse-run-config.d.ts +6 -5
  49. package/dist/cli/lib/parse-run-config.js +16 -9
  50. package/dist/cli/lib/runtime-defaults.d.ts +21 -0
  51. package/dist/cli/lib/runtime-defaults.js +79 -0
  52. package/dist/diagnosis/observe-mapper.js +14 -15
  53. package/dist/diagnosis/observe-producer.js +3 -1
  54. package/dist/diagnosis/studio-projection.js +14 -7
  55. package/dist/diagnosis/types.d.ts +2 -0
  56. package/dist/diagnosis/types.js +12 -0
  57. package/dist/doctor/endpoint-rule.js +2 -1
  58. package/dist/eval-core/artifact-file-names.js +18 -1
  59. package/dist/eval-core/artifact-index.d.ts +7 -11
  60. package/dist/eval-core/artifact-index.js +139 -80
  61. package/dist/eval-core/cache.d.ts +12 -3
  62. package/dist/eval-core/cache.js +89 -29
  63. package/dist/eval-core/comparability.js +10 -6
  64. package/dist/eval-core/evaluation-execution.d.ts +2 -1
  65. package/dist/eval-core/evaluation-execution.js +122 -37
  66. package/dist/eval-core/evaluation-job.d.ts +4 -1
  67. package/dist/eval-core/evaluation-job.js +4 -1
  68. package/dist/eval-core/evaluation-reporting.d.ts +15 -13
  69. package/dist/eval-core/evaluation-reporting.js +54 -52
  70. package/dist/eval-core/execution-strategy.d.ts +2 -0
  71. package/dist/eval-core/execution-strategy.js +11 -9
  72. package/dist/eval-core/fact-checker.js +15 -7
  73. package/dist/eval-core/holdout.js +3 -2
  74. package/dist/eval-core/judge-independence.d.ts +2 -2
  75. package/dist/eval-core/mock-hook.cjs +23 -6
  76. package/dist/eval-core/mocks-runtime.js +30 -8
  77. package/dist/eval-core/report-document.d.ts +12 -0
  78. package/dist/eval-core/report-document.js +1151 -0
  79. package/dist/eval-core/report-extensions.d.ts +4 -0
  80. package/dist/eval-core/report-extensions.js +500 -0
  81. package/dist/eval-core/report-file-migration.js +7 -2
  82. package/dist/eval-core/resume-compatibility.d.ts +31 -0
  83. package/dist/eval-core/resume-compatibility.js +141 -0
  84. package/dist/eval-core/sample-fingerprint.d.ts +12 -0
  85. package/dist/eval-core/sample-fingerprint.js +193 -0
  86. package/dist/eval-core/schema.js +86 -31
  87. package/dist/eval-core/verdict.d.ts +8 -4
  88. package/dist/eval-core/verdict.js +24 -10
  89. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
  90. package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
  91. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
  92. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
  93. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
  94. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
  95. package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
  96. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
  97. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
  98. package/dist/eval-workflows/evaluation-pipeline.js +12 -7
  99. package/dist/eval-workflows/run-evaluation.d.ts +9 -7
  100. package/dist/eval-workflows/run-evaluation.js +79 -51
  101. package/dist/executors/anthropic-api.js +65 -9
  102. package/dist/executors/claude-cli.js +16 -79
  103. package/dist/executors/claude-protocol.d.ts +28 -0
  104. package/dist/executors/claude-protocol.js +180 -0
  105. package/dist/executors/claude-sdk-trace.js +56 -28
  106. package/dist/executors/claude-sdk.d.ts +1 -0
  107. package/dist/executors/claude-sdk.js +39 -93
  108. package/dist/executors/codex-cli-trace.js +166 -31
  109. package/dist/executors/codex-cli.d.ts +6 -8
  110. package/dist/executors/codex-cli.js +49 -151
  111. package/dist/executors/codex-protocol.d.ts +24 -0
  112. package/dist/executors/codex-protocol.js +234 -0
  113. package/dist/executors/codex-sdk.js +68 -120
  114. package/dist/executors/gemini.js +88 -13
  115. package/dist/executors/index.d.ts +2 -3
  116. package/dist/executors/index.js +5 -3
  117. package/dist/executors/openai-api.js +70 -9
  118. package/dist/executors/runtime-fingerprint.js +88 -11
  119. package/dist/executors/script-command.d.ts +8 -0
  120. package/dist/executors/script-command.js +87 -0
  121. package/dist/executors/script.js +202 -29
  122. package/dist/executors/shared.d.ts +35 -3
  123. package/dist/executors/shared.js +113 -15
  124. package/dist/grading/assertions.d.ts +1 -1
  125. package/dist/grading/assertions.js +19 -9
  126. package/dist/grading/diagnostic.d.ts +9 -2
  127. package/dist/grading/diagnostic.js +25 -2
  128. package/dist/grading/index.js +10 -4
  129. package/dist/grading/judge.js +19 -6
  130. package/dist/grading/layered-scores.d.ts +2 -3
  131. package/dist/grading/layered-scores.js +2 -3
  132. package/dist/inputs/load-samples.d.ts +1 -2
  133. package/dist/inputs/load-samples.js +23 -1
  134. package/dist/inputs/mcp-resolver.js +6 -3
  135. package/dist/inputs/sample-document.d.ts +11 -0
  136. package/dist/inputs/sample-document.js +96 -0
  137. package/dist/managed/evidence.d.ts +1 -0
  138. package/dist/managed/evidence.js +1 -1
  139. package/dist/managed/store.js +200 -91
  140. package/dist/observability/codex-trace-adapter.d.ts +5 -0
  141. package/dist/observability/codex-trace-adapter.js +850 -0
  142. package/dist/observability/experience.d.ts +32 -6
  143. package/dist/observability/experience.js +2695 -459
  144. package/dist/observability/feedback-matchers.js +16 -1
  145. package/dist/observability/inbox-view-model.d.ts +1 -1
  146. package/dist/observability/inbox-view-model.js +19 -14
  147. package/dist/observability/inbox.d.ts +7 -1
  148. package/dist/observability/inbox.js +632 -124
  149. package/dist/observability/problem-patterns.js +2 -0
  150. package/dist/observability/review-state.d.ts +6 -0
  151. package/dist/observability/review-state.js +235 -63
  152. package/dist/observability/skill-chain-advisories.js +1 -1
  153. package/dist/observability/skill-chain.js +17 -4
  154. package/dist/observability/skill-health-analyzer.d.ts +32 -7
  155. package/dist/observability/skill-health-analyzer.js +194 -121
  156. package/dist/observability/skill-health-report.d.ts +10 -0
  157. package/dist/observability/skill-health-report.js +620 -0
  158. package/dist/observability/soft-standards/constants.d.ts +0 -1
  159. package/dist/observability/soft-standards/constants.js +0 -1
  160. package/dist/observability/soft-standards/index.d.ts +1 -1
  161. package/dist/observability/soft-standards/index.js +1 -1
  162. package/dist/observability/soft-standards/llm-extractor.js +8 -10
  163. package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
  164. package/dist/observability/soft-standards/skill-standards-store.js +59 -18
  165. package/dist/observability/soft-standards/types.d.ts +2 -2
  166. package/dist/observability/trace-adapter.d.ts +12 -7
  167. package/dist/observability/trace-adapter.js +11 -9
  168. package/dist/observability/trace-attribution.d.ts +13 -5
  169. package/dist/observability/trace-attribution.js +315 -21
  170. package/dist/observability/trace-ingestion.d.ts +9 -0
  171. package/dist/observability/trace-ingestion.js +80 -0
  172. package/dist/observability/trace-ir.d.ts +113 -0
  173. package/dist/observability/trace-ir.js +87 -0
  174. package/dist/observability/trace-segmenter.d.ts +19 -6
  175. package/dist/observability/trace-segmenter.js +377 -196
  176. package/dist/observability/trace-session-index.d.ts +19 -0
  177. package/dist/observability/trace-session-index.js +68 -0
  178. package/dist/observability/trace-source.d.ts +12 -4
  179. package/dist/observability/trace-source.js +939 -215
  180. package/dist/renderer/html-renderer.js +37 -6
  181. package/dist/renderer/icons.js +3 -0
  182. package/dist/renderer/observation-inbox-renderer.js +208 -90
  183. package/dist/renderer/skill-detail-renderer.js +452 -109
  184. package/dist/renderer/skill-health-renderer.js +69 -12
  185. package/dist/renderer/summary.js +28 -7
  186. package/dist/renderer/table.js +21 -4
  187. package/dist/renderer/test-view.d.ts +1 -0
  188. package/dist/renderer/test-view.js +44 -9
  189. package/dist/server/indexed-report-store.js +14 -18
  190. package/dist/server/job-store.js +64 -26
  191. package/dist/server/report-server.js +190 -78
  192. package/dist/server/report-store.js +57 -80
  193. package/dist/server/skill-index.js +143 -49
  194. package/dist/server/skill-insights.js +44 -5
  195. package/dist/shared/artifact-graph.d.ts +3 -0
  196. package/dist/shared/artifact-graph.js +224 -0
  197. package/dist/shared/assertion-types.d.ts +8 -0
  198. package/dist/shared/assertion-types.js +46 -0
  199. package/dist/shared/atomic-json.d.ts +8 -0
  200. package/dist/shared/atomic-json.js +33 -0
  201. package/dist/shared/diagnosis-schema.d.ts +9 -0
  202. package/dist/shared/diagnosis-schema.js +181 -0
  203. package/dist/shared/doctor-report.d.ts +3 -0
  204. package/dist/shared/doctor-report.js +103 -0
  205. package/dist/shared/evaluation-job.d.ts +6 -0
  206. package/dist/shared/evaluation-job.js +217 -0
  207. package/dist/shared/executor-result.d.ts +17 -0
  208. package/dist/shared/executor-result.js +221 -0
  209. package/dist/shared/file-lock.d.ts +12 -0
  210. package/dist/shared/file-lock.js +129 -0
  211. package/dist/shared/json-value.d.ts +5 -0
  212. package/dist/shared/json-value.js +36 -0
  213. package/dist/shared/keyed-mutex.d.ts +7 -0
  214. package/dist/shared/keyed-mutex.js +24 -0
  215. package/dist/shared/record-count.d.ts +8 -0
  216. package/dist/shared/record-count.js +43 -0
  217. package/dist/shared/sample-contract.d.ts +3 -0
  218. package/dist/shared/sample-contract.js +332 -0
  219. package/dist/shared/timestamp.d.ts +6 -0
  220. package/dist/shared/timestamp.js +64 -0
  221. package/dist/shared/token-usage.d.ts +19 -0
  222. package/dist/shared/token-usage.js +50 -0
  223. package/dist/shared/tool-call-status.d.ts +8 -0
  224. package/dist/shared/tool-call-status.js +28 -0
  225. package/dist/shared/tool-identity.d.ts +21 -0
  226. package/dist/shared/tool-identity.js +84 -0
  227. package/dist/shared/tool-search.js +73 -16
  228. package/dist/shared/trace-projection.d.ts +5 -0
  229. package/dist/shared/trace-projection.js +20 -0
  230. package/dist/shared/trace-source-kind.d.ts +3 -0
  231. package/dist/shared/trace-source-kind.js +12 -0
  232. package/dist/types/diagnosis.d.ts +2 -0
  233. package/dist/types/eval.d.ts +4 -0
  234. package/dist/types/executor.d.ts +32 -5
  235. package/dist/types/index.d.ts +1 -0
  236. package/dist/types/index.js +1 -0
  237. package/dist/types/judge.d.ts +2 -0
  238. package/dist/types/observability.d.ts +116 -9
  239. package/dist/types/report.d.ts +58 -6
  240. package/dist/types/skill-index.d.ts +7 -0
  241. package/dist/types/trace.d.ts +2 -0
  242. package/dist/types/trace.js +1 -0
  243. package/package.json +9 -5
@@ -1,27 +1,44 @@
1
1
  /** Skill segmentation and ResultEntry projection for loaded traces. */
2
- import { isToolResultFailureText } from './text-signals.js';
3
- import { extractAttributionSkillRef, extractBusinessActionSkillRef, extractCommandSkillRef, extractSkillReadFileRef, extractSkillScriptCommandRef, extractSkillToolUseRef, stripCommandEnvelopeText, } from './trace-attribution.js';
2
+ import { createHash } from 'node:crypto';
3
+ import { incrementRecordCount } from '../shared/record-count.js';
4
+ import { truncateToolCallsForPersistence, truncateTurnsForPersistence, } from '../shared/trace-projection.js';
5
+ import { sumTokenCounts, tokenCount } from '../shared/token-usage.js';
6
+ import { legacyCcSessionToTraceSession } from './trace-source.js';
7
+ import { normalizeTraceTimestamp } from './trace-ir.js';
8
+ import { extractAttributionSkillRefFromEvent, extractBusinessActionSkillRefFromEvent, extractCommandSkillRefFromEvent, extractSkillReadFileRefFromEvent, extractSkillScriptCommandRefFromEvent, extractSkillToolUseRefFromEvent, stripCommandEnvelopeText, } from './trace-attribution.js';
9
+ export const UNOBSERVED_TRACE_TIMESTAMP = '1970-01-01T00:00:00.000Z';
10
+ export function skillSegmentTimestampObserved(segment) {
11
+ return segment.timestampObserved
12
+ ?? segment.startTimestamp !== UNOBSERVED_TRACE_TIMESTAMP;
13
+ }
4
14
  // ---------- Segment by skill ----------
5
15
  /**
6
- * 扫描 session records, 按 skill 信号把 tool calls 切成多段。
16
+ * 扫描 Trace IR events, 按 skill 信号把 tool calls 切成多段。
7
17
  * 一个 session 可能产生 1-N 个 SkillSegment。
8
18
  */
9
- export function segmentBySkill(session) {
19
+ export function segmentTraceBySkill(session) {
10
20
  const segments = [];
21
+ const orderedEvents = session.events
22
+ .map((event, order) => ({ event, order }))
23
+ .sort((a, b) => a.event.sourceIndex - b.event.sourceIndex || a.order - b.order)
24
+ .map(({ event }) => event);
11
25
  let currentSkill = 'general';
12
26
  let currentSkillSource;
13
27
  let currentSegment = createEmptySegment(session, currentSkill, 0, 0);
14
28
  let segmentIndex = 0;
15
- // 用 tool_use_id → ToolCallInfo 的映射, 收到 tool_result 时回填 output / success
29
+ const lastSourceIndex = orderedEvents.reduce((max, event) => Math.max(max, event.sourceIndex), 0);
16
30
  const pendingToolUses = new Map();
17
- const markCurrentRecord = (recordIndex) => {
18
- currentSegment.endRecordIndex = Math.max(currentSegment.endRecordIndex ?? currentSegment.startRecordIndex ?? recordIndex, recordIndex);
31
+ const assistantTurnsBySourceIndex = new Map();
32
+ const humanTurnsBySourceIndex = new Map();
33
+ const invalidTokenUsage = new WeakSet();
34
+ const markSegmentRecord = (segment, recordIndex) => {
35
+ segment.endRecordIndex = Math.max(segment.endRecordIndex ?? segment.startRecordIndex ?? recordIndex, recordIndex);
19
36
  };
20
37
  const flushCurrent = (endRecordIndex) => {
21
38
  if (currentSegment.turns.length > 0 || currentSegment.toolCalls.length > 0) {
22
39
  if (typeof endRecordIndex === 'number') {
23
40
  const start = currentSegment.startRecordIndex ?? 0;
24
- currentSegment.endRecordIndex = Math.max(start, Math.min(session.records.length - 1, endRecordIndex));
41
+ currentSegment.endRecordIndex = Math.max(start, Math.min(lastSourceIndex, endRecordIndex));
25
42
  }
26
43
  segments.push(currentSegment);
27
44
  return true;
@@ -38,196 +55,331 @@ export function segmentBySkill(session) {
38
55
  currentSkill = ref.skillName;
39
56
  currentSkillSource = ref.pluginName;
40
57
  };
41
- for (const [recordIndex, raw] of session.records.entries()) {
42
- // records 是 unknown[], 按 type 字段做 structural type guard
43
- if (!raw || typeof raw !== 'object' || !('type' in raw))
44
- continue;
45
- const rec = raw;
46
- if (rec.type === 'user') {
47
- const u = rec;
48
- // 检测 skill 信号 2 (slash command)
49
- const cmdSkill = extractCommandSkillRef(u);
50
- if (cmdSkill && !isCurrentSkillRef(cmdSkill)) {
51
- startNewSegment(cmdSkill, recordIndex, u.timestamp, {
52
- source: 'command-name',
53
- confidence: 0.85,
54
- rawSkillRef: cmdSkill.rawSkillRef,
55
- pluginName: cmdSkill.pluginName,
56
- commandName: `/${cmdSkill.rawSkillRef}`,
57
- });
58
- }
59
- else if (!cmdSkill) {
60
- const businessActionSkill = extractBusinessActionSkillRef(u);
61
- if (businessActionSkill && !isCurrentSkillRef(businessActionSkill)) {
62
- startNewSegment(businessActionSkill, recordIndex, u.timestamp, {
63
- source: 'business-action',
64
- confidence: 0.85,
65
- rawSkillRef: businessActionSkill.rawSkillRef,
66
- pluginName: businessActionSkill.pluginName,
67
- commandName: businessActionSkill.rawSkillRef,
68
- });
69
- }
70
- else {
71
- const scriptSkill = extractSkillScriptCommandRef(u);
72
- if (scriptSkill && !isCurrentSkillRef(scriptSkill)) {
73
- startNewSegment(scriptSkill, recordIndex, u.timestamp, {
74
- source: 'skill-script',
75
- confidence: 0.75,
76
- rawSkillRef: scriptSkill.rawSkillRef,
77
- pluginName: scriptSkill.pluginName,
78
- commandName: scriptSkill.rawSkillRef,
79
- });
58
+ const eventGroups = groupEventsBySourceIndex(orderedEvents);
59
+ for (const eventGroup of eventGroups) {
60
+ const boundary = skillBoundaryForEventGroup(eventGroup, session, currentSegment);
61
+ if (boundary && !isCurrentSkillRef(boundary.ref)) {
62
+ startNewSegment(boundary.ref, eventGroup[0].sourceIndex, eventGroup.find((event) => event.timestamp)?.timestamp, boundary.attribution);
63
+ }
64
+ for (const event of eventsInCorrelationOrder(eventGroup)) {
65
+ const recordIndex = event.sourceIndex;
66
+ updateSegmentTimestamp(currentSegment, event.timestamp);
67
+ if (event.eventKind === 'message' && event.role === 'user') {
68
+ if (event.origin === 'human') {
69
+ const textContent = stripCommandEnvelopeText(event.text);
70
+ if (textContent) {
71
+ const existing = humanTurnsBySourceIndex.get(recordIndex);
72
+ if (existing && currentSegment.turns.includes(existing)) {
73
+ existing.content = existing.content
74
+ ? `${existing.content}\n${textContent}`
75
+ : textContent;
76
+ }
77
+ else {
78
+ const turn = { role: 'user', content: textContent };
79
+ currentSegment.turns.push(turn);
80
+ humanTurnsBySourceIndex.set(recordIndex, turn);
81
+ }
80
82
  }
81
83
  }
84
+ updateSegmentTimestamp(currentSegment, event.timestamp);
85
+ markSegmentRecord(currentSegment, recordIndex);
86
+ continue;
82
87
  }
83
- // 处理 tool_result(回填之前的 tool_use)
84
- if (typeof u.message.content !== 'string') {
85
- for (const part of u.message.content) {
86
- if (part.type === 'tool_result') {
87
- const pending = pendingToolUses.get(part.tool_use_id);
88
- if (pending) {
89
- const failed = part.is_error === true || isToolResultFailureText(part.content);
90
- pending.toolCall.output = part.content;
91
- pending.toolCall.success = !failed;
92
- if (failed) {
93
- pending.segmentRef.metrics.numToolFailures += 1;
94
- }
95
- pendingToolUses.delete(part.tool_use_id);
96
- }
88
+ if (event.eventKind === 'message' && event.role === 'assistant') {
89
+ updateSegmentSourceModel(currentSegment, session.sourceMetadata, event.model);
90
+ if (event.text) {
91
+ const existing = assistantTurnsBySourceIndex.get(recordIndex);
92
+ if (existing && currentSegment.turns.includes(existing)) {
93
+ existing.content = existing.content
94
+ ? `${existing.content}\n${event.text}`
95
+ : event.text;
96
+ }
97
+ else {
98
+ const turn = { role: 'assistant', content: event.text };
99
+ currentSegment.turns.push(turn);
100
+ assistantTurnsBySourceIndex.set(recordIndex, turn);
101
+ currentSegment.metrics.numTurns += 1;
97
102
  }
98
103
  }
104
+ updateSegmentTimestamp(currentSegment, event.timestamp);
105
+ markSegmentRecord(currentSegment, recordIndex);
106
+ continue;
99
107
  }
100
- // user text 合并到 tool turn(简化处理, 不强区分角色)
101
- const textContent = extractUserText(u);
102
- if (textContent) {
103
- currentSegment.turns.push({ role: 'tool', content: textContent });
104
- currentSegment.metrics.numTurns += 1;
105
- }
106
- updateSegmentTimestamp(currentSegment, u.timestamp);
107
- markCurrentRecord(recordIndex);
108
- continue;
109
- }
110
- if (rec.type === 'assistant') {
111
- const a = rec;
112
- // 检测 skill 信号 1 (Skill tool_use); 信号 3 (Read SKILL.md) 作 fallback。
113
- // OpenClaw 场景里一条用户消息可能包含多个业务动作,
114
- // 后续读取不同 SKILL.md 才是实际运行到哪个 skill 的稳定边界。
115
- const skillTool = extractSkillToolUseRef(a);
116
- if (skillTool && !isCurrentSkillRef(skillTool)) {
117
- startNewSegment(skillTool, recordIndex, a.timestamp, {
118
- source: 'skill-tool',
119
- confidence: 0.95,
120
- rawSkillRef: skillTool.rawSkillRef,
121
- pluginName: skillTool.pluginName,
122
- });
123
- }
124
- else if (!skillTool) {
125
- const attrSkill = extractAttributionSkillRef(a);
126
- if (attrSkill && !isCurrentSkillRef(attrSkill)) {
127
- startNewSegment(attrSkill, recordIndex, a.timestamp, {
128
- source: 'command-name',
129
- confidence: 0.85,
130
- rawSkillRef: attrSkill.rawSkillRef,
131
- pluginName: attrSkill.pluginName,
132
- commandName: `/${attrSkill.rawSkillRef}`,
133
- });
108
+ if (event.eventKind === 'tool_call') {
109
+ updateSegmentSourceModel(currentSegment, session.sourceMetadata, event.model);
110
+ const toolCall = {
111
+ tool: event.tool.name,
112
+ ...(event.tool.sourceName ? { sourceTool: event.tool.sourceName } : {}),
113
+ ...(event.tool.namespace ? { toolNamespace: event.tool.namespace } : {}),
114
+ ...(event.tool.provider ? { toolProvider: event.tool.provider } : {}),
115
+ input: event.input,
116
+ output: '',
117
+ status: 'unknown',
118
+ statusSource: 'unknown',
119
+ success: false,
120
+ messageIndex: recordIndex,
121
+ messageUuid: event.sourceEventId ?? event.eventId,
122
+ callInstanceId: event.callInstanceId ?? event.eventId,
123
+ toolUseId: event.callId,
124
+ timestamp: event.timestamp,
125
+ sourceTrace: session.sourcePath,
126
+ sourceKind: session.sourceKind,
127
+ traceRole: session.role,
128
+ traceLabel: session.label,
129
+ };
130
+ currentSegment.toolCalls.push(toolCall);
131
+ currentSegment.metrics.numToolCalls += 1;
132
+ const pending = pendingToolUses.get(event.callId) ?? [];
133
+ pending.push({ toolCall, segmentRef: currentSegment });
134
+ pendingToolUses.set(event.callId, pending);
135
+ let turn = assistantTurnsBySourceIndex.get(recordIndex);
136
+ if (!turn || !currentSegment.turns.includes(turn)) {
137
+ turn = { role: 'assistant', content: '' };
138
+ currentSegment.turns.push(turn);
139
+ assistantTurnsBySourceIndex.set(recordIndex, turn);
140
+ currentSegment.metrics.numTurns += 1;
134
141
  }
135
- else {
136
- const scriptSkill = extractSkillScriptCommandRef(a);
137
- if (scriptSkill && !isCurrentSkillRef(scriptSkill)) {
138
- startNewSegment(scriptSkill, recordIndex, a.timestamp, {
139
- source: 'skill-script',
140
- confidence: 0.7,
141
- rawSkillRef: scriptSkill.rawSkillRef,
142
- pluginName: scriptSkill.pluginName,
143
- commandName: scriptSkill.rawSkillRef,
144
- });
142
+ turn.toolCalls = [...(turn.toolCalls ?? []), toolCall];
143
+ updateSegmentTimestamp(currentSegment, event.timestamp);
144
+ markSegmentRecord(currentSegment, recordIndex);
145
+ continue;
146
+ }
147
+ if (event.eventKind === 'tool_result') {
148
+ const queue = pendingToolUses.get(event.callId);
149
+ const matchingIndex = event.callInstanceId
150
+ ? queue?.findIndex((candidate) => candidate.toolCall.callInstanceId === event.callInstanceId)
151
+ : undefined;
152
+ const pending = matchingIndex !== undefined && matchingIndex >= 0
153
+ ? queue?.splice(matchingIndex, 1)[0]
154
+ : event.callInstanceId
155
+ ? undefined
156
+ : queue?.shift();
157
+ if (pending) {
158
+ pending.toolCall.output = event.output;
159
+ pending.toolCall.status = event.status;
160
+ pending.toolCall.statusSource = event.statusSource;
161
+ pending.toolCall.success = event.status === 'success';
162
+ if (event.status === 'failure')
163
+ pending.segmentRef.metrics.numToolFailures += 1;
164
+ if (event.status === 'cancelled') {
165
+ pending.segmentRef.metrics.numToolCancelled =
166
+ (pending.segmentRef.metrics.numToolCancelled ?? 0) + 1;
167
+ }
168
+ if (event.status === 'unknown')
169
+ pending.segmentRef.metrics.numToolUnknown += 1;
170
+ updateSegmentTimestamp(pending.segmentRef, event.timestamp);
171
+ if (pending.segmentRef === currentSegment) {
172
+ markSegmentRecord(pending.segmentRef, recordIndex);
145
173
  }
146
174
  else {
147
- const readSkill = extractSkillReadFileRef(a);
148
- if (readSkill && !isCurrentSkillRef(readSkill) && shouldCutOnReadSkill(session, currentSegment)) {
149
- startNewSegment(readSkill, recordIndex, a.timestamp, {
150
- source: 'read-skill-md',
151
- confidence: 0.5,
152
- rawSkillRef: readSkill.rawSkillRef,
153
- pluginName: readSkill.pluginName,
154
- });
155
- }
175
+ // The result completes the originating call, but its source record
176
+ // belongs to the segment active after the explicit boundary.
177
+ markSegmentRecord(currentSegment, recordIndex);
156
178
  }
179
+ if (queue?.length === 0)
180
+ pendingToolUses.delete(event.callId);
157
181
  }
158
- }
159
- // 提取 tool_use → ToolCallInfo(success 先标 true, 等 tool_result 回填)
160
- const toolCalls = [];
161
- let assistantText = '';
162
- const assistantContent = Array.isArray(a.message.content) ? a.message.content : [];
163
- for (const part of assistantContent) {
164
- if (part.type === 'text' && part.text)
165
- assistantText += part.text;
166
- if (part.type === 'tool_use' && part.id && part.name) {
167
- const tc = {
168
- tool: part.name,
169
- input: part.input ?? {},
170
- output: '',
171
- success: true,
172
- messageIndex: recordIndex,
173
- messageUuid: a.uuid,
174
- toolUseId: part.id,
175
- timestamp: a.timestamp,
176
- sourceTrace: session.sourcePath,
177
- sourceKind: session.sourceKind,
178
- traceRole: session.traceRole,
179
- traceLabel: session.traceLabel,
180
- };
181
- toolCalls.push(tc);
182
- pendingToolUses.set(part.id, { toolCall: tc, segmentRef: currentSegment });
182
+ else {
183
+ updateSegmentTimestamp(currentSegment, event.timestamp);
184
+ markSegmentRecord(currentSegment, recordIndex);
183
185
  }
186
+ continue;
184
187
  }
185
- if (toolCalls.length > 0 || assistantText) {
186
- currentSegment.turns.push({
187
- role: 'assistant',
188
- content: assistantText,
189
- toolCalls: toolCalls.length > 0 ? toolCalls : undefined,
190
- });
191
- currentSegment.metrics.numTurns += 1;
192
- currentSegment.toolCalls.push(...toolCalls);
193
- currentSegment.metrics.numToolCalls += toolCalls.length;
194
- }
195
- // 累加 token usage
196
- const usage = a.message.usage;
197
- if (usage) {
198
- currentSegment.metrics.inputTokens += usage.input_tokens ?? 0;
199
- currentSegment.metrics.outputTokens += usage.output_tokens ?? 0;
200
- currentSegment.metrics.cacheReadTokens += usage.cache_read_input_tokens ?? 0;
201
- currentSegment.metrics.cacheCreationTokens += usage.cache_creation_input_tokens ?? 0;
188
+ if (event.eventKind === 'usage') {
189
+ if (!invalidTokenUsage.has(currentSegment)) {
190
+ const nextUsage = [
191
+ safeTokenCountAddition(currentSegment.metrics.inputTokens, event.inputTokens),
192
+ safeTokenCountAddition(currentSegment.metrics.outputTokens, event.outputTokens),
193
+ safeTokenCountAddition(currentSegment.metrics.cacheReadTokens, event.cacheReadTokens),
194
+ safeTokenCountAddition(currentSegment.metrics.cacheCreationTokens, event.cacheCreationTokens),
195
+ ];
196
+ if (nextUsage.some((value) => value === undefined)
197
+ || safeTokenCountSum(nextUsage) === undefined) {
198
+ currentSegment.metrics.inputTokens = 0;
199
+ currentSegment.metrics.outputTokens = 0;
200
+ currentSegment.metrics.cacheReadTokens = 0;
201
+ currentSegment.metrics.cacheCreationTokens = 0;
202
+ currentSegment.metrics.tokenUsageObserved = false;
203
+ invalidTokenUsage.add(currentSegment);
204
+ }
205
+ else {
206
+ [
207
+ currentSegment.metrics.inputTokens,
208
+ currentSegment.metrics.outputTokens,
209
+ currentSegment.metrics.cacheReadTokens,
210
+ currentSegment.metrics.cacheCreationTokens,
211
+ ] = nextUsage;
212
+ currentSegment.metrics.tokenUsageObserved = true;
213
+ }
214
+ }
215
+ updateSegmentSourceModel(currentSegment, session.sourceMetadata, event.model);
216
+ updateSegmentTimestamp(currentSegment, event.timestamp);
202
217
  }
203
- updateSegmentTimestamp(currentSegment, a.timestamp);
204
- markCurrentRecord(recordIndex);
205
- continue;
218
+ markSegmentRecord(currentSegment, recordIndex);
219
+ }
220
+ }
221
+ for (const queue of pendingToolUses.values()) {
222
+ for (const pending of queue) {
223
+ pending.segmentRef.metrics.numToolUnknown += 1;
206
224
  }
207
- // 其他 type(permission-mode / file-history-snapshot) 不产出事件,但仍属于当前
208
- // skill 生命周期窗口,用于保持下一次 skill 边界前的连续上下文。
209
- markCurrentRecord(recordIndex);
210
225
  }
211
- flushCurrent(session.records.length - 1);
212
- // 孤儿 tool_use(没对应 tool_result 的)保持 success=true, 但标记为未闭合
226
+ flushCurrent(lastSourceIndex);
213
227
  return segments;
214
228
  }
229
+ function eventsInCorrelationOrder(events) {
230
+ return events
231
+ .map((event, order) => ({ event, order }))
232
+ .sort((left, right) => correlationRank(left.event) - correlationRank(right.event) || left.order - right.order)
233
+ .map(({ event }) => event);
234
+ }
235
+ function correlationRank(event) {
236
+ if (event.eventKind === 'tool_call')
237
+ return 0;
238
+ if (event.eventKind === 'tool_result')
239
+ return 2;
240
+ return 1;
241
+ }
242
+ function groupEventsBySourceIndex(events) {
243
+ const groups = [];
244
+ for (const event of events) {
245
+ const current = groups.at(-1);
246
+ if (current?.[0]?.sourceIndex === event.sourceIndex)
247
+ current.push(event);
248
+ else
249
+ groups.push([event]);
250
+ }
251
+ return groups;
252
+ }
253
+ function skillBoundaryForEventGroup(events, session, currentSegment) {
254
+ const humanMessages = events.filter((event) => event.eventKind === 'message' && event.role === 'user' && event.origin === 'human');
255
+ const assistantMessages = events.filter((event) => event.eventKind === 'message' && event.role === 'assistant');
256
+ const toolCalls = events.filter((event) => event.eventKind === 'tool_call');
257
+ for (const event of toolCalls) {
258
+ const ref = extractSkillToolUseRefFromEvent(event);
259
+ if (ref) {
260
+ return {
261
+ ref,
262
+ attribution: {
263
+ source: 'skill-tool',
264
+ confidence: 0.95,
265
+ rawSkillRef: ref.rawSkillRef,
266
+ pluginName: ref.pluginName,
267
+ },
268
+ };
269
+ }
270
+ }
271
+ for (const event of humanMessages) {
272
+ const ref = extractCommandSkillRefFromEvent(event);
273
+ if (ref) {
274
+ return {
275
+ ref,
276
+ attribution: {
277
+ source: 'command-name',
278
+ confidence: 0.85,
279
+ rawSkillRef: ref.rawSkillRef,
280
+ pluginName: ref.pluginName,
281
+ commandName: `/${ref.rawSkillRef}`,
282
+ },
283
+ };
284
+ }
285
+ }
286
+ for (const event of humanMessages) {
287
+ const ref = extractBusinessActionSkillRefFromEvent(event);
288
+ if (ref) {
289
+ return {
290
+ ref,
291
+ attribution: {
292
+ source: 'business-action',
293
+ confidence: 0.85,
294
+ rawSkillRef: ref.rawSkillRef,
295
+ pluginName: ref.pluginName,
296
+ commandName: ref.rawSkillRef,
297
+ },
298
+ };
299
+ }
300
+ }
301
+ for (const event of assistantMessages) {
302
+ const ref = extractAttributionSkillRefFromEvent(event);
303
+ if (ref) {
304
+ return {
305
+ ref,
306
+ attribution: {
307
+ source: 'command-name',
308
+ confidence: 0.85,
309
+ rawSkillRef: ref.rawSkillRef,
310
+ pluginName: ref.pluginName,
311
+ commandName: `/${ref.rawSkillRef}`,
312
+ },
313
+ };
314
+ }
315
+ }
316
+ for (const event of [...humanMessages, ...assistantMessages, ...toolCalls]) {
317
+ const ref = extractSkillScriptCommandRefFromEvent(event);
318
+ if (ref) {
319
+ return {
320
+ ref,
321
+ attribution: {
322
+ source: 'skill-script',
323
+ confidence: event.eventKind === 'message' && event.role === 'user' ? 0.75 : 0.7,
324
+ rawSkillRef: ref.rawSkillRef,
325
+ pluginName: ref.pluginName,
326
+ commandName: ref.rawSkillRef,
327
+ },
328
+ };
329
+ }
330
+ }
331
+ for (const event of toolCalls) {
332
+ const ref = extractSkillReadFileRefFromEvent(event);
333
+ if (ref && shouldCutOnReadSkill(currentSegment)) {
334
+ return {
335
+ ref,
336
+ attribution: {
337
+ source: 'read-skill-md',
338
+ confidence: 0.5,
339
+ rawSkillRef: ref.rawSkillRef,
340
+ pluginName: ref.pluginName,
341
+ },
342
+ };
343
+ }
344
+ }
345
+ return null;
346
+ }
347
+ /** @deprecated Compatibility entry point for Claude-shaped fixtures. */
348
+ export function segmentBySkill(session) {
349
+ return segmentTraceBySkill('events' in session ? session : legacyCcSessionToTraceSession(session));
350
+ }
351
+ function updateSegmentSourceModel(segment, sessionMetadata, model) {
352
+ if (!model)
353
+ return;
354
+ const baseMetadata = { ...sessionMetadata };
355
+ delete baseMetadata.model;
356
+ const models = new Set(segment.sourceMetadata?.model?.split(', ').filter(Boolean) ?? []);
357
+ models.add(model);
358
+ segment.sourceMetadata = {
359
+ ...baseMetadata,
360
+ ...segment.sourceMetadata,
361
+ model: Array.from(models).join(', '),
362
+ };
363
+ }
215
364
  function createEmptySegment(session, skillName, index, recordIndex, timestamp, attribution) {
216
- const ts = timestamp ?? session.startTimestamp ?? new Date().toISOString();
365
+ const observedTimestamp = normalizeTraceTimestamp(timestamp);
366
+ const ts = observedTimestamp ?? UNOBSERVED_TRACE_TIMESTAMP;
217
367
  return {
218
368
  skillName,
219
369
  attribution: attribution ?? { source: 'general', confidence: 0.3 },
220
- sessionId: session.sessionGroupId ?? session.sessionId,
221
- traceSessionId: session.sessionId,
370
+ sessionId: session.rootRunId,
371
+ traceSessionId: session.runId,
372
+ traceId: session.traceId,
222
373
  sourceTrace: session.sourcePath,
223
374
  sourceKind: session.sourceKind,
224
- traceRole: session.traceRole,
225
- traceLabel: session.traceLabel,
375
+ traceRole: session.role,
376
+ traceLabel: session.label,
226
377
  segmentIndex: index,
227
378
  startRecordIndex: recordIndex,
228
379
  endRecordIndex: recordIndex,
229
380
  startTimestamp: ts,
230
381
  endTimestamp: ts,
382
+ timestampObserved: observedTimestamp !== undefined,
231
383
  cwd: session.cwd,
232
384
  turns: [],
233
385
  toolCalls: [],
@@ -237,17 +389,30 @@ function createEmptySegment(session, skillName, index, recordIndex, timestamp, a
237
389
  outputTokens: 0,
238
390
  cacheReadTokens: 0,
239
391
  cacheCreationTokens: 0,
392
+ tokenUsageObserved: false,
240
393
  numTurns: 0,
241
394
  numToolCalls: 0,
242
395
  numToolFailures: 0,
396
+ numToolCancelled: 0,
397
+ numToolUnknown: 0,
243
398
  },
244
399
  };
245
400
  }
246
401
  function updateSegmentTimestamp(seg, timestamp) {
247
- if (!timestamp)
402
+ const normalized = normalizeTraceTimestamp(timestamp);
403
+ if (!normalized)
248
404
  return;
249
- if (!seg.endTimestamp || timestamp > seg.endTimestamp)
250
- seg.endTimestamp = timestamp;
405
+ seg.timestampObserved = true;
406
+ if (seg.startTimestamp === UNOBSERVED_TRACE_TIMESTAMP) {
407
+ seg.startTimestamp = normalized;
408
+ seg.endTimestamp = normalized;
409
+ }
410
+ else {
411
+ if (normalized < seg.startTimestamp)
412
+ seg.startTimestamp = normalized;
413
+ if (!seg.endTimestamp || normalized > seg.endTimestamp)
414
+ seg.endTimestamp = normalized;
415
+ }
251
416
  // 重算 durationMs
252
417
  try {
253
418
  const start = new Date(seg.startTimestamp).getTime();
@@ -257,23 +422,22 @@ function updateSegmentTimestamp(seg, timestamp) {
257
422
  }
258
423
  catch { /* skip */ }
259
424
  }
260
- function shouldCutOnReadSkill(session, currentSegment) {
425
+ function shouldCutOnReadSkill(currentSegment) {
261
426
  return currentSegment.skillName === 'general'
262
- || currentSegment.attribution?.source === 'read-skill-md'
263
- || session.sourceKind === 'openclaw';
427
+ || currentSegment.attribution?.source === 'read-skill-md';
264
428
  }
265
- function extractUserText(record) {
266
- const content = record.message.content;
267
- if (typeof content === 'string')
268
- return stripCommandEnvelopeText(content);
269
- const parts = [];
270
- for (const p of content) {
271
- if (p.type === 'text')
272
- parts.push(stripCommandEnvelopeText(p.text));
273
- if (p.type === 'tool_result' && typeof p.content === 'string')
274
- parts.push(p.content);
429
+ function safeTokenCountAddition(current, value) {
430
+ const addition = tokenCount(value);
431
+ return current <= Number.MAX_SAFE_INTEGER - addition ? current + addition : undefined;
432
+ }
433
+ function safeTokenCountSum(values) {
434
+ let total = 0;
435
+ for (const value of values) {
436
+ if (total > Number.MAX_SAFE_INTEGER - value)
437
+ return undefined;
438
+ total += value;
275
439
  }
276
- return parts.join('\n');
440
+ return total;
277
441
  }
278
442
  // ---------- Segment → ResultEntry ----------
279
443
  /**
@@ -281,41 +445,58 @@ function extractUserText(record) {
281
445
  *
282
446
  * 映射规则(详见 docs/skill-health-spec.md):
283
447
  * - 每 segment 一个 ResultEntry
284
- * - sample_id = `${sessionId}:${segmentIndex}`
448
+ * - sample_id 锚定 traceId + skillName + startRecordIndex
285
449
  * - variant key = skill 名(复用 omk 的 variant 维度作为 skill 分组维度)
286
450
  */
287
451
  export function segmentsToResultEntries(segments) {
288
452
  return segments.map((seg) => ({
289
- sample_id: `${seg.sessionId}:${seg.segmentIndex}`,
453
+ sample_id: `trace:${createHash('sha256')
454
+ .update([
455
+ seg.traceId ?? seg.traceSessionId ?? seg.sessionId,
456
+ seg.skillName,
457
+ String(seg.startRecordIndex ?? 0),
458
+ ].join('\u0000'))
459
+ .digest('hex')
460
+ .slice(0, 32)}`,
290
461
  variants: {
291
462
  [seg.skillName]: buildVariantResult(seg),
292
463
  },
293
464
  }));
294
465
  }
295
466
  function buildVariantResult(seg) {
296
- const totalTokens = seg.metrics.inputTokens + seg.metrics.outputTokens;
297
- const toolSuccessRate = seg.metrics.numToolCalls > 0
298
- ? (seg.metrics.numToolCalls - seg.metrics.numToolFailures) / seg.metrics.numToolCalls
299
- : 1;
467
+ const totalTokens = sumTokenCounts(seg.metrics.inputTokens, seg.metrics.outputTokens, seg.metrics.cacheReadTokens, seg.metrics.cacheCreationTokens);
468
+ const cancelledToolCalls = seg.metrics.numToolCancelled ?? 0;
469
+ const comparableToolCalls = Math.max(0, seg.metrics.numToolCalls - seg.metrics.numToolUnknown - cancelledToolCalls);
470
+ const toolDistribution = {};
471
+ for (const toolCall of seg.toolCalls) {
472
+ incrementRecordCount(toolDistribution, toolCall.tool);
473
+ }
300
474
  return {
301
475
  ok: true,
302
476
  durationMs: seg.metrics.durationMs,
303
- durationApiMs: seg.metrics.durationMs,
477
+ durationApiMs: 0,
304
478
  inputTokens: seg.metrics.inputTokens,
305
479
  outputTokens: seg.metrics.outputTokens,
306
480
  totalTokens,
307
481
  cacheReadTokens: seg.metrics.cacheReadTokens,
308
482
  cacheCreationTokens: seg.metrics.cacheCreationTokens,
483
+ ...(!seg.metrics.tokenUsageObserved && { tokenUsageReportedByExecutor: false }),
309
484
  execCostUSD: 0,
310
485
  judgeCostUSD: 0,
311
486
  costUSD: 0,
487
+ costReportedByExecutor: false,
312
488
  numTurns: seg.metrics.numTurns,
313
489
  numToolCalls: seg.metrics.numToolCalls,
314
490
  numToolFailures: seg.metrics.numToolFailures,
315
- toolSuccessRate,
491
+ numToolCancelled: cancelledToolCalls,
492
+ numToolUnknown: seg.metrics.numToolUnknown,
493
+ ...(comparableToolCalls > 0 && {
494
+ toolSuccessRate: Number(((comparableToolCalls - seg.metrics.numToolFailures) / comparableToolCalls).toFixed(2)),
495
+ }),
316
496
  toolNames: Array.from(new Set(seg.toolCalls.map((tc) => tc.tool))),
497
+ toolDistribution,
317
498
  outputPreview: null,
318
- turns: seg.turns,
319
- toolCalls: seg.toolCalls,
499
+ turns: truncateTurnsForPersistence(seg.turns),
500
+ toolCalls: truncateToolCallsForPersistence(seg.toolCalls),
320
501
  };
321
502
  }