oh-my-knowledge 0.47.0 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -18
- package/README.zh.md +55 -23
- package/dist/analysis/coverage-analyzer.d.ts +1 -0
- package/dist/analysis/coverage-analyzer.js +125 -62
- package/dist/analysis/failure-clusterer.js +2 -1
- package/dist/analysis/gap-analyzer.d.ts +2 -2
- package/dist/analysis/gap-analyzer.js +13 -3
- package/dist/analysis/hedging-classifier.d.ts +2 -2
- package/dist/analysis/hedging-classifier.js +3 -4
- package/dist/analysis/report-diagnostics.js +9 -7
- package/dist/analysis/sample-diagnostics.js +6 -6
- package/dist/artifact-graph/doctor.js +15 -7
- package/dist/assets/agent-skills/omk/SKILL.md +27 -7
- package/dist/assets/agent-skills/omk/references/commands.md +19 -17
- package/dist/authoring/evolver.d.ts +10 -6
- package/dist/authoring/evolver.js +496 -83
- package/dist/authoring/generator.d.ts +3 -3
- package/dist/authoring/generator.js +5 -10
- package/dist/authoring/sample-fixer.d.ts +8 -6
- package/dist/authoring/sample-fixer.js +76 -5
- package/dist/cli/commands/doctor.js +31 -14
- package/dist/cli/commands/eval/index.d.ts +3 -0
- package/dist/cli/commands/eval/index.js +163 -17
- package/dist/cli/commands/evolve.d.ts +4 -4
- package/dist/cli/commands/evolve.js +27 -13
- package/dist/cli/commands/init.js +16 -3
- package/dist/cli/commands/observe/inbox.js +44 -21
- package/dist/cli/commands/observe/index.js +20 -11
- package/dist/cli/commands/observe/ingest.d.ts +3 -0
- package/dist/cli/commands/observe/ingest.js +35 -4
- package/dist/cli/commands/sample.d.ts +9 -3
- package/dist/cli/commands/sample.js +91 -74
- package/dist/cli/lib/cmd-flags.d.ts +1 -0
- package/dist/cli/lib/codex-model-hint.d.ts +9 -0
- package/dist/cli/lib/codex-model-hint.js +45 -0
- package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
- package/dist/cli/lib/generation-failure-hint.js +61 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +4 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +38 -6
- package/dist/cli/lib/i18n-dict/help.js +6 -6
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +13 -9
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +34 -2
- package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
- package/dist/cli/lib/llm-failure-classifier.js +8 -0
- package/dist/cli/lib/parse-run-config.d.ts +6 -5
- package/dist/cli/lib/parse-run-config.js +16 -9
- package/dist/cli/lib/runtime-defaults.d.ts +21 -0
- package/dist/cli/lib/runtime-defaults.js +79 -0
- package/dist/diagnosis/observe-mapper.js +14 -15
- package/dist/diagnosis/observe-producer.js +3 -1
- package/dist/diagnosis/studio-projection.js +14 -7
- package/dist/diagnosis/types.d.ts +2 -0
- package/dist/diagnosis/types.js +12 -0
- package/dist/doctor/endpoint-rule.js +2 -1
- package/dist/eval-core/artifact-file-names.js +18 -1
- package/dist/eval-core/artifact-index.d.ts +7 -11
- package/dist/eval-core/artifact-index.js +139 -80
- package/dist/eval-core/cache.d.ts +12 -3
- package/dist/eval-core/cache.js +89 -29
- package/dist/eval-core/comparability.js +10 -6
- package/dist/eval-core/evaluation-execution.d.ts +2 -1
- package/dist/eval-core/evaluation-execution.js +122 -37
- package/dist/eval-core/evaluation-job.d.ts +4 -1
- package/dist/eval-core/evaluation-job.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +15 -13
- package/dist/eval-core/evaluation-reporting.js +54 -52
- package/dist/eval-core/execution-strategy.d.ts +2 -0
- package/dist/eval-core/execution-strategy.js +11 -9
- package/dist/eval-core/fact-checker.js +15 -7
- package/dist/eval-core/holdout.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +2 -2
- package/dist/eval-core/mock-hook.cjs +23 -6
- package/dist/eval-core/mocks-runtime.js +30 -8
- package/dist/eval-core/report-document.d.ts +12 -0
- package/dist/eval-core/report-document.js +1151 -0
- package/dist/eval-core/report-extensions.d.ts +4 -0
- package/dist/eval-core/report-extensions.js +500 -0
- package/dist/eval-core/report-file-migration.js +7 -2
- package/dist/eval-core/resume-compatibility.d.ts +31 -0
- package/dist/eval-core/resume-compatibility.js +141 -0
- package/dist/eval-core/sample-fingerprint.d.ts +12 -0
- package/dist/eval-core/sample-fingerprint.js +193 -0
- package/dist/eval-core/schema.js +86 -31
- package/dist/eval-core/verdict.d.ts +8 -4
- package/dist/eval-core/verdict.js +24 -10
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
- package/dist/eval-workflows/evaluation-pipeline.js +12 -7
- package/dist/eval-workflows/run-evaluation.d.ts +9 -7
- package/dist/eval-workflows/run-evaluation.js +79 -51
- package/dist/executors/anthropic-api.js +65 -9
- package/dist/executors/claude-cli.js +16 -79
- package/dist/executors/claude-protocol.d.ts +28 -0
- package/dist/executors/claude-protocol.js +180 -0
- package/dist/executors/claude-sdk-trace.js +56 -28
- package/dist/executors/claude-sdk.d.ts +1 -0
- package/dist/executors/claude-sdk.js +39 -93
- package/dist/executors/codex-cli-trace.js +166 -31
- package/dist/executors/codex-cli.d.ts +6 -8
- package/dist/executors/codex-cli.js +49 -151
- package/dist/executors/codex-protocol.d.ts +24 -0
- package/dist/executors/codex-protocol.js +234 -0
- package/dist/executors/codex-sdk.js +68 -120
- package/dist/executors/gemini.js +88 -13
- package/dist/executors/index.d.ts +2 -3
- package/dist/executors/index.js +5 -3
- package/dist/executors/openai-api.js +70 -9
- package/dist/executors/runtime-fingerprint.js +88 -11
- package/dist/executors/script-command.d.ts +8 -0
- package/dist/executors/script-command.js +87 -0
- package/dist/executors/script.js +202 -29
- package/dist/executors/shared.d.ts +35 -3
- package/dist/executors/shared.js +113 -15
- package/dist/grading/assertions.d.ts +1 -1
- package/dist/grading/assertions.js +19 -9
- package/dist/grading/diagnostic.d.ts +9 -2
- package/dist/grading/diagnostic.js +25 -2
- package/dist/grading/index.js +10 -4
- package/dist/grading/judge.js +19 -6
- package/dist/grading/layered-scores.d.ts +2 -3
- package/dist/grading/layered-scores.js +2 -3
- package/dist/inputs/load-samples.d.ts +1 -2
- package/dist/inputs/load-samples.js +23 -1
- package/dist/inputs/mcp-resolver.js +6 -3
- package/dist/inputs/sample-document.d.ts +11 -0
- package/dist/inputs/sample-document.js +96 -0
- package/dist/managed/evidence.d.ts +1 -0
- package/dist/managed/evidence.js +1 -1
- package/dist/managed/store.js +200 -91
- package/dist/observability/codex-trace-adapter.d.ts +5 -0
- package/dist/observability/codex-trace-adapter.js +850 -0
- package/dist/observability/experience.d.ts +32 -6
- package/dist/observability/experience.js +2695 -459
- package/dist/observability/feedback-matchers.js +16 -1
- package/dist/observability/inbox-view-model.d.ts +2 -1
- package/dist/observability/inbox-view-model.js +20 -14
- package/dist/observability/inbox.d.ts +7 -1
- package/dist/observability/inbox.js +632 -124
- package/dist/observability/problem-patterns.js +2 -0
- package/dist/observability/review-state.d.ts +6 -0
- package/dist/observability/review-state.js +235 -63
- package/dist/observability/skill-chain-advisories.js +1 -1
- package/dist/observability/skill-chain.js +17 -4
- package/dist/observability/skill-health-analyzer.d.ts +32 -7
- package/dist/observability/skill-health-analyzer.js +194 -121
- package/dist/observability/skill-health-report.d.ts +10 -0
- package/dist/observability/skill-health-report.js +620 -0
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +1 -1
- package/dist/observability/soft-standards/index.js +1 -1
- package/dist/observability/soft-standards/llm-extractor.js +8 -10
- package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
- package/dist/observability/soft-standards/skill-standards-store.js +59 -18
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/observability/trace-adapter.d.ts +12 -7
- package/dist/observability/trace-adapter.js +11 -9
- package/dist/observability/trace-attribution.d.ts +13 -5
- package/dist/observability/trace-attribution.js +315 -21
- package/dist/observability/trace-ingestion.d.ts +9 -0
- package/dist/observability/trace-ingestion.js +80 -0
- package/dist/observability/trace-ir.d.ts +113 -0
- package/dist/observability/trace-ir.js +87 -0
- package/dist/observability/trace-segmenter.d.ts +19 -6
- package/dist/observability/trace-segmenter.js +377 -196
- package/dist/observability/trace-session-index.d.ts +19 -0
- package/dist/observability/trace-session-index.js +68 -0
- package/dist/observability/trace-source.d.ts +12 -4
- package/dist/observability/trace-source.js +939 -215
- package/dist/renderer/html-renderer.js +37 -6
- package/dist/renderer/icons.js +3 -0
- package/dist/renderer/observation-inbox-renderer.js +227 -91
- package/dist/renderer/skill-detail-renderer.js +452 -109
- package/dist/renderer/skill-health-renderer.js +69 -12
- package/dist/renderer/summary.js +28 -7
- package/dist/renderer/table.js +21 -4
- package/dist/renderer/test-view.d.ts +1 -0
- package/dist/renderer/test-view.js +44 -9
- package/dist/server/indexed-report-store.js +14 -18
- package/dist/server/job-store.js +64 -26
- package/dist/server/report-server.js +190 -78
- package/dist/server/report-store.js +57 -80
- package/dist/server/skill-index.js +143 -49
- package/dist/server/skill-insights.js +44 -5
- package/dist/shared/artifact-graph.d.ts +3 -0
- package/dist/shared/artifact-graph.js +224 -0
- package/dist/shared/assertion-types.d.ts +8 -0
- package/dist/shared/assertion-types.js +46 -0
- package/dist/shared/atomic-json.d.ts +8 -0
- package/dist/shared/atomic-json.js +33 -0
- package/dist/shared/diagnosis-schema.d.ts +9 -0
- package/dist/shared/diagnosis-schema.js +181 -0
- package/dist/shared/doctor-report.d.ts +3 -0
- package/dist/shared/doctor-report.js +103 -0
- package/dist/shared/evaluation-job.d.ts +6 -0
- package/dist/shared/evaluation-job.js +217 -0
- package/dist/shared/executor-result.d.ts +17 -0
- package/dist/shared/executor-result.js +221 -0
- package/dist/shared/file-lock.d.ts +12 -0
- package/dist/shared/file-lock.js +129 -0
- package/dist/shared/json-value.d.ts +5 -0
- package/dist/shared/json-value.js +36 -0
- package/dist/shared/keyed-mutex.d.ts +7 -0
- package/dist/shared/keyed-mutex.js +24 -0
- package/dist/shared/record-count.d.ts +8 -0
- package/dist/shared/record-count.js +43 -0
- package/dist/shared/sample-contract.d.ts +3 -0
- package/dist/shared/sample-contract.js +332 -0
- package/dist/shared/shell-quote.d.ts +2 -0
- package/dist/shared/shell-quote.js +7 -0
- package/dist/shared/timestamp.d.ts +6 -0
- package/dist/shared/timestamp.js +64 -0
- package/dist/shared/token-usage.d.ts +19 -0
- package/dist/shared/token-usage.js +50 -0
- package/dist/shared/tool-call-status.d.ts +8 -0
- package/dist/shared/tool-call-status.js +28 -0
- package/dist/shared/tool-identity.d.ts +21 -0
- package/dist/shared/tool-identity.js +84 -0
- package/dist/shared/tool-search.js +73 -16
- package/dist/shared/trace-projection.d.ts +5 -0
- package/dist/shared/trace-projection.js +20 -0
- package/dist/shared/trace-source-kind.d.ts +3 -0
- package/dist/shared/trace-source-kind.js +12 -0
- package/dist/types/diagnosis.d.ts +2 -0
- package/dist/types/eval.d.ts +4 -0
- package/dist/types/executor.d.ts +32 -5
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/judge.d.ts +2 -0
- package/dist/types/observability.d.ts +116 -9
- package/dist/types/report.d.ts +58 -6
- package/dist/types/skill-index.d.ts +7 -0
- package/dist/types/trace.d.ts +2 -0
- package/dist/types/trace.js +1 -0
- package/package.json +9 -5
|
@@ -6,25 +6,51 @@
|
|
|
6
6
|
* 定位是"真实使用 trace 的 skill 维度观察",不是通用 APM / 生产监控。
|
|
7
7
|
*
|
|
8
8
|
* 分析流水线:
|
|
9
|
-
* 1.
|
|
9
|
+
* 1. tracesToResultEntries(path) → segments + ResultEntry[]
|
|
10
10
|
* 2. 时间窗 / skill 白名单过滤
|
|
11
11
|
* 3. 按 skill name (variant key) 分别 computeCoverage + computeGapReport
|
|
12
12
|
* 4. 聚合 overall 指标 + 健康度色带
|
|
13
13
|
*/
|
|
14
14
|
import { buildKnowledgeIndex, computeCoverage } from '../analysis/coverage-analyzer.js';
|
|
15
15
|
import { computeGapReport } from '../analysis/gap-analyzer.js';
|
|
16
|
-
import {
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
16
|
+
import { segmentsToResultEntries, skillSegmentTimestampObserved, tracesToResultEntries, } from './trace-adapter.js';
|
|
17
|
+
import { legacyCcSessionToTraceSession } from './trace-source.js';
|
|
18
|
+
import { createTraceSessionIndex } from './trace-session-index.js';
|
|
19
|
+
import { setOwnRecordValue, sumRecordCounts } from '../shared/record-count.js';
|
|
20
|
+
import { checkedSumTokenCounts } from '../shared/token-usage.js';
|
|
20
21
|
function timestampLt(a, b) {
|
|
21
|
-
|
|
22
|
+
const left = Date.parse(a);
|
|
23
|
+
const right = Date.parse(b);
|
|
24
|
+
return Number.isFinite(left) && Number.isFinite(right) ? left < right : a < b;
|
|
25
|
+
}
|
|
26
|
+
function sumCounts(items, select) {
|
|
27
|
+
let total = 0;
|
|
28
|
+
for (const item of items) {
|
|
29
|
+
total = sumRecordCounts(total, select(item));
|
|
30
|
+
}
|
|
31
|
+
return total;
|
|
32
|
+
}
|
|
33
|
+
function sumNonNegativeFinite(items, select) {
|
|
34
|
+
let total = 0;
|
|
35
|
+
for (const item of items) {
|
|
36
|
+
const value = select(item);
|
|
37
|
+
if (!Number.isFinite(value) || value < 0) {
|
|
38
|
+
throw new TypeError(`Health metric must be a non-negative finite number, got ${String(value)}`);
|
|
39
|
+
}
|
|
40
|
+
total += value;
|
|
41
|
+
if (!Number.isFinite(total)) {
|
|
42
|
+
throw new RangeError('Health metric sum exceeds the finite number range');
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
return total;
|
|
22
46
|
}
|
|
23
47
|
/**
|
|
24
48
|
* 判断 segment 是否落在时间窗内(闭区间)。
|
|
25
49
|
*/
|
|
26
50
|
function withinTimeWindow(seg, from, to) {
|
|
27
|
-
if (from &&
|
|
51
|
+
if ((from || to) && !skillSegmentTimestampObserved(seg))
|
|
52
|
+
return false;
|
|
53
|
+
if (from && timestampLt(seg.endTimestamp, from))
|
|
28
54
|
return false;
|
|
29
55
|
if (to && timestampLt(to, seg.startTimestamp))
|
|
30
56
|
return false;
|
|
@@ -44,8 +70,10 @@ export function healthBandOf(weightedGapRate) {
|
|
|
44
70
|
/**
|
|
45
71
|
* 按 per-skill 失败率判定执行稳定性。阈值见 SkillHealth.stability。
|
|
46
72
|
*/
|
|
47
|
-
function
|
|
48
|
-
if (
|
|
73
|
+
export function toolStabilityOf(toolFailureRate, comparableToolCalls, totalToolCalls) {
|
|
74
|
+
if (totalToolCalls > 0 && comparableToolCalls === 0)
|
|
75
|
+
return 'unknown';
|
|
76
|
+
if (toolFailureRate >= 0.4 && comparableToolCalls >= 5)
|
|
49
77
|
return 'very-unstable';
|
|
50
78
|
if (toolFailureRate >= 0.2)
|
|
51
79
|
return 'unstable';
|
|
@@ -68,38 +96,40 @@ export function confidenceOf(segmentCount) {
|
|
|
68
96
|
* 聚合一组 segment 的 tokens / duration / turns. 平均值按 segment 数(非 toolCall 数)算。
|
|
69
97
|
*/
|
|
70
98
|
function aggregateUsage(skillSegs) {
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
inputTokens += s.metrics.inputTokens ?? 0;
|
|
79
|
-
outputTokens += s.metrics.outputTokens ?? 0;
|
|
80
|
-
cacheReadTokens += s.metrics.cacheReadTokens ?? 0;
|
|
81
|
-
cacheCreationTokens += s.metrics.cacheCreationTokens ?? 0;
|
|
82
|
-
durationMs += s.metrics.durationMs ?? 0;
|
|
83
|
-
numTurns += s.metrics.numTurns ?? 0;
|
|
84
|
-
}
|
|
85
|
-
const totalTokens = inputTokens + outputTokens + cacheReadTokens + cacheCreationTokens;
|
|
86
|
-
const n = skillSegs.length || 1;
|
|
87
|
-
return {
|
|
99
|
+
const observed = skillSegs.filter((segment) => segment.metrics.tokenUsageObserved);
|
|
100
|
+
const inputTokens = checkedSumTokenCounts(...observed.map((s) => s.metrics.inputTokens));
|
|
101
|
+
const outputTokens = checkedSumTokenCounts(...observed.map((s) => s.metrics.outputTokens));
|
|
102
|
+
const cacheReadTokens = checkedSumTokenCounts(...observed.map((s) => s.metrics.cacheReadTokens));
|
|
103
|
+
const cacheCreationTokens = checkedSumTokenCounts(...observed.map((s) => s.metrics.cacheCreationTokens));
|
|
104
|
+
const totalTokens = checkedSumTokenCounts(inputTokens, outputTokens, cacheReadTokens, cacheCreationTokens);
|
|
105
|
+
const tokenAggregateValid = [
|
|
88
106
|
inputTokens,
|
|
89
107
|
outputTokens,
|
|
90
108
|
cacheReadTokens,
|
|
91
109
|
cacheCreationTokens,
|
|
92
110
|
totalTokens,
|
|
111
|
+
].every((value) => value !== undefined);
|
|
112
|
+
const tokenObservedSegmentCount = tokenAggregateValid ? observed.length : 0;
|
|
113
|
+
const durationMs = sumNonNegativeFinite(skillSegs, (segment) => segment.metrics.durationMs);
|
|
114
|
+
const numTurns = sumCounts(skillSegs, (segment) => segment.metrics.numTurns);
|
|
115
|
+
const n = skillSegs.length || 1;
|
|
116
|
+
const tokenDivisor = tokenObservedSegmentCount || 1;
|
|
117
|
+
return {
|
|
118
|
+
inputTokens: inputTokens ?? 0,
|
|
119
|
+
outputTokens: outputTokens ?? 0,
|
|
120
|
+
cacheReadTokens: cacheReadTokens ?? 0,
|
|
121
|
+
cacheCreationTokens: cacheCreationTokens ?? 0,
|
|
122
|
+
totalTokens: totalTokens ?? 0,
|
|
123
|
+
tokenObservedSegmentCount,
|
|
124
|
+
tokenCoverage: skillSegs.length > 0
|
|
125
|
+
? Number((tokenObservedSegmentCount / skillSegs.length).toFixed(4))
|
|
126
|
+
: 1,
|
|
93
127
|
durationMs,
|
|
94
128
|
numTurns,
|
|
95
|
-
avgTokensPerSegment: Math.round(totalTokens /
|
|
129
|
+
avgTokensPerSegment: Math.round((totalTokens ?? 0) / tokenDivisor),
|
|
96
130
|
avgDurationMsPerSegment: Math.round(durationMs / n),
|
|
97
131
|
};
|
|
98
132
|
}
|
|
99
|
-
/**
|
|
100
|
-
* 推断 KB root: 没传 --kb 时,取第一个 assistant record 的 cwd。
|
|
101
|
-
* 如果跨多个 cwd,取第一个并 warn。
|
|
102
|
-
*/
|
|
103
133
|
function inferKbRoot(sessions) {
|
|
104
134
|
const cwds = new Set();
|
|
105
135
|
for (const s of sessions) {
|
|
@@ -114,90 +144,40 @@ function inferKbRoot(sessions) {
|
|
|
114
144
|
return cwds.values().next().value ?? null;
|
|
115
145
|
}
|
|
116
146
|
/**
|
|
117
|
-
*
|
|
147
|
+
* 主入口:从受支持的 trace 输入生成 SkillHealthReport。
|
|
118
148
|
*/
|
|
119
149
|
export function computeSkillHealthReport(tracePath, opts = {}) {
|
|
120
|
-
const { sessions, segments } =
|
|
121
|
-
|
|
122
|
-
let filtered = segments.filter((s) => withinTimeWindow(s, opts.from, opts.to));
|
|
123
|
-
if (opts.skills?.length) {
|
|
124
|
-
const allow = new Set(opts.skills);
|
|
125
|
-
filtered = filtered.filter((s) => allow.has(s.skillName));
|
|
126
|
-
}
|
|
127
|
-
const filteredEntries = segmentsToResultEntries(filtered);
|
|
128
|
-
// 推断 KB root
|
|
129
|
-
const kbRoot = opts.kbRoot ?? inferKbRoot(sessions);
|
|
130
|
-
const index = kbRoot ? buildKnowledgeIndex(kbRoot) : null;
|
|
131
|
-
// 按 skill 分组聚合
|
|
132
|
-
const skillNames = [...new Set(filtered.map((s) => s.skillName))];
|
|
133
|
-
const bySkill = {};
|
|
134
|
-
for (const skill of skillNames) {
|
|
135
|
-
const skillSegs = filtered.filter((s) => s.skillName === skill);
|
|
136
|
-
const coverage = index ? computeCoverage(filteredEntries, skill, index, kbRoot) : null;
|
|
137
|
-
const gap = computeGapReport(filteredEntries, skill);
|
|
138
|
-
// 挂 trace 源作水印(spec §六)
|
|
139
|
-
gap.testSetPath = tracePath;
|
|
140
|
-
const skillToolCalls = skillSegs.reduce((a, s) => a + s.metrics.numToolCalls, 0);
|
|
141
|
-
const skillFailures = skillSegs.reduce((a, s) => a + s.metrics.numToolFailures, 0);
|
|
142
|
-
const toolFailureRate = skillToolCalls > 0 ? Number((skillFailures / skillToolCalls).toFixed(4)) : 0;
|
|
143
|
-
bySkill[skill] = {
|
|
144
|
-
skillName: skill,
|
|
145
|
-
segmentCount: skillSegs.length,
|
|
146
|
-
toolCallCount: skillToolCalls,
|
|
147
|
-
toolFailureCount: skillFailures,
|
|
148
|
-
toolFailureRate,
|
|
149
|
-
stability: stabilityOf(toolFailureRate),
|
|
150
|
-
confidence: confidenceOf(skillSegs.length),
|
|
151
|
-
usage: aggregateUsage(skillSegs),
|
|
152
|
-
coverage,
|
|
153
|
-
gap,
|
|
154
|
-
};
|
|
155
|
-
}
|
|
156
|
-
// Overall 聚合(加权平均,权重 = 每个 skill 的 segment 数)
|
|
157
|
-
const totalSegments = filtered.length;
|
|
158
|
-
const totalGap = Object.values(bySkill).reduce((a, h) => a + h.gap.samplesWithGap, 0);
|
|
159
|
-
const totalWeighted = Object.values(bySkill).reduce((a, h) => a + h.gap.weightedGapRate * h.gap.sampleCount, 0);
|
|
160
|
-
const gapRate = totalSegments > 0 ? Number((totalGap / totalSegments).toFixed(4)) : 0;
|
|
161
|
-
const weightedGapRate = totalSegments > 0 ? Number((totalWeighted / totalSegments).toFixed(4)) : 0;
|
|
162
|
-
// meta
|
|
163
|
-
const totalToolCalls = filtered.reduce((a, s) => a + s.metrics.numToolCalls, 0);
|
|
164
|
-
const totalFailures = filtered.reduce((a, s) => a + s.metrics.numToolFailures, 0);
|
|
165
|
-
const timeRange = filtered.length > 0
|
|
166
|
-
? {
|
|
167
|
-
from: filtered.reduce((m, s) => (timestampLt(s.startTimestamp, m) ? s.startTimestamp : m), filtered[0].startTimestamp),
|
|
168
|
-
to: filtered.reduce((m, s) => (timestampLt(m, s.endTimestamp) ? s.endTimestamp : m), filtered[0].endTimestamp),
|
|
169
|
-
}
|
|
170
|
-
: { from: '', to: '' };
|
|
171
|
-
return {
|
|
172
|
-
kind: 'observe-health',
|
|
173
|
-
meta: {
|
|
174
|
-
tracePath,
|
|
175
|
-
kbPath: kbRoot,
|
|
176
|
-
sessionCount: sessions.length,
|
|
177
|
-
segmentCount: totalSegments,
|
|
178
|
-
messageCount: sessions.reduce((a, s) => a + s.records.length, 0),
|
|
179
|
-
toolCallCount: totalToolCalls,
|
|
180
|
-
toolFailureRate: totalToolCalls > 0 ? Number((totalFailures / totalToolCalls).toFixed(4)) : 0,
|
|
181
|
-
timeRange,
|
|
182
|
-
generatedAt: new Date().toISOString(),
|
|
183
|
-
},
|
|
184
|
-
bySkill,
|
|
185
|
-
overall: { gapRate, weightedGapRate, healthBand: healthBandOf(weightedGapRate), confidence: confidenceOf(totalSegments) },
|
|
186
|
-
};
|
|
150
|
+
const { sessions, segments, ingestion } = tracesToResultEntries(tracePath);
|
|
151
|
+
return computeSkillHealthFromSegments(segments, sessions, tracePath, opts, ingestion);
|
|
187
152
|
}
|
|
188
153
|
/**
|
|
189
154
|
* 便利入口:直接从已准备好的 segments(跳过 loadCcSessions)算 report。
|
|
190
155
|
* 用于测试 / 已手工组装过 segments 的场景。
|
|
191
156
|
*/
|
|
192
|
-
export function computeSkillHealthFromSegments(segments, sessions, tracePath, opts = {}) {
|
|
193
|
-
const
|
|
194
|
-
const
|
|
195
|
-
?
|
|
196
|
-
:
|
|
157
|
+
export function computeSkillHealthFromSegments(segments, sessions, tracePath, opts = {}, ingestion) {
|
|
158
|
+
const candidateSegs = segments.filter((segment) => segment.skillName !== 'general');
|
|
159
|
+
const scopedCandidateSegs = opts.skills?.length
|
|
160
|
+
? candidateSegs.filter((segment) => opts.skills.includes(segment.skillName))
|
|
161
|
+
: candidateSegs;
|
|
162
|
+
const excludedUntimestampedSegmentCount = opts.from || opts.to
|
|
163
|
+
? scopedCandidateSegs.filter((segment) => !skillSegmentTimestampObserved(segment)).length
|
|
164
|
+
: 0;
|
|
165
|
+
const finalSegs = scopedCandidateSegs.filter((segment) => withinTimeWindow(segment, opts.from, opts.to));
|
|
197
166
|
const finalEntries = segmentsToResultEntries(finalSegs);
|
|
198
|
-
return buildReport(finalSegs, finalEntries, sessions, tracePath, opts);
|
|
167
|
+
return buildReport(finalSegs, finalEntries, sessionsForSegments(sessions, finalSegs), tracePath, opts, ingestion, excludedUntimestampedSegmentCount);
|
|
168
|
+
}
|
|
169
|
+
function sessionsForSegments(sessions, segments) {
|
|
170
|
+
if (segments.length === 0)
|
|
171
|
+
return [];
|
|
172
|
+
const traceSessions = sessions.map((session) => 'events' in session ? session : legacyCcSessionToTraceSession(session));
|
|
173
|
+
const index = createTraceSessionIndex(traceSessions);
|
|
174
|
+
const selectedTraceIds = new Set(segments.flatMap((segment) => {
|
|
175
|
+
const session = index.resolve(segment);
|
|
176
|
+
return session ? [session.traceId] : [];
|
|
177
|
+
}));
|
|
178
|
+
return sessions.filter((_, position) => selectedTraceIds.has(traceSessions[position].traceId));
|
|
199
179
|
}
|
|
200
|
-
function buildReport(segments, entries, sessions, tracePath, opts) {
|
|
180
|
+
function buildReport(segments, entries, sessions, tracePath, opts, ingestion, excludedUntimestampedSegmentCount = 0) {
|
|
201
181
|
const kbRoot = opts.kbRoot ?? inferKbRoot(sessions);
|
|
202
182
|
const index = kbRoot ? buildKnowledgeIndex(kbRoot) : null;
|
|
203
183
|
const skillNames = [...new Set(segments.map((s) => s.skillName))];
|
|
@@ -207,33 +187,53 @@ function buildReport(segments, entries, sessions, tracePath, opts) {
|
|
|
207
187
|
const coverage = index ? computeCoverage(entries, skill, index, kbRoot) : null;
|
|
208
188
|
const gap = computeGapReport(entries, skill);
|
|
209
189
|
gap.testSetPath = tracePath;
|
|
210
|
-
const skillToolCalls = skillSegs
|
|
211
|
-
const skillFailures = skillSegs
|
|
212
|
-
const
|
|
213
|
-
|
|
190
|
+
const skillToolCalls = sumCounts(skillSegs, (segment) => segment.metrics.numToolCalls);
|
|
191
|
+
const skillFailures = sumCounts(skillSegs, (segment) => segment.metrics.numToolFailures);
|
|
192
|
+
const skillCancelled = sumCounts(skillSegs, (segment) => segment.metrics.numToolCancelled ?? 0);
|
|
193
|
+
const skillUnknown = sumCounts(skillSegs, (segment) => segment.metrics.numToolUnknown ?? 0);
|
|
194
|
+
const skillResolved = Math.max(0, skillToolCalls - skillUnknown);
|
|
195
|
+
const skillComparable = Math.max(0, skillResolved - skillCancelled);
|
|
196
|
+
const toolFailureRate = skillComparable > 0 ? Number((skillFailures / skillComparable).toFixed(4)) : 0;
|
|
197
|
+
setOwnRecordValue(bySkill, skill, {
|
|
214
198
|
skillName: skill,
|
|
215
199
|
segmentCount: skillSegs.length,
|
|
216
200
|
toolCallCount: skillToolCalls,
|
|
217
201
|
toolFailureCount: skillFailures,
|
|
202
|
+
toolCancelledCount: skillCancelled,
|
|
203
|
+
toolUnknownCount: skillUnknown,
|
|
204
|
+
toolResolvedCount: skillResolved,
|
|
205
|
+
toolOutcomeCoverage: skillToolCalls > 0
|
|
206
|
+
? Number((skillResolved / skillToolCalls).toFixed(4))
|
|
207
|
+
: 1,
|
|
218
208
|
toolFailureRate,
|
|
219
|
-
stability:
|
|
209
|
+
stability: toolStabilityOf(toolFailureRate, skillComparable, skillToolCalls),
|
|
220
210
|
confidence: confidenceOf(skillSegs.length),
|
|
221
211
|
usage: aggregateUsage(skillSegs),
|
|
222
212
|
coverage,
|
|
223
213
|
gap,
|
|
224
|
-
};
|
|
214
|
+
});
|
|
225
215
|
}
|
|
226
216
|
const totalSegments = segments.length;
|
|
227
|
-
const
|
|
228
|
-
const
|
|
217
|
+
const healthRows = Object.values(bySkill);
|
|
218
|
+
const totalGap = sumCounts(healthRows, (health) => health.gap.samplesWithGap);
|
|
219
|
+
const totalWeighted = sumNonNegativeFinite(healthRows, (health) => health.gap.weightedGapRate * health.gap.sampleCount);
|
|
229
220
|
const gapRate = totalSegments > 0 ? Number((totalGap / totalSegments).toFixed(4)) : 0;
|
|
230
221
|
const weightedGapRate = totalSegments > 0 ? Number((totalWeighted / totalSegments).toFixed(4)) : 0;
|
|
231
|
-
const totalToolCalls = segments
|
|
232
|
-
const totalFailures = segments
|
|
233
|
-
const
|
|
222
|
+
const totalToolCalls = sumCounts(segments, (segment) => segment.metrics.numToolCalls);
|
|
223
|
+
const totalFailures = sumCounts(segments, (segment) => segment.metrics.numToolFailures);
|
|
224
|
+
const totalCancelled = sumCounts(segments, (segment) => segment.metrics.numToolCancelled ?? 0);
|
|
225
|
+
const totalUnknown = sumCounts(segments, (segment) => segment.metrics.numToolUnknown ?? 0);
|
|
226
|
+
const totalResolved = Math.max(0, totalToolCalls - totalUnknown);
|
|
227
|
+
const totalComparable = Math.max(0, totalResolved - totalCancelled);
|
|
228
|
+
const timestampedSegments = segments.filter(skillSegmentTimestampObserved);
|
|
229
|
+
const timeRange = timestampedSegments.length > 0
|
|
234
230
|
? {
|
|
235
|
-
from:
|
|
236
|
-
|
|
231
|
+
from: timestampedSegments.reduce((minimum, segment) => timestampLt(segment.startTimestamp, minimum)
|
|
232
|
+
? segment.startTimestamp
|
|
233
|
+
: minimum, timestampedSegments[0].startTimestamp),
|
|
234
|
+
to: timestampedSegments.reduce((maximum, segment) => timestampLt(maximum, segment.endTimestamp)
|
|
235
|
+
? segment.endTimestamp
|
|
236
|
+
: maximum, timestampedSegments[0].endTimestamp),
|
|
237
237
|
}
|
|
238
238
|
: { from: '', to: '' };
|
|
239
239
|
return {
|
|
@@ -243,13 +243,86 @@ function buildReport(segments, entries, sessions, tracePath, opts) {
|
|
|
243
243
|
kbPath: kbRoot,
|
|
244
244
|
sessionCount: sessions.length,
|
|
245
245
|
segmentCount: totalSegments,
|
|
246
|
-
messageCount: sessions
|
|
246
|
+
messageCount: scopedSessionMessageCount(sessions, segments),
|
|
247
|
+
timestampedSegmentCount: timestampedSegments.length,
|
|
248
|
+
timestampCoverage: totalSegments > 0
|
|
249
|
+
? Number((timestampedSegments.length / totalSegments).toFixed(4))
|
|
250
|
+
: 1,
|
|
251
|
+
excludedUntimestampedSegmentCount,
|
|
247
252
|
toolCallCount: totalToolCalls,
|
|
248
|
-
|
|
253
|
+
toolCancelledCount: totalCancelled,
|
|
254
|
+
toolUnknownCount: totalUnknown,
|
|
255
|
+
toolResolvedCount: totalResolved,
|
|
256
|
+
toolOutcomeCoverage: totalToolCalls > 0
|
|
257
|
+
? Number((totalResolved / totalToolCalls).toFixed(4))
|
|
258
|
+
: 1,
|
|
259
|
+
toolFailureRate: totalComparable > 0 ? Number((totalFailures / totalComparable).toFixed(4)) : 0,
|
|
249
260
|
timeRange,
|
|
250
261
|
generatedAt: new Date().toISOString(),
|
|
262
|
+
...(ingestion ? { ingestion } : {}),
|
|
251
263
|
},
|
|
252
264
|
bySkill,
|
|
253
265
|
overall: { gapRate, weightedGapRate, healthBand: healthBandOf(weightedGapRate), confidence: confidenceOf(totalSegments) },
|
|
254
266
|
};
|
|
255
267
|
}
|
|
268
|
+
function scopedSessionMessageCount(sessions, segments) {
|
|
269
|
+
const traceSessions = sessions.map((session) => 'events' in session ? session : legacyCcSessionToTraceSession(session));
|
|
270
|
+
const index = createTraceSessionIndex(traceSessions);
|
|
271
|
+
const rangesByTraceId = new Map();
|
|
272
|
+
const unboundedTraceIds = new Set();
|
|
273
|
+
for (const segment of segments) {
|
|
274
|
+
const session = index.resolve(segment);
|
|
275
|
+
if (!session)
|
|
276
|
+
continue;
|
|
277
|
+
if (segment.startRecordIndex === undefined
|
|
278
|
+
|| segment.endRecordIndex === undefined) {
|
|
279
|
+
unboundedTraceIds.add(session.traceId);
|
|
280
|
+
continue;
|
|
281
|
+
}
|
|
282
|
+
const ranges = rangesByTraceId.get(session.traceId) ?? [];
|
|
283
|
+
ranges.push({
|
|
284
|
+
start: segment.startRecordIndex,
|
|
285
|
+
end: segment.endRecordIndex,
|
|
286
|
+
});
|
|
287
|
+
rangesByTraceId.set(session.traceId, ranges);
|
|
288
|
+
}
|
|
289
|
+
let total = 0;
|
|
290
|
+
for (const [position, session] of sessions.entries()) {
|
|
291
|
+
if (!('events' in session)) {
|
|
292
|
+
total = sumRecordCounts(total, sessionMessageCount(session));
|
|
293
|
+
continue;
|
|
294
|
+
}
|
|
295
|
+
const traceId = traceSessions[position].traceId;
|
|
296
|
+
const ranges = unboundedTraceIds.has(traceId)
|
|
297
|
+
? undefined
|
|
298
|
+
: rangesByTraceId.get(traceId);
|
|
299
|
+
const count = !ranges
|
|
300
|
+
? sessionMessageCount(session)
|
|
301
|
+
: session.events.filter((event) => event.eventKind === 'message'
|
|
302
|
+
&& ranges.some((range) => event.sourceIndex >= range.start && event.sourceIndex <= range.end)).length;
|
|
303
|
+
total = sumRecordCounts(total, count);
|
|
304
|
+
}
|
|
305
|
+
return total;
|
|
306
|
+
}
|
|
307
|
+
function sessionMessageCount(session) {
|
|
308
|
+
if ('events' in session) {
|
|
309
|
+
return session.events.filter((event) => event.eventKind === 'message').length;
|
|
310
|
+
}
|
|
311
|
+
return session.records.filter((record) => {
|
|
312
|
+
if (!record || typeof record !== 'object')
|
|
313
|
+
return false;
|
|
314
|
+
const typed = record;
|
|
315
|
+
if (typed.type !== 'user' && typed.type !== 'assistant')
|
|
316
|
+
return false;
|
|
317
|
+
const content = typed.message?.content;
|
|
318
|
+
if (typeof content === 'string')
|
|
319
|
+
return content.trim().length > 0;
|
|
320
|
+
if (!Array.isArray(content))
|
|
321
|
+
return false;
|
|
322
|
+
return content.some((part) => part
|
|
323
|
+
&& typeof part === 'object'
|
|
324
|
+
&& part.type === 'text'
|
|
325
|
+
&& typeof part.text === 'string'
|
|
326
|
+
&& Boolean(part.text.trim()));
|
|
327
|
+
}).length;
|
|
328
|
+
}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import type { SkillHealthReport } from './skill-health-analyzer.js';
|
|
2
|
+
/**
|
|
3
|
+
* Parse persisted observe-health data at the storage boundary.
|
|
4
|
+
*
|
|
5
|
+
* Legacy reports may omit additive fields such as usage, confidence and
|
|
6
|
+
* tool-outcome coverage. Those fields are reconstructed from authoritative
|
|
7
|
+
* counts. Contradictory counts/rates are rejected; derived labels are
|
|
8
|
+
* recomputed because their threshold formula can evolve between releases.
|
|
9
|
+
*/
|
|
10
|
+
export declare function parseSkillHealthReport(value: unknown): SkillHealthReport | null;
|