oh-my-knowledge 0.48.0 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -18
- package/README.zh.md +55 -23
- package/dist/analysis/coverage-analyzer.d.ts +1 -0
- package/dist/analysis/coverage-analyzer.js +125 -62
- package/dist/analysis/failure-clusterer.js +2 -1
- package/dist/analysis/gap-analyzer.d.ts +2 -2
- package/dist/analysis/gap-analyzer.js +13 -3
- package/dist/analysis/hedging-classifier.d.ts +2 -2
- package/dist/analysis/hedging-classifier.js +3 -4
- package/dist/analysis/report-diagnostics.js +9 -7
- package/dist/analysis/sample-diagnostics.js +6 -6
- package/dist/artifact-graph/doctor.js +15 -7
- package/dist/assets/agent-skills/omk/SKILL.md +27 -7
- package/dist/assets/agent-skills/omk/references/commands.md +18 -17
- package/dist/authoring/evolver.d.ts +10 -6
- package/dist/authoring/evolver.js +496 -83
- package/dist/authoring/generator.d.ts +3 -3
- package/dist/authoring/generator.js +5 -10
- package/dist/authoring/sample-fixer.d.ts +8 -6
- package/dist/authoring/sample-fixer.js +76 -5
- package/dist/cli/commands/doctor.js +31 -14
- package/dist/cli/commands/eval/index.d.ts +3 -0
- package/dist/cli/commands/eval/index.js +163 -17
- package/dist/cli/commands/evolve.d.ts +4 -4
- package/dist/cli/commands/evolve.js +27 -13
- package/dist/cli/commands/init.js +16 -3
- package/dist/cli/commands/observe/inbox.js +28 -21
- package/dist/cli/commands/observe/index.js +20 -11
- package/dist/cli/commands/observe/ingest.d.ts +3 -0
- package/dist/cli/commands/observe/ingest.js +30 -2
- package/dist/cli/commands/sample.d.ts +6 -3
- package/dist/cli/commands/sample.js +72 -68
- package/dist/cli/lib/codex-model-hint.d.ts +9 -0
- package/dist/cli/lib/codex-model-hint.js +45 -0
- package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
- package/dist/cli/lib/generation-failure-hint.js +61 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +4 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +38 -6
- package/dist/cli/lib/i18n-dict/help.js +6 -6
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +13 -9
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +34 -2
- package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
- package/dist/cli/lib/llm-failure-classifier.js +8 -0
- package/dist/cli/lib/parse-run-config.d.ts +6 -5
- package/dist/cli/lib/parse-run-config.js +16 -9
- package/dist/cli/lib/runtime-defaults.d.ts +21 -0
- package/dist/cli/lib/runtime-defaults.js +79 -0
- package/dist/diagnosis/observe-mapper.js +14 -15
- package/dist/diagnosis/observe-producer.js +3 -1
- package/dist/diagnosis/studio-projection.js +14 -7
- package/dist/diagnosis/types.d.ts +2 -0
- package/dist/diagnosis/types.js +12 -0
- package/dist/doctor/endpoint-rule.js +2 -1
- package/dist/eval-core/artifact-file-names.js +18 -1
- package/dist/eval-core/artifact-index.d.ts +7 -11
- package/dist/eval-core/artifact-index.js +139 -80
- package/dist/eval-core/cache.d.ts +12 -3
- package/dist/eval-core/cache.js +89 -29
- package/dist/eval-core/comparability.js +10 -6
- package/dist/eval-core/evaluation-execution.d.ts +2 -1
- package/dist/eval-core/evaluation-execution.js +122 -37
- package/dist/eval-core/evaluation-job.d.ts +4 -1
- package/dist/eval-core/evaluation-job.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +15 -13
- package/dist/eval-core/evaluation-reporting.js +54 -52
- package/dist/eval-core/execution-strategy.d.ts +2 -0
- package/dist/eval-core/execution-strategy.js +11 -9
- package/dist/eval-core/fact-checker.js +15 -7
- package/dist/eval-core/holdout.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +2 -2
- package/dist/eval-core/mock-hook.cjs +23 -6
- package/dist/eval-core/mocks-runtime.js +30 -8
- package/dist/eval-core/report-document.d.ts +12 -0
- package/dist/eval-core/report-document.js +1151 -0
- package/dist/eval-core/report-extensions.d.ts +4 -0
- package/dist/eval-core/report-extensions.js +500 -0
- package/dist/eval-core/report-file-migration.js +7 -2
- package/dist/eval-core/resume-compatibility.d.ts +31 -0
- package/dist/eval-core/resume-compatibility.js +141 -0
- package/dist/eval-core/sample-fingerprint.d.ts +12 -0
- package/dist/eval-core/sample-fingerprint.js +193 -0
- package/dist/eval-core/schema.js +86 -31
- package/dist/eval-core/verdict.d.ts +8 -4
- package/dist/eval-core/verdict.js +24 -10
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
- package/dist/eval-workflows/evaluation-pipeline.js +12 -7
- package/dist/eval-workflows/run-evaluation.d.ts +9 -7
- package/dist/eval-workflows/run-evaluation.js +79 -51
- package/dist/executors/anthropic-api.js +65 -9
- package/dist/executors/claude-cli.js +16 -79
- package/dist/executors/claude-protocol.d.ts +28 -0
- package/dist/executors/claude-protocol.js +180 -0
- package/dist/executors/claude-sdk-trace.js +56 -28
- package/dist/executors/claude-sdk.d.ts +1 -0
- package/dist/executors/claude-sdk.js +39 -93
- package/dist/executors/codex-cli-trace.js +166 -31
- package/dist/executors/codex-cli.d.ts +6 -8
- package/dist/executors/codex-cli.js +49 -151
- package/dist/executors/codex-protocol.d.ts +24 -0
- package/dist/executors/codex-protocol.js +234 -0
- package/dist/executors/codex-sdk.js +68 -120
- package/dist/executors/gemini.js +88 -13
- package/dist/executors/index.d.ts +2 -3
- package/dist/executors/index.js +5 -3
- package/dist/executors/openai-api.js +70 -9
- package/dist/executors/runtime-fingerprint.js +88 -11
- package/dist/executors/script-command.d.ts +8 -0
- package/dist/executors/script-command.js +87 -0
- package/dist/executors/script.js +202 -29
- package/dist/executors/shared.d.ts +35 -3
- package/dist/executors/shared.js +113 -15
- package/dist/grading/assertions.d.ts +1 -1
- package/dist/grading/assertions.js +19 -9
- package/dist/grading/diagnostic.d.ts +9 -2
- package/dist/grading/diagnostic.js +25 -2
- package/dist/grading/index.js +10 -4
- package/dist/grading/judge.js +19 -6
- package/dist/grading/layered-scores.d.ts +2 -3
- package/dist/grading/layered-scores.js +2 -3
- package/dist/inputs/load-samples.d.ts +1 -2
- package/dist/inputs/load-samples.js +23 -1
- package/dist/inputs/mcp-resolver.js +6 -3
- package/dist/inputs/sample-document.d.ts +11 -0
- package/dist/inputs/sample-document.js +96 -0
- package/dist/managed/evidence.d.ts +1 -0
- package/dist/managed/evidence.js +1 -1
- package/dist/managed/store.js +200 -91
- package/dist/observability/codex-trace-adapter.d.ts +5 -0
- package/dist/observability/codex-trace-adapter.js +850 -0
- package/dist/observability/experience.d.ts +32 -6
- package/dist/observability/experience.js +2695 -459
- package/dist/observability/feedback-matchers.js +16 -1
- package/dist/observability/inbox-view-model.d.ts +1 -1
- package/dist/observability/inbox-view-model.js +19 -14
- package/dist/observability/inbox.d.ts +7 -1
- package/dist/observability/inbox.js +632 -124
- package/dist/observability/problem-patterns.js +2 -0
- package/dist/observability/review-state.d.ts +6 -0
- package/dist/observability/review-state.js +235 -63
- package/dist/observability/skill-chain-advisories.js +1 -1
- package/dist/observability/skill-chain.js +17 -4
- package/dist/observability/skill-health-analyzer.d.ts +32 -7
- package/dist/observability/skill-health-analyzer.js +194 -121
- package/dist/observability/skill-health-report.d.ts +10 -0
- package/dist/observability/skill-health-report.js +620 -0
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +1 -1
- package/dist/observability/soft-standards/index.js +1 -1
- package/dist/observability/soft-standards/llm-extractor.js +8 -10
- package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
- package/dist/observability/soft-standards/skill-standards-store.js +59 -18
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/observability/trace-adapter.d.ts +12 -7
- package/dist/observability/trace-adapter.js +11 -9
- package/dist/observability/trace-attribution.d.ts +13 -5
- package/dist/observability/trace-attribution.js +315 -21
- package/dist/observability/trace-ingestion.d.ts +9 -0
- package/dist/observability/trace-ingestion.js +80 -0
- package/dist/observability/trace-ir.d.ts +113 -0
- package/dist/observability/trace-ir.js +87 -0
- package/dist/observability/trace-segmenter.d.ts +19 -6
- package/dist/observability/trace-segmenter.js +377 -196
- package/dist/observability/trace-session-index.d.ts +19 -0
- package/dist/observability/trace-session-index.js +68 -0
- package/dist/observability/trace-source.d.ts +12 -4
- package/dist/observability/trace-source.js +939 -215
- package/dist/renderer/html-renderer.js +37 -6
- package/dist/renderer/icons.js +3 -0
- package/dist/renderer/observation-inbox-renderer.js +208 -90
- package/dist/renderer/skill-detail-renderer.js +452 -109
- package/dist/renderer/skill-health-renderer.js +69 -12
- package/dist/renderer/summary.js +28 -7
- package/dist/renderer/table.js +21 -4
- package/dist/renderer/test-view.d.ts +1 -0
- package/dist/renderer/test-view.js +44 -9
- package/dist/server/indexed-report-store.js +14 -18
- package/dist/server/job-store.js +64 -26
- package/dist/server/report-server.js +190 -78
- package/dist/server/report-store.js +57 -80
- package/dist/server/skill-index.js +143 -49
- package/dist/server/skill-insights.js +44 -5
- package/dist/shared/artifact-graph.d.ts +3 -0
- package/dist/shared/artifact-graph.js +224 -0
- package/dist/shared/assertion-types.d.ts +8 -0
- package/dist/shared/assertion-types.js +46 -0
- package/dist/shared/atomic-json.d.ts +8 -0
- package/dist/shared/atomic-json.js +33 -0
- package/dist/shared/diagnosis-schema.d.ts +9 -0
- package/dist/shared/diagnosis-schema.js +181 -0
- package/dist/shared/doctor-report.d.ts +3 -0
- package/dist/shared/doctor-report.js +103 -0
- package/dist/shared/evaluation-job.d.ts +6 -0
- package/dist/shared/evaluation-job.js +217 -0
- package/dist/shared/executor-result.d.ts +17 -0
- package/dist/shared/executor-result.js +221 -0
- package/dist/shared/file-lock.d.ts +12 -0
- package/dist/shared/file-lock.js +129 -0
- package/dist/shared/json-value.d.ts +5 -0
- package/dist/shared/json-value.js +36 -0
- package/dist/shared/keyed-mutex.d.ts +7 -0
- package/dist/shared/keyed-mutex.js +24 -0
- package/dist/shared/record-count.d.ts +8 -0
- package/dist/shared/record-count.js +43 -0
- package/dist/shared/sample-contract.d.ts +3 -0
- package/dist/shared/sample-contract.js +332 -0
- package/dist/shared/timestamp.d.ts +6 -0
- package/dist/shared/timestamp.js +64 -0
- package/dist/shared/token-usage.d.ts +19 -0
- package/dist/shared/token-usage.js +50 -0
- package/dist/shared/tool-call-status.d.ts +8 -0
- package/dist/shared/tool-call-status.js +28 -0
- package/dist/shared/tool-identity.d.ts +21 -0
- package/dist/shared/tool-identity.js +84 -0
- package/dist/shared/tool-search.js +73 -16
- package/dist/shared/trace-projection.d.ts +5 -0
- package/dist/shared/trace-projection.js +20 -0
- package/dist/shared/trace-source-kind.d.ts +3 -0
- package/dist/shared/trace-source-kind.js +12 -0
- package/dist/types/diagnosis.d.ts +2 -0
- package/dist/types/eval.d.ts +4 -0
- package/dist/types/executor.d.ts +32 -5
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/judge.d.ts +2 -0
- package/dist/types/observability.d.ts +116 -9
- package/dist/types/report.d.ts +58 -6
- package/dist/types/skill-index.d.ts +7 -0
- package/dist/types/trace.d.ts +2 -0
- package/dist/types/trace.js +1 -0
- package/package.json +9 -5
|
@@ -11,7 +11,18 @@
|
|
|
11
11
|
* - 跨层共享类型集中在 types/,单一来源
|
|
12
12
|
*/
|
|
13
13
|
import type { DiagnosisBundle } from './diagnosis.js';
|
|
14
|
+
import type { ToolCallStatus } from './executor.js';
|
|
15
|
+
import type { TraceSourceKind } from './trace.js';
|
|
14
16
|
import type { SkillHardRule, SkillWorkflow } from '../shared/hard-rules.js';
|
|
17
|
+
export interface TraceIngestionSummary {
|
|
18
|
+
fileCount: number;
|
|
19
|
+
sourceRecordCount: number;
|
|
20
|
+
parsedRecordCount: number;
|
|
21
|
+
malformedRecordCount: number;
|
|
22
|
+
ignoredValueCount: number;
|
|
23
|
+
unknownEventCount: number;
|
|
24
|
+
filteredSessionCount: number;
|
|
25
|
+
}
|
|
15
26
|
export interface TraceSourceMetadata {
|
|
16
27
|
channel?: string;
|
|
17
28
|
sender?: string;
|
|
@@ -35,10 +46,12 @@ export interface ObservationReviewStateEntry {
|
|
|
35
46
|
metricKey?: ObservationMetricKey;
|
|
36
47
|
metricScope?: ObservationMetricScope;
|
|
37
48
|
metricScopeId?: string;
|
|
49
|
+
traceId?: string;
|
|
38
50
|
sourceTrace?: string;
|
|
39
51
|
sessionId?: string;
|
|
40
52
|
messageIndex?: number;
|
|
41
53
|
messageUuid?: string;
|
|
54
|
+
callInstanceId?: string;
|
|
42
55
|
toolUseId?: string;
|
|
43
56
|
snippet?: string;
|
|
44
57
|
}
|
|
@@ -57,10 +70,12 @@ export interface ObservationReviewStateUpdate {
|
|
|
57
70
|
metricKey?: ObservationMetricKey;
|
|
58
71
|
metricScope?: ObservationMetricScope;
|
|
59
72
|
metricScopeId?: string;
|
|
73
|
+
traceId?: string;
|
|
60
74
|
sourceTrace?: string;
|
|
61
75
|
sessionId?: string;
|
|
62
76
|
messageIndex?: number;
|
|
63
77
|
messageUuid?: string;
|
|
78
|
+
callInstanceId?: string;
|
|
64
79
|
toolUseId?: string;
|
|
65
80
|
snippet?: string;
|
|
66
81
|
}
|
|
@@ -69,12 +84,14 @@ export type ExperienceProblemSignal = 'user_correction' | 'negative_feedback' |
|
|
|
69
84
|
export interface ExperienceProblemEvidenceRef {
|
|
70
85
|
id: string;
|
|
71
86
|
kind: string;
|
|
87
|
+
traceId?: string;
|
|
72
88
|
sourceTrace: string;
|
|
73
89
|
sessionId: string;
|
|
74
90
|
messageIndex?: number;
|
|
75
91
|
logicalMessageIndex?: number;
|
|
76
92
|
sourceLineIndex?: number;
|
|
77
93
|
messageUuid?: string;
|
|
94
|
+
callInstanceId?: string;
|
|
78
95
|
toolUseId?: string;
|
|
79
96
|
timestamp?: string;
|
|
80
97
|
role?: 'user' | 'assistant' | 'tool' | 'other';
|
|
@@ -95,12 +112,14 @@ export interface ExperienceProblemPattern {
|
|
|
95
112
|
export interface ProblemTimelineEvent {
|
|
96
113
|
id: string;
|
|
97
114
|
kind: string;
|
|
115
|
+
traceId?: string;
|
|
98
116
|
sourceTrace: string;
|
|
99
117
|
sessionId: string;
|
|
100
118
|
messageIndex?: number;
|
|
101
119
|
logicalMessageIndex?: number;
|
|
102
120
|
sourceLineIndex?: number;
|
|
103
121
|
messageUuid?: string;
|
|
122
|
+
callInstanceId?: string;
|
|
104
123
|
toolUseId?: string;
|
|
105
124
|
timestamp?: string;
|
|
106
125
|
role?: 'user' | 'assistant' | 'tool' | 'other';
|
|
@@ -119,10 +138,15 @@ export interface SkillChainAdvisory {
|
|
|
119
138
|
shortLabel: string;
|
|
120
139
|
}
|
|
121
140
|
export type ObservationSignalType = 'failed_search' | 'repeated_failure' | 'hedging' | 'explicit_marker';
|
|
122
|
-
|
|
141
|
+
/** @deprecated Prefer TraceSourceKind for new source-neutral APIs. */
|
|
142
|
+
export type ObservationSourceKind = TraceSourceKind;
|
|
123
143
|
export type ObservationSeverityReasonCode = 'knowledge_gap_suspected' | 'repeated_failure_suspected' | 'explicit_gap_marker' | 'exploratory_probe' | 'skill_asset_unavailable' | 'soft_hedging_signal' | 'tool_or_runtime_noise';
|
|
124
144
|
export type ObservationSignalSubtype = 'hard_miss' | 'repeated_failure' | 'exploratory_miss' | 'tool_error' | 'permission_error' | 'bash_probe' | 'not_found' | 'transient_file_missing' | 'skill_asset_read_failed' | 'permission_denied' | 'tool_limit' | 'tool_failure' | 'regex_only' | 'llm_classified' | 'marker';
|
|
125
145
|
export interface ObservationEvidence {
|
|
146
|
+
traceId?: string;
|
|
147
|
+
sessionId?: string;
|
|
148
|
+
sourceTrace?: string;
|
|
149
|
+
sourceKind?: ObservationSourceKind;
|
|
126
150
|
tool?: string;
|
|
127
151
|
query?: string;
|
|
128
152
|
path?: string;
|
|
@@ -131,6 +155,7 @@ export interface ObservationEvidence {
|
|
|
131
155
|
markerToken?: string;
|
|
132
156
|
messageIndex?: number;
|
|
133
157
|
messageUuid?: string;
|
|
158
|
+
callInstanceId?: string;
|
|
134
159
|
toolUseId?: string;
|
|
135
160
|
segmentTimestamp?: string;
|
|
136
161
|
}
|
|
@@ -154,6 +179,8 @@ export interface ObservationInboxItem {
|
|
|
154
179
|
artifactHash?: string;
|
|
155
180
|
cwd?: string;
|
|
156
181
|
sessionId: string;
|
|
182
|
+
/** Physical evidence stream identity; unlike sessionId, unique across reused run ids. */
|
|
183
|
+
traceId?: string;
|
|
157
184
|
sourceTrace: string;
|
|
158
185
|
sourceKind: ObservationSourceKind;
|
|
159
186
|
signalType: ObservationSignalType;
|
|
@@ -168,11 +195,15 @@ export interface ObservationInboxItem {
|
|
|
168
195
|
firstSeen: string;
|
|
169
196
|
lastSeen: string;
|
|
170
197
|
occurrences: number;
|
|
198
|
+
/** Occurrences backed by an observed source timestamp. */
|
|
199
|
+
timestampedOccurrences?: number;
|
|
171
200
|
recentSessionIds: string[];
|
|
201
|
+
recentTraceIds?: string[];
|
|
172
202
|
representativeEvidence: ObservationEvidence[];
|
|
173
203
|
}
|
|
174
204
|
export interface ObservationSessionTimeRange {
|
|
175
205
|
sessionId: string;
|
|
206
|
+
traceId?: string;
|
|
176
207
|
sessionGroupId?: string;
|
|
177
208
|
sourceTrace: string;
|
|
178
209
|
sourceKind: ObservationSourceKind;
|
|
@@ -196,12 +227,15 @@ export interface ObservationInboxReport {
|
|
|
196
227
|
durationMs?: number;
|
|
197
228
|
};
|
|
198
229
|
sessionTimeRanges?: ObservationSessionTimeRange[];
|
|
230
|
+
ingestion?: TraceIngestionSummary;
|
|
199
231
|
segmentCount: number;
|
|
200
232
|
itemCount: number;
|
|
201
233
|
skillInvocationCounts?: Record<string, number>;
|
|
202
234
|
skillSessionCounts?: Record<string, number>;
|
|
203
235
|
skillInvocationLastSeen?: Record<string, string>;
|
|
204
236
|
skillToolCallCounts?: Record<string, Record<string, number>>;
|
|
237
|
+
timestampedSegmentCount?: number;
|
|
238
|
+
timestampCoverage?: number;
|
|
205
239
|
};
|
|
206
240
|
items: ObservationInboxItem[];
|
|
207
241
|
experience?: ObservationExperienceReport;
|
|
@@ -254,6 +288,7 @@ export type ExperienceRuleFindingCode = 'high_observation_seen' | 'medium_observ
|
|
|
254
288
|
export interface ExperienceEvidenceRef {
|
|
255
289
|
id: string;
|
|
256
290
|
kind: ExperienceEvidenceKind;
|
|
291
|
+
traceId?: string;
|
|
257
292
|
sourceTrace: string;
|
|
258
293
|
sessionId: string;
|
|
259
294
|
traceRole?: 'standalone' | 'main' | 'subagent';
|
|
@@ -262,6 +297,8 @@ export interface ExperienceEvidenceRef {
|
|
|
262
297
|
logicalMessageIndex?: number;
|
|
263
298
|
sourceLineIndex?: number;
|
|
264
299
|
messageUuid?: string;
|
|
300
|
+
/** Source-neutral identity for one concrete tool-call occurrence. */
|
|
301
|
+
callInstanceId?: string;
|
|
265
302
|
toolUseId?: string;
|
|
266
303
|
timestamp?: string;
|
|
267
304
|
role?: 'user' | 'assistant' | 'tool' | 'other';
|
|
@@ -271,17 +308,22 @@ export interface ExperienceEvidenceRef {
|
|
|
271
308
|
export interface ExperienceTimelineEvent extends ExperienceEvidenceRef {
|
|
272
309
|
order: number;
|
|
273
310
|
toolName?: string;
|
|
311
|
+
toolStatus?: ToolCallStatus;
|
|
274
312
|
isError?: boolean;
|
|
275
313
|
fullText?: string;
|
|
276
314
|
}
|
|
277
315
|
export interface ExperienceTimelineBranch {
|
|
278
316
|
id: string;
|
|
279
317
|
label: string;
|
|
318
|
+
sessionId: string;
|
|
319
|
+
traceId?: string;
|
|
280
320
|
sourceTrace: string;
|
|
281
321
|
traceRole: 'main' | 'subagent' | 'standalone';
|
|
282
322
|
attachTo?: {
|
|
323
|
+
traceId?: string;
|
|
283
324
|
sourceTrace: string;
|
|
284
325
|
messageIndex?: number;
|
|
326
|
+
callInstanceId?: string;
|
|
285
327
|
toolUseId?: string;
|
|
286
328
|
label?: string;
|
|
287
329
|
};
|
|
@@ -292,6 +334,20 @@ export interface ExperienceTimelineTree {
|
|
|
292
334
|
main: ExperienceTimelineEvent[];
|
|
293
335
|
branches: ExperienceTimelineBranch[];
|
|
294
336
|
}
|
|
337
|
+
export interface ExperienceTraceTimeline {
|
|
338
|
+
id: string;
|
|
339
|
+
sessionGroupKey: string;
|
|
340
|
+
sessionId: string;
|
|
341
|
+
eventCount: number;
|
|
342
|
+
tree: ExperienceTimelineTree;
|
|
343
|
+
}
|
|
344
|
+
export interface ExperienceTraceRecordRange {
|
|
345
|
+
traceId: string;
|
|
346
|
+
sourceTrace: string;
|
|
347
|
+
startRecordIndex: number;
|
|
348
|
+
endRecordIndex: number;
|
|
349
|
+
eventCount: number;
|
|
350
|
+
}
|
|
295
351
|
export interface ExperienceEvidenceChain {
|
|
296
352
|
userMessageCount: number;
|
|
297
353
|
runtimeContextCount: number;
|
|
@@ -391,10 +447,13 @@ export interface ExperienceSessionStorySubagentDispatch {
|
|
|
391
447
|
id: string;
|
|
392
448
|
order: number;
|
|
393
449
|
branchId: string;
|
|
450
|
+
childSessionId: string;
|
|
451
|
+
traceId: string;
|
|
394
452
|
label: string;
|
|
395
453
|
sourceTrace: string;
|
|
396
454
|
attachTo?: {
|
|
397
455
|
messageIndex?: number;
|
|
456
|
+
callInstanceId?: string;
|
|
398
457
|
toolUseId?: string;
|
|
399
458
|
label?: string;
|
|
400
459
|
};
|
|
@@ -419,6 +478,7 @@ export interface ExperienceGoalEvidenceRef {
|
|
|
419
478
|
export interface ExperienceMessageRange {
|
|
420
479
|
startMessageIndex: number;
|
|
421
480
|
endMessageIndex: number;
|
|
481
|
+
traceId?: string;
|
|
422
482
|
sourceTrace?: string;
|
|
423
483
|
sessionId?: string;
|
|
424
484
|
}
|
|
@@ -521,6 +581,7 @@ export interface ExperienceSessionStoryGraphEdge {
|
|
|
521
581
|
}
|
|
522
582
|
export interface ExperienceSessionStory {
|
|
523
583
|
schemaVersion: 1;
|
|
584
|
+
contextRef?: string;
|
|
524
585
|
summary: string;
|
|
525
586
|
invocationCount: number;
|
|
526
587
|
goalSliceCount: number;
|
|
@@ -539,6 +600,13 @@ export interface ExperienceSessionStory {
|
|
|
539
600
|
nodes: ExperienceSessionStoryNode[];
|
|
540
601
|
answers: ExperienceSessionStoryAnswer[];
|
|
541
602
|
}
|
|
603
|
+
export interface ExperienceStoryContext {
|
|
604
|
+
id: string;
|
|
605
|
+
sessionGroupKey: string;
|
|
606
|
+
goalSlices: ExperienceSessionStoryGoalSlice[];
|
|
607
|
+
subagentDispatches: ExperienceSessionStorySubagentDispatch[];
|
|
608
|
+
episodes: ExperienceEpisode[];
|
|
609
|
+
}
|
|
542
610
|
export interface ExperienceReviewerReport {
|
|
543
611
|
schemaVersion: 1;
|
|
544
612
|
mode: 'deterministic_milestone_1' | 'deterministic_session_story';
|
|
@@ -554,6 +622,8 @@ export interface ExperienceReviewerReport {
|
|
|
554
622
|
oneLookMetrics: {
|
|
555
623
|
toolCallCount: number;
|
|
556
624
|
toolFailureCount: number;
|
|
625
|
+
toolCancelledCount?: number;
|
|
626
|
+
toolUnknownCount?: number;
|
|
557
627
|
userMessageCount: number;
|
|
558
628
|
userFollowUpCount: number;
|
|
559
629
|
assistantDeliverySignalCount: number;
|
|
@@ -570,10 +640,15 @@ export interface ExperienceReviewerReport {
|
|
|
570
640
|
outputTokens: number;
|
|
571
641
|
cacheReadTokens: number;
|
|
572
642
|
cacheCreationTokens: number;
|
|
643
|
+
/** Missing on legacy reviewer reports; readers must treat that as unknown coverage. */
|
|
644
|
+
observedInvocationCount?: number;
|
|
645
|
+
invocationCount?: number;
|
|
646
|
+
coverage?: number;
|
|
573
647
|
attribution: 'skill_segment';
|
|
574
648
|
};
|
|
575
649
|
};
|
|
576
650
|
sessionStory: ExperienceSessionStory;
|
|
651
|
+
sessionStoryRef?: 'session';
|
|
577
652
|
authorSuggestions: string[];
|
|
578
653
|
traceLinks: ExperienceEvidenceRef[];
|
|
579
654
|
}
|
|
@@ -581,10 +656,12 @@ export interface ExperienceGoalSlice {
|
|
|
581
656
|
id: string;
|
|
582
657
|
skillName: string;
|
|
583
658
|
sessionId: string;
|
|
659
|
+
traceId?: string;
|
|
584
660
|
sourceTrace: string;
|
|
585
661
|
cwd?: string;
|
|
586
662
|
startTimestamp: string;
|
|
587
663
|
endTimestamp: string;
|
|
664
|
+
timestampObserved?: boolean;
|
|
588
665
|
sliceReasonCode: ExperienceGoalSliceReasonCode;
|
|
589
666
|
sliceConfidence: 'low' | 'medium' | 'high';
|
|
590
667
|
inferredUserGoal?: string;
|
|
@@ -608,6 +685,9 @@ export interface ExperienceReviewIndicators {
|
|
|
608
685
|
repeatedExecutionCount: number;
|
|
609
686
|
toolCallCount: number;
|
|
610
687
|
toolFailureCount: number;
|
|
688
|
+
toolCancelledCount?: number;
|
|
689
|
+
/** Runtime did not expose a trustworthy terminal outcome. */
|
|
690
|
+
toolUnknownCount?: number;
|
|
611
691
|
highObservationCount: number;
|
|
612
692
|
mediumObservationCount: number;
|
|
613
693
|
hedgingCount: number;
|
|
@@ -619,15 +699,21 @@ export interface ExperienceInvocationMetrics {
|
|
|
619
699
|
outputTokens: number;
|
|
620
700
|
cacheReadTokens: number;
|
|
621
701
|
cacheCreationTokens: number;
|
|
702
|
+
/** False means counters are placeholders because the trace exposed no valid usage event. */
|
|
703
|
+
tokenUsageObserved: boolean;
|
|
622
704
|
numTurns: number;
|
|
623
705
|
numToolCalls: number;
|
|
624
706
|
numToolFailures: number;
|
|
707
|
+
numToolCancelled?: number;
|
|
708
|
+
/** Unresolved or source-unknown outcomes; excluded from failure-rate denominators. */
|
|
709
|
+
numToolUnknown?: number;
|
|
625
710
|
}
|
|
626
711
|
export interface ExperienceInvocation {
|
|
627
712
|
id: string;
|
|
628
713
|
skillName: string;
|
|
629
714
|
sessionId: string;
|
|
630
715
|
sessionGroupKey: string;
|
|
716
|
+
traceId?: string;
|
|
631
717
|
sourceTrace: string;
|
|
632
718
|
sourceKind: ObservationSourceKind;
|
|
633
719
|
entrypoint?: string;
|
|
@@ -637,6 +723,7 @@ export interface ExperienceInvocation {
|
|
|
637
723
|
goalSliceId: string;
|
|
638
724
|
startTimestamp: string;
|
|
639
725
|
endTimestamp: string;
|
|
726
|
+
timestampObserved?: boolean;
|
|
640
727
|
attribution: {
|
|
641
728
|
source: string;
|
|
642
729
|
confidence: number;
|
|
@@ -653,6 +740,8 @@ export interface ExperienceInvocation {
|
|
|
653
740
|
problemPatterns: ExperienceProblemPattern[];
|
|
654
741
|
relatedObservationIds: string[];
|
|
655
742
|
evidenceRefs: ExperienceEvidenceRef[];
|
|
743
|
+
timelineRef?: string;
|
|
744
|
+
timelineEventIds?: string[];
|
|
656
745
|
timeline: ExperienceTimelineEvent[];
|
|
657
746
|
}
|
|
658
747
|
export interface ExperienceSessionSummary {
|
|
@@ -669,6 +758,8 @@ export interface ExperienceSessionSummary {
|
|
|
669
758
|
sourceSessionDurationMs?: number;
|
|
670
759
|
startTimestamp: string;
|
|
671
760
|
endTimestamp: string;
|
|
761
|
+
timestampedInvocationCount?: number;
|
|
762
|
+
timestampCoverage?: number;
|
|
672
763
|
invocationIds: string[];
|
|
673
764
|
goalSliceIds: string[];
|
|
674
765
|
reviewPriority: ExperienceReviewPriority;
|
|
@@ -680,22 +771,34 @@ export interface ExperienceSessionSummary {
|
|
|
680
771
|
assistiveInference: ExperienceAssistiveInference;
|
|
681
772
|
problemPatterns: ExperienceProblemPattern[];
|
|
682
773
|
relatedObservationIds: string[];
|
|
774
|
+
timelineRef?: string;
|
|
775
|
+
timelinePreviewEventIds?: string[];
|
|
683
776
|
timelinePreview: ExperienceTimelineEvent[];
|
|
684
777
|
fullSessionTimeline: ExperienceTimelineEvent[];
|
|
685
778
|
timelineTree?: ExperienceTimelineTree;
|
|
686
779
|
timelineScope: {
|
|
687
780
|
mode: 'skill_segment_window';
|
|
688
|
-
|
|
689
|
-
segmentEndRecordIndex?: number;
|
|
690
|
-
previewStartRecordIndex?: number;
|
|
691
|
-
previewEndRecordIndex?: number;
|
|
692
|
-
sessionStartRecordIndex: number;
|
|
693
|
-
sessionEndRecordIndex: number;
|
|
781
|
+
segmentEventCount: number;
|
|
694
782
|
previewEventCount: number;
|
|
695
783
|
fullSessionEventCount: number;
|
|
784
|
+
segmentRecordRanges: ExperienceTraceRecordRange[];
|
|
785
|
+
previewRecordRanges: ExperienceTraceRecordRange[];
|
|
786
|
+
sessionRecordRanges: ExperienceTraceRecordRange[];
|
|
696
787
|
truncated: boolean;
|
|
697
788
|
omittedBeforeCount: number;
|
|
698
789
|
omittedAfterCount: number;
|
|
790
|
+
/** @deprecated v2 only. Record indexes are local to one physical trace. */
|
|
791
|
+
segmentStartRecordIndex?: number;
|
|
792
|
+
/** @deprecated v2 only. Record indexes are local to one physical trace. */
|
|
793
|
+
segmentEndRecordIndex?: number;
|
|
794
|
+
/** @deprecated v2 only. Record indexes are local to one physical trace. */
|
|
795
|
+
previewStartRecordIndex?: number;
|
|
796
|
+
/** @deprecated v2 only. Record indexes are local to one physical trace. */
|
|
797
|
+
previewEndRecordIndex?: number;
|
|
798
|
+
/** @deprecated v2 only. Record indexes are local to one physical trace. */
|
|
799
|
+
sessionStartRecordIndex?: number;
|
|
800
|
+
/** @deprecated v2 only. Record indexes are local to one physical trace. */
|
|
801
|
+
sessionEndRecordIndex?: number;
|
|
699
802
|
};
|
|
700
803
|
attributionSources: string[];
|
|
701
804
|
pluginNames: string[];
|
|
@@ -725,6 +828,8 @@ export interface ExperienceSkillSummary {
|
|
|
725
828
|
toolCounts: Record<string, number>;
|
|
726
829
|
firstSeen: string;
|
|
727
830
|
lastSeen: string;
|
|
831
|
+
timestampedInvocationCount?: number;
|
|
832
|
+
timestampCoverage?: number;
|
|
728
833
|
reviewFirstSessionCount: number;
|
|
729
834
|
sampleReviewSessionCount: number;
|
|
730
835
|
indicators: ExperienceReviewIndicators;
|
|
@@ -736,7 +841,7 @@ export interface ExperienceSkillSummary {
|
|
|
736
841
|
}
|
|
737
842
|
export interface ObservationExperienceReport {
|
|
738
843
|
kind: 'observe-experience';
|
|
739
|
-
schemaVersion:
|
|
844
|
+
schemaVersion: 3;
|
|
740
845
|
scope: 'evidence-only';
|
|
741
846
|
generatedAt: string;
|
|
742
847
|
meta: {
|
|
@@ -747,6 +852,8 @@ export interface ObservationExperienceReport {
|
|
|
747
852
|
noteCodes: Array<'no_llm_judge' | 'no_auto_verdict' | 'default_goal_slice_is_allowed' | 'deterministic_assistive_inference'>;
|
|
748
853
|
};
|
|
749
854
|
goalSlices: ExperienceGoalSlice[];
|
|
855
|
+
traceTimelines: ExperienceTraceTimeline[];
|
|
856
|
+
storyContexts: ExperienceStoryContext[];
|
|
750
857
|
invocations: ExperienceInvocation[];
|
|
751
858
|
sessions: ExperienceSessionSummary[];
|
|
752
859
|
skills: ExperienceSkillSummary[];
|
|
@@ -763,7 +870,7 @@ export interface ObservationRuntimeCheck {
|
|
|
763
870
|
evidenceSnippets: string[];
|
|
764
871
|
}
|
|
765
872
|
export type SkillRuntimeEvidencePackSourceType = 'tool_call' | 'tool_result' | 'assistant_message' | 'user_feedback' | 'artifact' | 'runtime_context' | 'skill_context' | 'unknown';
|
|
766
|
-
export interface SkillRuntimeEvidencePackRef extends Pick<ExperienceEvidenceRef, 'id' | 'kind' | 'sourceTrace' | 'sessionId' | 'messageUuid' | 'messageIndex' | 'logicalMessageIndex' | 'sourceLineIndex' | 'toolUseId' | 'timestamp' | 'role' | 'label' | 'snippet'> {
|
|
873
|
+
export interface SkillRuntimeEvidencePackRef extends Pick<ExperienceEvidenceRef, 'id' | 'kind' | 'sourceTrace' | 'sessionId' | 'messageUuid' | 'messageIndex' | 'logicalMessageIndex' | 'sourceLineIndex' | 'callInstanceId' | 'toolUseId' | 'timestamp' | 'role' | 'label' | 'snippet'> {
|
|
767
874
|
sourceType: SkillRuntimeEvidencePackSourceType;
|
|
768
875
|
toolName?: string;
|
|
769
876
|
isError?: boolean;
|
package/dist/types/report.d.ts
CHANGED
|
@@ -10,11 +10,14 @@ export interface VariantResult {
|
|
|
10
10
|
totalTokens: number;
|
|
11
11
|
cacheReadTokens: number;
|
|
12
12
|
cacheCreationTokens: number;
|
|
13
|
+
/** Mirrors `ExecResult.tokenUsageReportedByExecutor`.
|
|
14
|
+
* False means token counters are placeholders for unavailable telemetry. */
|
|
15
|
+
tokenUsageReportedByExecutor?: boolean;
|
|
13
16
|
execCostUSD: number;
|
|
14
17
|
judgeCostUSD: number;
|
|
15
18
|
/** Diagnostic 自身花费(USD)。仅在 failed-assertion 触发 diagnostic 且 executor 报告了 cost 时有值。
|
|
16
|
-
* Diagnostic
|
|
17
|
-
* `judgeCostReportedByExecutor
|
|
19
|
+
* Diagnostic 跟随首位 judge executor,所以 cost-reported 语义汇入
|
|
20
|
+
* `judgeCostReportedByExecutor`,不再单独引一个 VariantResult flag。
|
|
18
21
|
* v0.30 新增,旧报告无此字段 — 老 costUSD 仍等于 execCostUSD + judgeCostUSD,新 costUSD
|
|
19
22
|
* 含 diagnostic,跨版本汇总时按字段是否存在判断。 */
|
|
20
23
|
diagnosticCostUSD?: number;
|
|
@@ -32,12 +35,18 @@ export interface VariantResult {
|
|
|
32
35
|
* report cost (currently codex). Default undefined ⇒ reported. */
|
|
33
36
|
judgeCostReportedByExecutor?: boolean;
|
|
34
37
|
numTurns: number;
|
|
38
|
+
/** Executor attempts used to obtain the final output. Missing means one. */
|
|
39
|
+
attemptCount?: number;
|
|
35
40
|
fullNumTurns?: number;
|
|
36
41
|
numSubAgents?: number;
|
|
37
42
|
assistantTurns?: number;
|
|
38
43
|
toolTurns?: number;
|
|
39
44
|
numToolCalls?: number;
|
|
40
45
|
numToolFailures?: number;
|
|
46
|
+
/** Calls explicitly cancelled by the runtime; distinct from execution failures. */
|
|
47
|
+
numToolCancelled?: number;
|
|
48
|
+
/** Calls without an authoritative completion state; excluded from success-rate denominator. */
|
|
49
|
+
numToolUnknown?: number;
|
|
41
50
|
toolSuccessRate?: number;
|
|
42
51
|
toolNames?: string[];
|
|
43
52
|
/** per-sample tool call distribution (tool name → call count).
|
|
@@ -93,6 +102,9 @@ export interface VariantResult {
|
|
|
93
102
|
timing?: {
|
|
94
103
|
execMs: number;
|
|
95
104
|
gradeMs: number;
|
|
105
|
+
/** Wall-clock time spent on the optional failure diagnostic call. */
|
|
106
|
+
diagnosticMs?: number;
|
|
107
|
+
/** execMs + gradeMs + (diagnosticMs ?? 0). */
|
|
96
108
|
totalMs: number;
|
|
97
109
|
};
|
|
98
110
|
}
|
|
@@ -105,6 +117,12 @@ export interface VariantSummary {
|
|
|
105
117
|
avgInputTokens: number;
|
|
106
118
|
avgOutputTokens: number;
|
|
107
119
|
avgTotalTokens: number;
|
|
120
|
+
/**
|
|
121
|
+
* Share of all sample results whose executor reported token usage.
|
|
122
|
+
* Missing means legacy reports where usage was assumed fully observed.
|
|
123
|
+
* Token averages use only successful results with observed usage.
|
|
124
|
+
*/
|
|
125
|
+
tokenUsageCoverageRate?: number;
|
|
108
126
|
/** sum(ok-sample 的 costUSD)。等于 totalExecCostUSD + totalJudgeCostUSD + totalDiagnosticCostUSD。
|
|
109
127
|
* 注意:仅含执行成功且未被 per-sample budget 标 overrun 的 sample,跟 meta.totalCostUSD(全量
|
|
110
128
|
* 累计,含失败 sample)语义不同 — 那是历史 ok-filter 行为,跟 v0.30 的 diagnostic 引入无关。 */
|
|
@@ -132,6 +150,8 @@ export interface VariantSummary {
|
|
|
132
150
|
avgToolTurns?: number;
|
|
133
151
|
avgToolCalls?: number;
|
|
134
152
|
avgToolFailures?: number;
|
|
153
|
+
avgToolCancelled?: number;
|
|
154
|
+
avgToolUnknown?: number;
|
|
135
155
|
toolSuccessRate?: number;
|
|
136
156
|
toolDistribution?: Record<string, number>;
|
|
137
157
|
traceCoverageRate?: number;
|
|
@@ -264,14 +284,28 @@ export interface ReportMeta {
|
|
|
264
284
|
* Value 2 marked the artifactHashes tree-hash era for local dir-skills (git dir-skills still
|
|
265
285
|
* SKILL.md-only). Value 3 extends whole-tree hashing to git dir-skills via isolated copies
|
|
266
286
|
* (all dir-skills bind). Value 4 keeps the v3 hash/binding semantics but marks the canonical
|
|
267
|
-
* top-level discriminant era, so external consumers can version-gate the JSON shape.
|
|
268
|
-
*
|
|
269
|
-
*
|
|
287
|
+
* top-level discriminant era, so external consumers can version-gate the JSON shape. Value 5
|
|
288
|
+
* changes `sampleHashes` to bind the complete Sample contract plus external mock fixtures and
|
|
289
|
+
* the statically resolvable ESM custom-assertion import graph. Drift / lineage consumers gate on `>= 2`
|
|
290
|
+
* (tree-hash era); git-dir-skill binding additionally requires `>= 3`; complete sample
|
|
291
|
+
* fingerprints require `>= 5`. */
|
|
270
292
|
schemaVersion?: number;
|
|
271
|
-
/** SHA256-12 of every sample's
|
|
293
|
+
/** SHA256-12 of every sample's measurement contract (sample_id → hash).
|
|
294
|
+
* schemaVersion >= 5 binds all inline fields, mock return_file content, and the statically
|
|
295
|
+
* resolvable ESM custom-assertion import graph. Earlier versions used a partial inline-field
|
|
296
|
+
* hash. */
|
|
272
297
|
sampleHashes?: Record<string, string>;
|
|
273
298
|
/** SHA256-12 of the LLM judge prompt template. Different hash = judge changed semantics. */
|
|
274
299
|
judgePromptHash?: string;
|
|
300
|
+
/** Failure-diagnostic execution contract. Diagnostic is independent from judge scoring,
|
|
301
|
+
* so it remains configured under --no-judge unless --no-diagnostic is set. */
|
|
302
|
+
diagnostic?: {
|
|
303
|
+
enabled: boolean;
|
|
304
|
+
executor?: string;
|
|
305
|
+
model?: string;
|
|
306
|
+
runtime?: ExecutorRuntimeFingerprint;
|
|
307
|
+
promptHash?: string;
|
|
308
|
+
};
|
|
275
309
|
/** Runtime fingerprint for the executor that produced tested outputs.
|
|
276
310
|
* Legacy/common field. Prefer executorRuntimes for variant-level audit; this
|
|
277
311
|
* field is the common runtime when all variants match, otherwise a representative
|
|
@@ -331,6 +365,17 @@ export interface ReportMeta {
|
|
|
331
365
|
evolve?: {
|
|
332
366
|
skillName: string;
|
|
333
367
|
skillPath?: string;
|
|
368
|
+
/** 本次 evolve 实际发生的改写、修复与评测总成本;不同于报告内 result 的测量成本。 */
|
|
369
|
+
processCostUSD?: number;
|
|
370
|
+
/** false 表示 processCostUSD 只是可观测 provider 成本的下界。 */
|
|
371
|
+
processCostReported?: boolean;
|
|
372
|
+
/** 每个 round 的原始评测身份。合并报告是 aggregate,不复用任一源报告的 request / run / job。 */
|
|
373
|
+
sourceReports?: Array<{
|
|
374
|
+
round: number;
|
|
375
|
+
accepted: boolean;
|
|
376
|
+
reportId: string;
|
|
377
|
+
variant: string;
|
|
378
|
+
}>;
|
|
334
379
|
};
|
|
335
380
|
}
|
|
336
381
|
export interface ResultEntry {
|
|
@@ -354,10 +399,17 @@ export interface ResultEntry {
|
|
|
354
399
|
export interface SampleSnapshot {
|
|
355
400
|
sample_id: string;
|
|
356
401
|
prompt: string;
|
|
402
|
+
/** Runtime context that can change executor-visible project state. */
|
|
403
|
+
cwd?: string;
|
|
357
404
|
rubric?: string;
|
|
358
405
|
context?: string;
|
|
406
|
+
dimensions?: Record<string, string>;
|
|
359
407
|
assertions?: import('./eval.js').Assertion[];
|
|
360
408
|
mocks?: import('./eval.js').Mock[];
|
|
409
|
+
mocksStrict?: boolean;
|
|
410
|
+
environment?: import('./eval.js').SampleEnvironment;
|
|
411
|
+
allowedTools?: string[];
|
|
412
|
+
expectedTools?: string[];
|
|
361
413
|
capability?: string[];
|
|
362
414
|
difficulty?: import('./eval.js').SampleDifficulty;
|
|
363
415
|
construct?: string;
|
|
@@ -36,8 +36,13 @@ export interface SkillObserveSnapshot {
|
|
|
36
36
|
generatedAt: string;
|
|
37
37
|
healthBand: 'green' | 'yellow' | 'red';
|
|
38
38
|
failureRate: number;
|
|
39
|
+
toolCallCount?: number;
|
|
40
|
+
toolResolvedCount?: number;
|
|
41
|
+
toolCancelledCount?: number;
|
|
42
|
+
toolUnknownCount?: number;
|
|
39
43
|
segmentCount: number;
|
|
40
44
|
gapRate: number;
|
|
45
|
+
stability?: 'stable' | 'unstable' | 'very-unstable' | 'unknown';
|
|
41
46
|
/** 统计可信度(按 segment 数)。underpowered 时下游 insight / card 不应触发硬红或 high severity。
|
|
42
47
|
* 历史快照缺此字段时由 segmentCount 兜底推导。 */
|
|
43
48
|
confidence: 'high' | 'low' | 'underpowered';
|
|
@@ -56,6 +61,8 @@ export interface SkillGraphNodePreview {
|
|
|
56
61
|
nodeKind: string;
|
|
57
62
|
label: string;
|
|
58
63
|
status?: string;
|
|
64
|
+
/** Studio evidence view 用于把 assertion / diagnostic 归到对应 sample。 */
|
|
65
|
+
parentSampleStableKey?: string;
|
|
59
66
|
coverage?: 'declared' | 'undeclared';
|
|
60
67
|
coveredBySamples?: string[];
|
|
61
68
|
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "oh-my-knowledge",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.49.0",
|
|
4
4
|
"packageManager": "yarn@4.16.0",
|
|
5
5
|
"description": "Evaluation framework for LLM knowledge inputs — prompts, RAG corpora, skills, agent workflows. Fix the model, vary the artifact. Built-in statistical rigor: bootstrap CI, Krippendorff α, length-debias, saturation curves.",
|
|
6
6
|
"type": "module",
|
|
@@ -76,6 +76,9 @@
|
|
|
76
76
|
"skill-evaluation",
|
|
77
77
|
"knowledge-engineering",
|
|
78
78
|
"prompt-engineering",
|
|
79
|
+
"codex",
|
|
80
|
+
"codex-cli",
|
|
81
|
+
"openai",
|
|
79
82
|
"claude-code",
|
|
80
83
|
"claude",
|
|
81
84
|
"llm",
|
|
@@ -93,13 +96,14 @@
|
|
|
93
96
|
"license": "MIT",
|
|
94
97
|
"dependencies": {
|
|
95
98
|
"@anthropic-ai/claude-agent-sdk": "^0.3.143",
|
|
96
|
-
"@anthropic-ai/sdk": "^0.
|
|
99
|
+
"@anthropic-ai/sdk": "^0.112.3",
|
|
97
100
|
"@inquirer/prompts": "^8.4.3",
|
|
98
101
|
"@modelcontextprotocol/sdk": "^1.29.0",
|
|
99
102
|
"@oclif/core": "^4",
|
|
100
|
-
"@openai/codex-sdk": "0.
|
|
103
|
+
"@openai/codex-sdk": "0.144.6",
|
|
101
104
|
"ajv": "^8.18.0",
|
|
102
105
|
"chart.js": "^4.5.1",
|
|
106
|
+
"es-module-lexer": "^2.0.0",
|
|
103
107
|
"js-yaml": "^4.1.1",
|
|
104
108
|
"simple-statistics": "^7.8.9",
|
|
105
109
|
"zod": "^4.4.3"
|
|
@@ -112,11 +116,11 @@
|
|
|
112
116
|
"@types/node": "^25.5.0",
|
|
113
117
|
"eslint": "^10.1.0",
|
|
114
118
|
"husky": "^9.1.7",
|
|
115
|
-
"lint-staged": "17.0
|
|
119
|
+
"lint-staged": "17.1.0",
|
|
116
120
|
"npm-run-all2": "^9.0.1",
|
|
117
121
|
"typescript": "^6.0.2",
|
|
118
122
|
"typescript-eslint": "^8.58.0",
|
|
119
123
|
"vitepress": "^1.6.4",
|
|
120
|
-
"vitest": "4.1.
|
|
124
|
+
"vitest": "4.1.10"
|
|
121
125
|
}
|
|
122
126
|
}
|