oh-my-knowledge 0.42.0 → 0.44.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/README.zh.md +1 -1
- package/dist/artifact-graph/doctor.d.ts +21 -0
- package/dist/artifact-graph/doctor.js +569 -0
- package/dist/artifact-graph/eval.d.ts +16 -0
- package/dist/artifact-graph/eval.js +351 -0
- package/dist/assets/agent-skills/omk/SKILL.md +9 -9
- package/dist/assets/agent-skills/omk/references/commands.md +1 -1
- package/dist/cli/commands/doctor.js +62 -61
- package/dist/cli/commands/eval/index.js +20 -2
- package/dist/cli/commands/init.js +1 -0
- package/dist/cli/commands/observe/index.d.ts +2 -2
- package/dist/cli/commands/observe/index.js +8 -7
- package/dist/cli/commands/sample.js +22 -16
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +8 -0
- package/dist/cli/lib/i18n-dict/help.js +12 -10
- package/dist/cli/lib/parse-run-config/samples-discovery.d.ts +5 -7
- package/dist/cli/lib/parse-run-config/samples-discovery.js +10 -32
- package/dist/cli/lib/parse-run-config.js +4 -4
- package/dist/cli/lib/resolve-skill-input.js +10 -12
- package/dist/doctor/index.js +2 -7
- package/dist/doctor/messages.js +2 -2
- package/dist/eval-core/artifact-file-names.d.ts +15 -0
- package/dist/eval-core/artifact-file-names.js +45 -0
- package/dist/eval-core/evaluation-reporting.d.ts +1 -1
- package/dist/eval-core/evaluation-reporting.js +32 -12
- package/dist/eval-core/measurement-dirs.js +13 -7
- package/dist/eval-core/report-file-migration.d.ts +10 -0
- package/dist/eval-core/report-file-migration.js +90 -0
- package/dist/inputs/sample-locator.d.ts +23 -0
- package/dist/inputs/sample-locator.js +195 -0
- package/dist/inputs/skill-loader.js +7 -17
- package/dist/observability/inbox.js +7 -3
- package/dist/renderer/html-renderer.js +3 -2
- package/dist/renderer/skill-detail-renderer.js +832 -0
- package/dist/renderer/skill-list-renderer.js +2 -7
- package/dist/server/report-server.js +55 -15
- package/dist/server/report-store.js +17 -9
- package/dist/server/skill-index.d.ts +6 -0
- package/dist/server/skill-index.js +399 -15
- package/dist/types/artifact-graph.d.ts +93 -0
- package/dist/types/artifact-graph.js +1 -0
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/skill-index.d.ts +38 -0
- package/package.json +1 -1
|
@@ -0,0 +1,351 @@
|
|
|
1
|
+
import { createHash } from 'node:crypto';
|
|
2
|
+
import { existsSync, mkdirSync, writeFileSync } from 'node:fs';
|
|
3
|
+
import { basename, dirname, join } from 'node:path';
|
|
4
|
+
import { graphFileName } from '../eval-core/artifact-file-names.js';
|
|
5
|
+
function shortHash(input) {
|
|
6
|
+
return createHash('sha256').update(input).digest('hex').slice(0, 12);
|
|
7
|
+
}
|
|
8
|
+
function jsonPointerToken(value) {
|
|
9
|
+
return value.replaceAll('~', '~0').replaceAll('/', '~1');
|
|
10
|
+
}
|
|
11
|
+
function sampleSetHash(sampleHashes) {
|
|
12
|
+
if (!sampleHashes || Object.keys(sampleHashes).length === 0)
|
|
13
|
+
return undefined;
|
|
14
|
+
const canonical = Object.entries(sampleHashes)
|
|
15
|
+
.sort(([a], [b]) => a.localeCompare(b))
|
|
16
|
+
.map(([id, hash]) => `${id}:${hash}`)
|
|
17
|
+
.join('|');
|
|
18
|
+
return shortHash(canonical);
|
|
19
|
+
}
|
|
20
|
+
function statusFromScore(score) {
|
|
21
|
+
// Display band only. This is not the verdict gate and intentionally does not
|
|
22
|
+
// reference DEFAULT_GATE_THRESHOLD or statistical significance decisions.
|
|
23
|
+
if (score === undefined || !Number.isFinite(score))
|
|
24
|
+
return 'unknown';
|
|
25
|
+
if (score >= 4)
|
|
26
|
+
return 'ok';
|
|
27
|
+
if (score >= 3)
|
|
28
|
+
return 'warning';
|
|
29
|
+
return 'failed';
|
|
30
|
+
}
|
|
31
|
+
function assertionStatus(result) {
|
|
32
|
+
// Assertion topology status only. Pure LLM-scored samples have no assertion
|
|
33
|
+
// pass/fail edge, so they stay unknown here even when compositeScore is high.
|
|
34
|
+
if (result.error || !result.ok)
|
|
35
|
+
return 'failed';
|
|
36
|
+
const details = result.assertions?.details;
|
|
37
|
+
if (!details || details.length === 0)
|
|
38
|
+
return 'unknown';
|
|
39
|
+
return details.every((detail) => detail.passed) ? 'ok' : 'failed';
|
|
40
|
+
}
|
|
41
|
+
function artifactBinding(variant, hash) {
|
|
42
|
+
if (hash && hash !== 'no-skill') {
|
|
43
|
+
return { bindingStrength: 'content-hash', keys: { artifactHash: hash } };
|
|
44
|
+
}
|
|
45
|
+
return { bindingStrength: 'name-only', keys: { variantName: variant } };
|
|
46
|
+
}
|
|
47
|
+
function sampleBinding(sampleId, hash) {
|
|
48
|
+
if (hash) {
|
|
49
|
+
return { bindingStrength: 'content-hash', keys: { sampleHash: hash } };
|
|
50
|
+
}
|
|
51
|
+
return { bindingStrength: 'name-only', keys: { sampleId } };
|
|
52
|
+
}
|
|
53
|
+
function variantConfigByName(report) {
|
|
54
|
+
return new Map((report.meta.variantConfigs ?? []).map((config) => [config.variant, config]));
|
|
55
|
+
}
|
|
56
|
+
function scopeArtifactKind(report) {
|
|
57
|
+
const kinds = new Set((report.meta.variantConfigs ?? [])
|
|
58
|
+
.map((config) => config.artifactKind)
|
|
59
|
+
.filter((kind) => kind !== 'baseline'));
|
|
60
|
+
return kinds.size === 1 ? [...kinds][0] : undefined;
|
|
61
|
+
}
|
|
62
|
+
function sampleEvidence(report, sampleId) {
|
|
63
|
+
return [{
|
|
64
|
+
sourceKind: 'sample',
|
|
65
|
+
sourceId: sampleId,
|
|
66
|
+
selector: { selectorKind: 'sample-id', value: sampleId },
|
|
67
|
+
contentHash: report.meta.sampleHashes?.[sampleId],
|
|
68
|
+
label: sampleId,
|
|
69
|
+
}];
|
|
70
|
+
}
|
|
71
|
+
function evalResultEvidence(report, resultIndex, variant) {
|
|
72
|
+
return [{
|
|
73
|
+
sourceKind: 'eval-report',
|
|
74
|
+
sourceId: report.id,
|
|
75
|
+
selector: {
|
|
76
|
+
selectorKind: 'json-pointer',
|
|
77
|
+
value: `/results/${resultIndex}/variants/${jsonPointerToken(variant)}`,
|
|
78
|
+
},
|
|
79
|
+
label: `${variant} result`,
|
|
80
|
+
}];
|
|
81
|
+
}
|
|
82
|
+
function assertionEvidence(report, sampleId, index, resultIndex, variant) {
|
|
83
|
+
if (report.sampleSnapshots?.[sampleId]?.assertions?.[index]) {
|
|
84
|
+
return [{
|
|
85
|
+
sourceKind: 'sample',
|
|
86
|
+
sourceId: sampleId,
|
|
87
|
+
selector: {
|
|
88
|
+
selectorKind: 'json-pointer',
|
|
89
|
+
value: `/sampleSnapshots/${jsonPointerToken(sampleId)}/assertions/${index}`,
|
|
90
|
+
},
|
|
91
|
+
contentHash: report.meta.sampleHashes?.[sampleId],
|
|
92
|
+
label: `assertion ${index + 1}`,
|
|
93
|
+
}];
|
|
94
|
+
}
|
|
95
|
+
if (resultIndex !== undefined && variant !== undefined) {
|
|
96
|
+
return [{
|
|
97
|
+
sourceKind: 'eval-report',
|
|
98
|
+
sourceId: report.id,
|
|
99
|
+
selector: {
|
|
100
|
+
selectorKind: 'json-pointer',
|
|
101
|
+
value: `/results/${resultIndex}/variants/${jsonPointerToken(variant)}/assertions/details/${index}`,
|
|
102
|
+
},
|
|
103
|
+
label: `${variant} assertion ${index + 1}`,
|
|
104
|
+
}];
|
|
105
|
+
}
|
|
106
|
+
return [{
|
|
107
|
+
sourceKind: 'sample',
|
|
108
|
+
sourceId: sampleId,
|
|
109
|
+
contentHash: report.meta.sampleHashes?.[sampleId],
|
|
110
|
+
label: `assertion ${index + 1}`,
|
|
111
|
+
}];
|
|
112
|
+
}
|
|
113
|
+
function sampleStableKey(report, sampleId) {
|
|
114
|
+
const hash = report.meta.sampleHashes?.[sampleId];
|
|
115
|
+
return hash ? `v1:sample:${hash}` : `v1:sample:${report.id}:${sampleId}`;
|
|
116
|
+
}
|
|
117
|
+
function assertionStableKey(report, sampleId, index) {
|
|
118
|
+
return `${sampleStableKey(report, sampleId)}:assertion:${index}`;
|
|
119
|
+
}
|
|
120
|
+
function sampleAttrs(snapshot) {
|
|
121
|
+
if (!snapshot)
|
|
122
|
+
return undefined;
|
|
123
|
+
const display = {};
|
|
124
|
+
if (snapshot.capability?.length)
|
|
125
|
+
display.capability = snapshot.capability;
|
|
126
|
+
if (snapshot.construct)
|
|
127
|
+
display.construct = snapshot.construct;
|
|
128
|
+
if (snapshot.difficulty)
|
|
129
|
+
display.difficulty = snapshot.difficulty;
|
|
130
|
+
if (snapshot.provenance)
|
|
131
|
+
display.provenance = snapshot.provenance;
|
|
132
|
+
if (snapshot.tripwire)
|
|
133
|
+
display.tripwire = true;
|
|
134
|
+
if (snapshot.assertions?.length)
|
|
135
|
+
display.assertionCount = snapshot.assertions.length;
|
|
136
|
+
return Object.keys(display).length > 0 ? { display } : undefined;
|
|
137
|
+
}
|
|
138
|
+
export function evalGraphDirForReportOutput(reportOutputDir) {
|
|
139
|
+
return basename(reportOutputDir) === 'reports'
|
|
140
|
+
? join(dirname(reportOutputDir), 'graphs', 'eval')
|
|
141
|
+
: join(reportOutputDir, 'graphs', 'eval');
|
|
142
|
+
}
|
|
143
|
+
export function buildEvalArtifactGraph(options) {
|
|
144
|
+
const { report, sourcePath } = options;
|
|
145
|
+
const generatedAt = options.generatedAt ?? new Date().toISOString();
|
|
146
|
+
const nodes = [];
|
|
147
|
+
const edges = [];
|
|
148
|
+
const nodeIdsByStableKey = new Map();
|
|
149
|
+
const configs = variantConfigByName(report);
|
|
150
|
+
const addNode = (stableKey, nodeKind, nodeRole, label, extra = {}) => {
|
|
151
|
+
const existing = nodeIdsByStableKey.get(stableKey);
|
|
152
|
+
if (existing)
|
|
153
|
+
return existing;
|
|
154
|
+
const id = `node:${shortHash(stableKey)}`;
|
|
155
|
+
nodeIdsByStableKey.set(stableKey, id);
|
|
156
|
+
nodes.push({
|
|
157
|
+
id,
|
|
158
|
+
stableKey,
|
|
159
|
+
nodeKind,
|
|
160
|
+
nodeRole,
|
|
161
|
+
layer: 'measurement',
|
|
162
|
+
label,
|
|
163
|
+
...extra,
|
|
164
|
+
});
|
|
165
|
+
return id;
|
|
166
|
+
};
|
|
167
|
+
const addEdge = (fromNodeId, toNodeId, edgeKind, extra = {}) => {
|
|
168
|
+
const id = `edge:${shortHash(`${fromNodeId}|${edgeKind}|${toNodeId}|${edges.length}`)}`;
|
|
169
|
+
edges.push({
|
|
170
|
+
id,
|
|
171
|
+
fromNodeId,
|
|
172
|
+
toNodeId,
|
|
173
|
+
edgeKind,
|
|
174
|
+
layer: 'measurement',
|
|
175
|
+
...extra,
|
|
176
|
+
});
|
|
177
|
+
};
|
|
178
|
+
const variantNodeIds = new Map();
|
|
179
|
+
for (const variant of report.meta.variants) {
|
|
180
|
+
const artifactHash = report.meta.artifactHashes?.[variant];
|
|
181
|
+
const config = configs.get(variant);
|
|
182
|
+
const variantNodeId = addNode(`v1:variant:${report.id}:${variant}`, 'variant', 'entity', variant, {
|
|
183
|
+
status: statusFromScore(report.summary?.[variant]?.avgCompositeScore),
|
|
184
|
+
binding: artifactBinding(variant, artifactHash),
|
|
185
|
+
metrics: {
|
|
186
|
+
...(report.summary?.[variant]?.avgCompositeScore !== undefined
|
|
187
|
+
? { avgCompositeScore: report.summary[variant].avgCompositeScore }
|
|
188
|
+
: {}),
|
|
189
|
+
...(report.summary?.[variant]?.totalSamples !== undefined
|
|
190
|
+
? { totalSamples: report.summary[variant].totalSamples }
|
|
191
|
+
: {}),
|
|
192
|
+
},
|
|
193
|
+
attrs: {
|
|
194
|
+
display: {
|
|
195
|
+
...(config ? {
|
|
196
|
+
artifactKind: config.artifactKind,
|
|
197
|
+
artifactSource: config.artifactSource,
|
|
198
|
+
experimentRole: config.experimentRole,
|
|
199
|
+
executionStrategy: config.executionStrategy,
|
|
200
|
+
} : {}),
|
|
201
|
+
},
|
|
202
|
+
},
|
|
203
|
+
evidenceRefs: [{
|
|
204
|
+
sourceKind: 'eval-report',
|
|
205
|
+
sourceId: report.id,
|
|
206
|
+
selector: { selectorKind: 'json-pointer', value: `/summary/${jsonPointerToken(variant)}` },
|
|
207
|
+
label: `${variant} summary`,
|
|
208
|
+
}],
|
|
209
|
+
});
|
|
210
|
+
variantNodeIds.set(variant, variantNodeId);
|
|
211
|
+
if (config?.artifactKind === 'skill' && artifactHash && artifactHash !== 'no-skill') {
|
|
212
|
+
const skillNodeId = addNode(`v1:skill:${artifactHash}`, 'skill', 'entity', variant, {
|
|
213
|
+
binding: { bindingStrength: 'content-hash', keys: { artifactHash } },
|
|
214
|
+
attrs: {
|
|
215
|
+
display: {
|
|
216
|
+
variant,
|
|
217
|
+
sourceLocator: config.locator,
|
|
218
|
+
},
|
|
219
|
+
},
|
|
220
|
+
evidenceRefs: [{
|
|
221
|
+
sourceKind: 'eval-report',
|
|
222
|
+
sourceId: report.id,
|
|
223
|
+
selector: { selectorKind: 'json-pointer', value: `/meta/artifactHashes/${jsonPointerToken(variant)}` },
|
|
224
|
+
contentHash: artifactHash,
|
|
225
|
+
label: `${variant} artifact hash`,
|
|
226
|
+
}],
|
|
227
|
+
});
|
|
228
|
+
addEdge(variantNodeId, skillNodeId, 'derived_from');
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
for (const [sampleId, snapshot] of Object.entries(report.sampleSnapshots ?? {})) {
|
|
232
|
+
const sampleNodeId = addNode(sampleStableKey(report, sampleId), 'sample', 'entity', sampleId, {
|
|
233
|
+
binding: sampleBinding(sampleId, report.meta.sampleHashes?.[sampleId]),
|
|
234
|
+
attrs: sampleAttrs(snapshot),
|
|
235
|
+
evidenceRefs: sampleEvidence(report, sampleId),
|
|
236
|
+
});
|
|
237
|
+
snapshot.assertions?.forEach((assertion, index) => {
|
|
238
|
+
const assertionNodeId = addNode(assertionStableKey(report, sampleId, index), 'assertion', 'entity', `assertion: ${assertion.type}`, {
|
|
239
|
+
attrs: { display: { type: assertion.type, weight: assertion.weight ?? 1 } },
|
|
240
|
+
evidenceRefs: assertionEvidence(report, sampleId, index),
|
|
241
|
+
});
|
|
242
|
+
addEdge(sampleNodeId, assertionNodeId, 'contains');
|
|
243
|
+
});
|
|
244
|
+
}
|
|
245
|
+
for (const [resultIndex, result] of report.results.entries()) {
|
|
246
|
+
const sampleNodeId = addNode(sampleStableKey(report, result.sample_id), 'sample', 'entity', result.sample_id, {
|
|
247
|
+
binding: sampleBinding(result.sample_id, report.meta.sampleHashes?.[result.sample_id]),
|
|
248
|
+
attrs: sampleAttrs(report.sampleSnapshots?.[result.sample_id]),
|
|
249
|
+
evidenceRefs: sampleEvidence(report, result.sample_id),
|
|
250
|
+
});
|
|
251
|
+
for (const [variant, variantResult] of Object.entries(result.variants)) {
|
|
252
|
+
const variantNodeId = variantNodeIds.get(variant);
|
|
253
|
+
if (!variantNodeId)
|
|
254
|
+
continue;
|
|
255
|
+
addEdge(variantNodeId, sampleNodeId, 'evaluates', {
|
|
256
|
+
status: assertionStatus(variantResult),
|
|
257
|
+
evidenceRefs: evalResultEvidence(report, resultIndex, variant),
|
|
258
|
+
});
|
|
259
|
+
const evalResultNodeId = addNode(`v1:eval-result:${report.id}:${variant}:${result.sample_id}`, 'eval_result', 'observation', `${variant} / ${result.sample_id}`, {
|
|
260
|
+
status: assertionStatus(variantResult),
|
|
261
|
+
metrics: {
|
|
262
|
+
durationMs: variantResult.durationMs,
|
|
263
|
+
costUSD: variantResult.costUSD,
|
|
264
|
+
...(variantResult.compositeScore !== undefined ? { compositeScore: variantResult.compositeScore } : {}),
|
|
265
|
+
...(variantResult.llmScore !== undefined ? { llmScore: variantResult.llmScore } : {}),
|
|
266
|
+
...(variantResult.assertions ? { assertionScore: variantResult.assertions.score } : {}),
|
|
267
|
+
},
|
|
268
|
+
attrs: {
|
|
269
|
+
display: {
|
|
270
|
+
ok: variantResult.ok,
|
|
271
|
+
...(variantResult.error ? { error: variantResult.error } : {}),
|
|
272
|
+
},
|
|
273
|
+
},
|
|
274
|
+
evidenceRefs: evalResultEvidence(report, resultIndex, variant),
|
|
275
|
+
});
|
|
276
|
+
addEdge(evalResultNodeId, variantNodeId, 'derived_from');
|
|
277
|
+
addEdge(evalResultNodeId, sampleNodeId, 'evaluates');
|
|
278
|
+
variantResult.assertions?.details.forEach((detail, index) => {
|
|
279
|
+
const assertionNodeId = addNode(assertionStableKey(report, result.sample_id, index), 'assertion', 'entity', `assertion: ${detail.type}`, {
|
|
280
|
+
attrs: { display: { type: detail.type, weight: detail.weight } },
|
|
281
|
+
evidenceRefs: assertionEvidence(report, result.sample_id, index, resultIndex, variant),
|
|
282
|
+
});
|
|
283
|
+
addEdge(evalResultNodeId, assertionNodeId, detail.passed ? 'passes' : 'fails', {
|
|
284
|
+
status: detail.passed ? 'ok' : 'failed',
|
|
285
|
+
evidenceRefs: evalResultEvidence(report, resultIndex, variant),
|
|
286
|
+
});
|
|
287
|
+
});
|
|
288
|
+
for (const [dimension, dimensionResult] of Object.entries(variantResult.dimensions ?? {})) {
|
|
289
|
+
const dimensionNodeId = addNode(`v1:judge-dimension:${report.id}:${variant}:${result.sample_id}:${dimension}`, 'judge_dimension', 'observation', dimension, {
|
|
290
|
+
status: statusFromScore(dimensionResult.score),
|
|
291
|
+
metrics: { score: dimensionResult.score },
|
|
292
|
+
attrs: { display: { reason: dimensionResult.reason } },
|
|
293
|
+
evidenceRefs: evalResultEvidence(report, resultIndex, variant),
|
|
294
|
+
});
|
|
295
|
+
addEdge(dimensionNodeId, evalResultNodeId, 'derived_from');
|
|
296
|
+
}
|
|
297
|
+
if (variantResult.diagnostic) {
|
|
298
|
+
const diagnosticNodeId = addNode(`v1:diagnostic:${report.id}:${variant}:${result.sample_id}`, 'diagnostic', 'observation', `diagnostic: ${variant} / ${result.sample_id}`, {
|
|
299
|
+
status: variantResult.diagnostic.ok ? 'warning' : 'failed',
|
|
300
|
+
attrs: {
|
|
301
|
+
display: {
|
|
302
|
+
rootCause: variantResult.diagnostic.rootCause,
|
|
303
|
+
failureModes: variantResult.diagnostic.failureModes ?? [],
|
|
304
|
+
},
|
|
305
|
+
},
|
|
306
|
+
evidenceRefs: evalResultEvidence(report, resultIndex, variant),
|
|
307
|
+
});
|
|
308
|
+
addEdge(diagnosticNodeId, evalResultNodeId, 'diagnoses', {
|
|
309
|
+
status: variantResult.diagnostic.ok ? 'warning' : 'failed',
|
|
310
|
+
});
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
return {
|
|
315
|
+
documentKind: 'artifact-graph',
|
|
316
|
+
schemaVersion: 1,
|
|
317
|
+
graphId: `eval:${report.id}`,
|
|
318
|
+
generatedAt,
|
|
319
|
+
source: {
|
|
320
|
+
sourceKind: 'eval',
|
|
321
|
+
sourceId: report.id,
|
|
322
|
+
sourcePath,
|
|
323
|
+
cliVersion: report.meta.cliVersion,
|
|
324
|
+
},
|
|
325
|
+
scope: {
|
|
326
|
+
cwd: process.cwd(),
|
|
327
|
+
artifactKind: scopeArtifactKind(report),
|
|
328
|
+
sourceLocator: report.meta.request?.samplesPath,
|
|
329
|
+
sampleSetHash: sampleSetHash(report.meta.sampleHashes),
|
|
330
|
+
},
|
|
331
|
+
nodes,
|
|
332
|
+
edges,
|
|
333
|
+
summaries: [{
|
|
334
|
+
summaryKind: 'coverage',
|
|
335
|
+
title: 'Eval measurement graph',
|
|
336
|
+
severity: report.results.some((result) => Object.values(result.variants).some((variant) => assertionStatus(variant) === 'failed'))
|
|
337
|
+
? 'medium'
|
|
338
|
+
: 'info',
|
|
339
|
+
}],
|
|
340
|
+
};
|
|
341
|
+
}
|
|
342
|
+
export function persistEvalGraphSidecar(options) {
|
|
343
|
+
const graphDir = evalGraphDirForReportOutput(options.outputDir);
|
|
344
|
+
if (!existsSync(graphDir))
|
|
345
|
+
mkdirSync(graphDir, { recursive: true });
|
|
346
|
+
const fileStem = options.fileStem ?? options.report.id;
|
|
347
|
+
const graphPath = join(graphDir, graphFileName(fileStem));
|
|
348
|
+
const graph = buildEvalArtifactGraph(options);
|
|
349
|
+
writeFileSync(graphPath, JSON.stringify(graph, null, 2));
|
|
350
|
+
return { graphPath };
|
|
351
|
+
}
|
|
@@ -36,20 +36,20 @@ omk CLI 顶层命令包括:`init` / `install` / `list` / `promote` / `rollback
|
|
|
36
36
|
| 查看受管 skill 状态 | → `omk list` |
|
|
37
37
|
| 按证据接受 / 回退某版本 | → `omk promote` / `omk rollback` |
|
|
38
38
|
|
|
39
|
-
如果用户意图不明确,先扫描当前项目结构(skills/
|
|
39
|
+
如果用户意图不明确,先扫描当前项目结构(skills/ 目录、项目级 eval-samples 文件、skill 私有 `.omk/samples.*`),然后推荐最合适的操作。
|
|
40
40
|
|
|
41
41
|
## 第三步:检测项目结构
|
|
42
42
|
|
|
43
43
|
使用 Glob 和 Read 工具检查:
|
|
44
44
|
|
|
45
45
|
1. `skills/` 目录下有哪些 skill 文件(`.md` 或 `*/SKILL.md`)
|
|
46
|
-
2.
|
|
47
|
-
3. 是否有 `skills/*.eval-samples.json
|
|
46
|
+
2. 是否存在项目级 `eval-samples.json` / `eval-samples.yaml` / `eval-samples.yml`,或目录 skill 私有的 `<skill>/.omk/samples.json`
|
|
47
|
+
3. 是否有 `skills/*.eval-samples.json`(扁平 skill 的每 skill 配对文件 → `--batch` 模式)
|
|
48
48
|
|
|
49
49
|
根据检测结果决定:
|
|
50
50
|
|
|
51
|
-
- 多个 skill + 各自的 eval-samples → 建议 `--batch` 批量模式
|
|
52
|
-
- 多个 skill +
|
|
51
|
+
- 多个 skill + 各自的 `.omk/samples.*` 或扁平 skill paired eval-samples → 建议 `--batch` 批量模式
|
|
52
|
+
- 多个 skill + 共享项目级 eval-samples → 建议版本对比模式
|
|
53
53
|
- 只有一个 skill → 建议 `baseline` 对照(`omk eval --control baseline --treatment <skill>`)或 `omk evolve` 改进
|
|
54
54
|
- 没有 eval-samples → 先 `omk sample <skill>` 生成
|
|
55
55
|
|
|
@@ -104,19 +104,19 @@ evolve 默认开**显著性接受门**:候选只在相对当前最优**统计
|
|
|
104
104
|
|
|
105
105
|
```bash
|
|
106
106
|
# 为单个 skill 生成
|
|
107
|
-
omk sample skills/my-skill.md
|
|
107
|
+
omk sample skills/my-skill/SKILL.md
|
|
108
108
|
|
|
109
109
|
# 显式指定数量(不指定时 LLM 根据 skill 类型自动决定 4-8 条)
|
|
110
|
-
omk sample skills/my-skill.md --count 8
|
|
110
|
+
omk sample skills/my-skill/SKILL.md --count 8
|
|
111
111
|
|
|
112
112
|
# 自然语言指定重点覆盖场景
|
|
113
|
-
omk sample skills/my-skill.md --focus "重点覆盖搜索失败 / 权限拒绝 / 跨工具 fallback 路径"
|
|
113
|
+
omk sample skills/my-skill/SKILL.md --focus "重点覆盖搜索失败 / 权限拒绝 / 跨工具 fallback 路径"
|
|
114
114
|
|
|
115
115
|
# 为 skill 目录下所有缺测试集的 skill 批量生成
|
|
116
116
|
omk sample --batch
|
|
117
117
|
```
|
|
118
118
|
|
|
119
|
-
|
|
119
|
+
输出位置:目录 skill(`<skill>/SKILL.md`)→ `<skill>/.omk/samples.json`(标准);扁平 `.md` 单次生成 → 当前目录 `eval-samples.json`(项目级兜底);扁平 `.md` 的 `--batch` 兼容生成 `<skill-dir>/<name>.eval-samples.json`。
|
|
120
120
|
|
|
121
121
|
### 体检 skill 写法
|
|
122
122
|
|
|
@@ -101,7 +101,7 @@ omk eval [flags]
|
|
|
101
101
|
- `--report-only` `boolean`:生成报告并打印 verdict,但始终 exit 0(不参与 CI gate)。
|
|
102
102
|
- `--resume` `option`:从某次失败 run 续跑
|
|
103
103
|
- `--retry` `option`:失败 sample 重试次数
|
|
104
|
-
- `--samples` `option
|
|
104
|
+
- `--samples` `option`:用例文件路径。默认项目级 eval-samples.json,也接受 .yaml/.yml;单 treatment 时可自动发现 <skill>/.omk/。
|
|
105
105
|
- `--skill-dir` `option`:skill 目录,默认 skills
|
|
106
106
|
- `--skip-connectivity` `boolean`:跳 LLM 连通性预检
|
|
107
107
|
- `--skip-doctor` `boolean`:escape hatch:跳 doctor 健康检查门禁(默认强制启用)。沙箱 mock 提供依赖时绕开 doctor 物理路径误报;garbage-in 风险自负。
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
1
|
+
import { mkdirSync, readFileSync, readdirSync, unlinkSync, writeFileSync } from 'node:fs';
|
|
2
|
+
import { join, resolve } from 'node:path';
|
|
3
3
|
import { Args, Flags } from '@oclif/core';
|
|
4
4
|
import { LANG_FLAG, bilingual } from '../oclif/i18n.js';
|
|
5
5
|
import { BaseCommand } from '../oclif/base-command.js';
|
|
@@ -9,50 +9,11 @@ import { tCli } from '../lib/i18n.js';
|
|
|
9
9
|
import { makeDoctorProgress } from '../lib/progress.js';
|
|
10
10
|
import { DEFAULT_DOCTORS_DIR } from '../../eval-core/default-dirs.js';
|
|
11
11
|
import { indexDoctorWrite, removeDoctorCard } from '../../eval-core/artifact-index.js';
|
|
12
|
+
import { doctorReportFileStem, isReportFileName, reportFilePath } from '../../eval-core/artifact-file-names.js';
|
|
13
|
+
import { migrateLegacyReportFiles } from '../../eval-core/report-file-migration.js';
|
|
12
14
|
import { projectDoctorsDir, globalDoctorsDir } from '../../eval-core/measurement-dirs.js';
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
for (const name of DEFAULT_SAMPLE_FILENAMES) {
|
|
16
|
-
const candidate = join(dir, name);
|
|
17
|
-
if (existsSync(candidate))
|
|
18
|
-
return candidate;
|
|
19
|
-
}
|
|
20
|
-
return null;
|
|
21
|
-
}
|
|
22
|
-
function sampleSearchDirs(target, cwd) {
|
|
23
|
-
const dirs = [];
|
|
24
|
-
const add = (dir) => {
|
|
25
|
-
const abs = resolve(dir);
|
|
26
|
-
if (!dirs.includes(abs))
|
|
27
|
-
dirs.push(abs);
|
|
28
|
-
};
|
|
29
|
-
if (target) {
|
|
30
|
-
const absTarget = resolve(target);
|
|
31
|
-
if (existsSync(absTarget)) {
|
|
32
|
-
const stat = statSync(absTarget);
|
|
33
|
-
if (stat.isDirectory()) {
|
|
34
|
-
add(absTarget);
|
|
35
|
-
add(dirname(absTarget));
|
|
36
|
-
add(dirname(dirname(absTarget)));
|
|
37
|
-
}
|
|
38
|
-
else {
|
|
39
|
-
const parent = dirname(absTarget);
|
|
40
|
-
add(parent);
|
|
41
|
-
add(dirname(parent));
|
|
42
|
-
}
|
|
43
|
-
}
|
|
44
|
-
}
|
|
45
|
-
add(cwd);
|
|
46
|
-
return dirs;
|
|
47
|
-
}
|
|
48
|
-
function findDefaultSamplesPath(target, cwd) {
|
|
49
|
-
for (const dir of sampleSearchDirs(target, cwd)) {
|
|
50
|
-
const samplesPath = findSamplesInDir(dir);
|
|
51
|
-
if (samplesPath)
|
|
52
|
-
return samplesPath;
|
|
53
|
-
}
|
|
54
|
-
return null;
|
|
55
|
-
}
|
|
15
|
+
import { findDoctorDeprecatedSamplesHint, findDoctorSamplesPath } from '../../inputs/sample-locator.js';
|
|
16
|
+
import { persistDoctorGraphSidecars, removeDoctorGraphSidecars } from '../../artifact-graph/doctor.js';
|
|
56
17
|
export default class Doctor extends BaseCommand {
|
|
57
18
|
static description = bilingual({
|
|
58
19
|
zh: '体检 omk 工作目录,检查 skill 配置 / 依赖 / executor 连通性。',
|
|
@@ -186,7 +147,16 @@ export default class Doctor extends BaseCommand {
|
|
|
186
147
|
const timeoutSec = flags.timeout != null ? Number(flags.timeout) : defaultTimeoutSec;
|
|
187
148
|
const timeoutMs = Math.max(1000, Math.floor((Number.isFinite(timeoutSec) ? timeoutSec : defaultTimeoutSec) * 1000));
|
|
188
149
|
const cwd = process.cwd();
|
|
189
|
-
const samplesPath = flags.samples ? resolve(flags.samples) :
|
|
150
|
+
const samplesPath = flags.samples ? resolve(flags.samples) : findDoctorSamplesPath(target, cwd);
|
|
151
|
+
const deprecatedSamplesHint = !flags.samples && !samplesPath
|
|
152
|
+
? findDoctorDeprecatedSamplesHint(target, cwd)
|
|
153
|
+
: null;
|
|
154
|
+
if (deprecatedSamplesHint) {
|
|
155
|
+
process.stderr.write(tCli('cli.common.deprecated_skill_samples_path', lang, {
|
|
156
|
+
oldPath: deprecatedSamplesHint.oldPath,
|
|
157
|
+
newPath: deprecatedSamplesHint.newPath,
|
|
158
|
+
}));
|
|
159
|
+
}
|
|
190
160
|
let samples;
|
|
191
161
|
let requires;
|
|
192
162
|
if (samplesPath) {
|
|
@@ -270,7 +240,7 @@ export default class Doctor extends BaseCommand {
|
|
|
270
240
|
}
|
|
271
241
|
persistDoctorReport(report, flags['output-dir']
|
|
272
242
|
? resolve(flags['output-dir'])
|
|
273
|
-
: (flags.global ? globalDoctorsDir() : projectDoctorsDir()));
|
|
243
|
+
: (flags.global ? globalDoctorsDir() : projectDoctorsDir()), lang);
|
|
274
244
|
if (flags.fix) {
|
|
275
245
|
const existing = report;
|
|
276
246
|
if (existing.outcome !== 'failed') {
|
|
@@ -288,47 +258,76 @@ export default class Doctor extends BaseCommand {
|
|
|
288
258
|
// 每个 skill 最多保留多少份历史 doctor 报告(避免无界增长拖慢 studio 启动 +
|
|
289
259
|
// scanDoctorReports 扫盘成本)。50 = ~每天 1 跑撑 1.5 个月 sparkline,够用。
|
|
290
260
|
const DOCTOR_HISTORY_MAX_PER_SKILL = 50;
|
|
291
|
-
function persistDoctorReport(report, outputDir) {
|
|
261
|
+
function persistDoctorReport(report, outputDir, lang = 'zh') {
|
|
292
262
|
const dir = outputDir ?? DEFAULT_DOCTORS_DIR;
|
|
293
263
|
mkdirSync(dir, { recursive: true });
|
|
294
|
-
|
|
264
|
+
migrateLegacyReportFiles(dir, 'doctor');
|
|
295
265
|
for (const skill of report.skills) {
|
|
296
|
-
const counts = {
|
|
266
|
+
const counts = {
|
|
267
|
+
pass: 0,
|
|
268
|
+
warn: 0,
|
|
269
|
+
fail: 0,
|
|
270
|
+
skipped: 0,
|
|
271
|
+
};
|
|
297
272
|
for (const r of skill.results) {
|
|
298
273
|
const s = r.status;
|
|
299
274
|
if (s in counts)
|
|
300
275
|
counts[s]++;
|
|
301
276
|
}
|
|
277
|
+
const outcome = skill.status === 'fail' ? 'failed' : skill.status === 'warn' ? 'warnings_only' : 'passed';
|
|
302
278
|
const perSkill = {
|
|
303
279
|
...report,
|
|
304
280
|
skills: [skill],
|
|
305
|
-
ruleStats: {
|
|
281
|
+
ruleStats: {
|
|
282
|
+
pass: counts.pass,
|
|
283
|
+
warn: counts.warn,
|
|
284
|
+
fail: counts.fail,
|
|
285
|
+
skipped: counts.skipped,
|
|
286
|
+
total: skill.results.length,
|
|
287
|
+
},
|
|
306
288
|
totals: {
|
|
307
289
|
pass: skill.status === 'pass' ? 1 : 0,
|
|
308
290
|
warn: skill.status === 'warn' ? 1 : 0,
|
|
309
291
|
fail: skill.status === 'fail' ? 1 : 0,
|
|
310
292
|
},
|
|
311
|
-
outcome
|
|
293
|
+
outcome,
|
|
312
294
|
};
|
|
313
|
-
const
|
|
314
|
-
const
|
|
315
|
-
const filePath = join(dir, `${cardId}.json`);
|
|
295
|
+
const cardId = doctorReportFileStem(skill.skillName, report.id);
|
|
296
|
+
const filePath = reportFilePath(dir, cardId);
|
|
316
297
|
writeFileSync(filePath, JSON.stringify(perSkill, null, 2), 'utf8');
|
|
317
298
|
// 产物发现索引:per-skill 报告落项目本地后,best-effort 追加全局轻卡片,让 studio 跨项目聚合。
|
|
318
299
|
indexDoctorWrite({
|
|
319
300
|
id: cardId, path: filePath, skillName: skill.skillName, reportId: report.id, timestamp: report.timestamp,
|
|
320
301
|
status: skill.status, passCount: counts.pass, warnCount: counts.warn, failCount: counts.fail,
|
|
321
302
|
}, dir);
|
|
303
|
+
try {
|
|
304
|
+
persistDoctorGraphSidecars({
|
|
305
|
+
report: perSkill,
|
|
306
|
+
skill,
|
|
307
|
+
sourcePath: filePath,
|
|
308
|
+
outputDir: dir,
|
|
309
|
+
fileStem: cardId,
|
|
310
|
+
lang,
|
|
311
|
+
});
|
|
312
|
+
}
|
|
313
|
+
catch (err) {
|
|
314
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
315
|
+
const warning = lang === 'zh'
|
|
316
|
+
? `⚠️ doctor graph sidecar 写入失败:${message}\n`
|
|
317
|
+
: `⚠️ failed to write doctor graph sidecar: ${message}\n`;
|
|
318
|
+
process.stderr.write(warning);
|
|
319
|
+
}
|
|
322
320
|
pruneDoctorHistory(dir, skill.skillName, DOCTOR_HISTORY_MAX_PER_SKILL);
|
|
323
321
|
}
|
|
324
322
|
}
|
|
325
|
-
// 写入新报告后调用:扫 dir 里属于该 skill 的所有 single-skill doctor
|
|
323
|
+
// 写入新报告后调用:扫 dir 里属于该 skill 的所有 single-skill doctor report,
|
|
326
324
|
// 按 timestamp 倒排,保留 maxKeep 份最近的,其余删。按 content 匹配 skillName 不
|
|
327
|
-
//
|
|
325
|
+
// 看文件名,所以清理逻辑不依赖 readdir 顺序或 stem 推断 skill 名。
|
|
328
326
|
export function pruneDoctorHistory(dir, skillName, maxKeep) {
|
|
327
|
+
migrateLegacyReportFiles(dir, 'doctor');
|
|
329
328
|
const candidates = [];
|
|
330
329
|
for (const file of readdirSync(dir)) {
|
|
331
|
-
if (!file
|
|
330
|
+
if (!isReportFileName(file))
|
|
332
331
|
continue;
|
|
333
332
|
try {
|
|
334
333
|
const data = JSON.parse(readFileSync(join(dir, file), 'utf-8'));
|
|
@@ -337,20 +336,22 @@ export function pruneDoctorHistory(dir, skillName, maxKeep) {
|
|
|
337
336
|
continue;
|
|
338
337
|
if (data.skills[0].skillName !== skillName)
|
|
339
338
|
continue;
|
|
340
|
-
candidates.push({ file, timestamp: data.timestamp });
|
|
339
|
+
candidates.push({ file, graphStem: doctorReportFileStem(skillName, data.id), timestamp: data.timestamp });
|
|
341
340
|
}
|
|
342
341
|
catch { /* skip corrupt / unrelated json */ }
|
|
343
342
|
}
|
|
344
343
|
if (candidates.length <= maxKeep)
|
|
345
344
|
return;
|
|
346
345
|
candidates.sort((a, b) => b.timestamp.localeCompare(a.timestamp));
|
|
347
|
-
for (const { file } of candidates.slice(maxKeep)) {
|
|
346
|
+
for (const { file, graphStem } of candidates.slice(maxKeep)) {
|
|
348
347
|
try {
|
|
349
348
|
unlinkSync(join(dir, file));
|
|
350
349
|
}
|
|
351
350
|
catch { /* ignore */ }
|
|
352
351
|
// 连带删卡片:否则被 prune 掉的报告会经 listDoctorCards 合并在本项目 studio「复活」(正文已删、卡片还在)。
|
|
353
352
|
// 卡片 id = 文件 stem(`{name}-{id}`),与 indexDoctorWrite 写入口径一致。
|
|
354
|
-
|
|
353
|
+
const doctorStem = file.replace(/\.report\.json$/, '');
|
|
354
|
+
removeDoctorCard(doctorStem);
|
|
355
|
+
removeDoctorGraphSidecars(dir, graphStem);
|
|
355
356
|
}
|
|
356
357
|
}
|
|
@@ -10,6 +10,7 @@ import { computeRunTally } from '../../lib/run-tally.js';
|
|
|
10
10
|
import { DEFAULT_BOOTSTRAP_SAMPLES } from '../../../eval-core/bootstrap.js';
|
|
11
11
|
import { DEFAULT_GATE_THRESHOLD } from '../../../eval-core/verdict.js';
|
|
12
12
|
import { EVALUATION_REPORT_SCHEMA_VERSION } from '../../../eval-core/evaluation-reporting.js';
|
|
13
|
+
import { findSingleTreatmentDeprecatedSamplesHint, hasUsableSamplesPath, } from '../../../inputs/sample-locator.js';
|
|
13
14
|
function isDryRunReport(report) {
|
|
14
15
|
return Boolean(report && typeof report === 'object' && report.dryRun === true);
|
|
15
16
|
}
|
|
@@ -180,6 +181,23 @@ async function announceSavedReport({ report, filePath, reportsDir, values, lang,
|
|
|
180
181
|
}
|
|
181
182
|
async function runEval(_args, flags, lang) {
|
|
182
183
|
const { values, config, evalConfig } = parseRunConfig({ ...flags });
|
|
184
|
+
if (!values.batch && !hasUsableSamplesPath(config.samplesPath)) {
|
|
185
|
+
const treatmentRaw = typeof values.treatment === 'string' ? values.treatment : '';
|
|
186
|
+
const treatments = treatmentRaw.split(',').map((v) => v.trim()).filter(Boolean);
|
|
187
|
+
const deprecatedSamplesHint = !values.samples && !evalConfig?.samples && treatments.length === 1
|
|
188
|
+
? findSingleTreatmentDeprecatedSamplesHint(treatments[0], config.skillDir, process.cwd())
|
|
189
|
+
: null;
|
|
190
|
+
if (deprecatedSamplesHint) {
|
|
191
|
+
process.stderr.write(tCli('cli.common.deprecated_skill_samples_path', lang, {
|
|
192
|
+
oldPath: deprecatedSamplesHint.oldPath,
|
|
193
|
+
newPath: deprecatedSamplesHint.newPath,
|
|
194
|
+
}));
|
|
195
|
+
}
|
|
196
|
+
console.error(tCli('cli.common.error_prefix', lang, {
|
|
197
|
+
message: tCli('cli.common.samples_not_found', lang, { path: config.samplesPath }),
|
|
198
|
+
}));
|
|
199
|
+
throw new CliExit(1);
|
|
200
|
+
}
|
|
183
201
|
const { runEvaluation, runMultiple, runBatchEvaluation } = await import('../../../eval-workflows/run-evaluation.js');
|
|
184
202
|
config.onProgress = makeOnProgress(lang);
|
|
185
203
|
const repeatRaw = values.repeat;
|
|
@@ -377,8 +395,8 @@ export default class Eval extends BaseCommand {
|
|
|
377
395
|
}),
|
|
378
396
|
samples: Flags.string({
|
|
379
397
|
description: bilingual({
|
|
380
|
-
zh: '
|
|
381
|
-
en: 'Samples
|
|
398
|
+
zh: '用例文件路径。默认项目级 eval-samples.json,也接受 .yaml/.yml;单 treatment 时可自动发现 <skill>/.omk/。',
|
|
399
|
+
en: 'Samples path. Defaults to project-level eval-samples.json (also .yaml/.yml); single-treatment runs can auto-discover <skill>/.omk/.',
|
|
382
400
|
}),
|
|
383
401
|
}),
|
|
384
402
|
'skill-dir': Flags.string({
|