oh-my-knowledge 0.41.0 → 0.43.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -2
- package/README.zh.md +7 -2
- package/dist/analysis/report-diagnostics.d.ts +8 -1
- package/dist/analysis/report-diagnostics.js +82 -1
- package/dist/artifact-graph/doctor.d.ts +21 -0
- package/dist/artifact-graph/doctor.js +569 -0
- package/dist/assets/agent-skills/omk/SKILL.md +9 -9
- package/dist/assets/agent-skills/omk/references/commands.md +2 -1
- package/dist/authoring/evolver.d.ts +3 -14
- package/dist/authoring/evolver.js +1 -52
- package/dist/authoring/generator.d.ts +24 -0
- package/dist/authoring/generator.js +33 -6
- package/dist/cli/commands/doctor.js +62 -61
- package/dist/cli/commands/eval/index.d.ts +1 -0
- package/dist/cli/commands/eval/index.js +60 -7
- package/dist/cli/commands/init.js +11 -7
- package/dist/cli/commands/observe/index.d.ts +2 -2
- package/dist/cli/commands/observe/index.js +8 -7
- package/dist/cli/commands/sample.js +22 -16
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +8 -0
- package/dist/cli/lib/i18n-dict/help.js +12 -10
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +14 -11
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +4 -0
- package/dist/cli/lib/parse-run-config/samples-discovery.d.ts +5 -7
- package/dist/cli/lib/parse-run-config/samples-discovery.js +10 -32
- package/dist/cli/lib/parse-run-config.d.ts +3 -0
- package/dist/cli/lib/parse-run-config.js +4 -4
- package/dist/cli/lib/resolve-skill-input.js +10 -12
- package/dist/doctor/messages.js +2 -2
- package/dist/eval-core/artifact-file-names.d.ts +15 -0
- package/dist/eval-core/artifact-file-names.js +46 -0
- package/dist/eval-core/evaluation-job.d.ts +2 -1
- package/dist/eval-core/evaluation-job.js +2 -1
- package/dist/eval-core/evaluation-reporting.js +9 -4
- package/dist/eval-core/holdout.d.ts +66 -0
- package/dist/eval-core/holdout.js +118 -0
- package/dist/eval-core/measurement-dirs.js +13 -7
- package/dist/eval-core/report-file-migration.d.ts +10 -0
- package/dist/eval-core/report-file-migration.js +90 -0
- package/dist/eval-core/verdict.d.ts +44 -1
- package/dist/eval-core/verdict.js +175 -13
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -1
- package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
- package/dist/eval-workflows/evaluation-pipeline.js +2 -1
- package/dist/eval-workflows/run-evaluation.d.ts +5 -2
- package/dist/eval-workflows/run-evaluation.js +8 -5
- package/dist/inputs/eval-config.js +6 -0
- package/dist/inputs/sample-locator.d.ts +23 -0
- package/dist/inputs/sample-locator.js +195 -0
- package/dist/inputs/skill-loader.js +7 -17
- package/dist/observability/inbox.js +7 -3
- package/dist/renderer/summary.js +36 -3
- package/dist/server/report-server.js +10 -4
- package/dist/server/report-store.js +17 -9
- package/dist/server/skill-index.js +16 -11
- package/dist/types/artifact-graph.d.ts +93 -0
- package/dist/types/artifact-graph.js +1 -0
- package/dist/types/eval.d.ts +7 -0
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/report.d.ts +49 -0
- package/package.json +1 -1
|
@@ -8,7 +8,7 @@ import { projectReportsDir, globalReportsDir } from '../eval-core/measurement-di
|
|
|
8
8
|
import { analyzeResults } from '../analysis/report-diagnostics.js';
|
|
9
9
|
import { loadSamples } from '../inputs/load-samples.js';
|
|
10
10
|
import { hashArtifactSource } from '../inputs/content-hash.js';
|
|
11
|
-
import {
|
|
11
|
+
import { MIN_HOLDOUT_SUBSET, pickByStride, splitHoldout, subsetCompositeScore } from '../eval-core/holdout.js';
|
|
12
12
|
import { bootstrapDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
|
|
13
13
|
import { fixSamples } from './sample-fixer.js';
|
|
14
14
|
const IMPROVE_SYSTEM_PROMPT = `你是一个 AI 提示词改进专家。你的任务是分析评测结果中的薄弱环节,针对性地改进 skill(系统提示词),使其在评测中获得更高的分数。
|
|
@@ -165,43 +165,11 @@ export function extractWeakSamples(report, variantKey, count = 5, sampleIdFilter
|
|
|
165
165
|
.sort((a, b) => a.compositeScore - b.compositeScore)
|
|
166
166
|
.slice(0, count);
|
|
167
167
|
}
|
|
168
|
-
/** Below this many samples on any side, a split is too small to be meaningful —
|
|
169
|
-
* evolve falls back to full-set scoring and warns. */
|
|
170
|
-
const MIN_HOLDOUT_SUBSET = 3;
|
|
171
168
|
/** Below this many decision (val) samples the bootstrap diff CI almost never
|
|
172
169
|
* excludes 0 for realistic effect sizes, so the significance gate would reject
|
|
173
170
|
* every candidate. Under that floor evolve degrades to the point-estimate accept
|
|
174
171
|
* and flags `gate.underpowered`. */
|
|
175
172
|
export const MIN_GATE_SAMPLES = 8;
|
|
176
|
-
/** Pick `count` ids at an even stride across `ids` (deterministic, no RNG) so the
|
|
177
|
-
* picked subset is representative of the ordering and stable across rounds/runs. */
|
|
178
|
-
function pickByStride(ids, count) {
|
|
179
|
-
const picked = new Set();
|
|
180
|
-
if (count <= 0)
|
|
181
|
-
return picked;
|
|
182
|
-
const stride = ids.length / count;
|
|
183
|
-
for (let k = 0; k < count; k++)
|
|
184
|
-
picked.add(ids[Math.floor(k * stride)]);
|
|
185
|
-
return picked;
|
|
186
|
-
}
|
|
187
|
-
/**
|
|
188
|
-
* Deterministically split sample ids into train / holdout by `ratio` (fraction
|
|
189
|
-
* held out). Holdout members are picked at an even stride so the partition is
|
|
190
|
-
* representative of the ordering, and the split is stable across rounds and runs
|
|
191
|
-
* (no RNG). Returns null when ratio ≤ 0 or either side would drop below
|
|
192
|
-
* MIN_HOLDOUT_SUBSET — the caller then scores on the full set.
|
|
193
|
-
*/
|
|
194
|
-
export function splitHoldout(sampleIds, ratio) {
|
|
195
|
-
if (!(ratio > 0) || sampleIds.length === 0)
|
|
196
|
-
return null;
|
|
197
|
-
const holdoutCount = Math.round(sampleIds.length * ratio);
|
|
198
|
-
const trainCount = sampleIds.length - holdoutCount;
|
|
199
|
-
if (holdoutCount < MIN_HOLDOUT_SUBSET || trainCount < MIN_HOLDOUT_SUBSET)
|
|
200
|
-
return null;
|
|
201
|
-
const holdoutIds = pickByStride(sampleIds, holdoutCount);
|
|
202
|
-
const trainIds = new Set(sampleIds.filter((id) => !holdoutIds.has(id)));
|
|
203
|
-
return { trainIds, holdoutIds };
|
|
204
|
-
}
|
|
205
173
|
/**
|
|
206
174
|
* Deterministically split sample ids into train / val / test. `val` is carved
|
|
207
175
|
* first at an even stride; `test` is carved at an even stride over what remains,
|
|
@@ -223,25 +191,6 @@ export function splitTrainValTest(sampleIds, valRatio, testRatio) {
|
|
|
223
191
|
const trainIds = new Set(sampleIds.filter((id) => !valIds.has(id) && !testIds.has(id)));
|
|
224
192
|
return { trainIds, valIds, testIds };
|
|
225
193
|
}
|
|
226
|
-
/**
|
|
227
|
-
* Mean composite over the subset of a report's results whose sample_id is in
|
|
228
|
-
* `ids`, using the same aggregation as the full-run summary
|
|
229
|
-
* (`buildVariantSummary`) so train / holdout scores stay comparable to the
|
|
230
|
-
* headline composite. Returns 0 when the subset has no scorable entries.
|
|
231
|
-
*/
|
|
232
|
-
function subsetCompositeScore(report, variantKey, ids) {
|
|
233
|
-
const entries = [];
|
|
234
|
-
for (const r of report.results) {
|
|
235
|
-
if (!ids.has(r.sample_id))
|
|
236
|
-
continue;
|
|
237
|
-
const v = r.variants[variantKey];
|
|
238
|
-
if (v)
|
|
239
|
-
entries.push(v);
|
|
240
|
-
}
|
|
241
|
-
if (entries.length === 0)
|
|
242
|
-
return 0;
|
|
243
|
-
return buildVariantSummary(entries).avgCompositeScore ?? 0;
|
|
244
|
-
}
|
|
245
194
|
/**
|
|
246
195
|
* Per-sample composite scores over the subset of a report's results whose
|
|
247
196
|
* sample_id is in `ids`, in result order. Feeds `bootstrapDiffCI` for the
|
|
@@ -31,6 +31,30 @@ export declare function generateSamples({ skillContent, count, model, executorNa
|
|
|
31
31
|
costUSD: number;
|
|
32
32
|
}>;
|
|
33
33
|
type TraceSignalItem = Pick<ObservationInboxItem, 'skillName' | 'signalType' | 'signalSubtype' | 'severity' | 'evidence' | 'messageWindow' | 'occurrences'>;
|
|
34
|
+
export interface StratifiedTraceSignal extends TraceSignalItem {
|
|
35
|
+
/** Share of total occurrences across all signals (0-1) — drives proportional
|
|
36
|
+
* sample allocation in the prompt. */
|
|
37
|
+
weight: number;
|
|
38
|
+
}
|
|
39
|
+
/**
|
|
40
|
+
* Rank trace signals by frequency and annotate each with its share of the total
|
|
41
|
+
* occurrences. Lets `omk sample --from-traces` allocate samples *proportional to
|
|
42
|
+
* how often a failure actually happened* instead of a flat "1-2 per signal" — a
|
|
43
|
+
* failure seen 100× deserves more regression coverage than one seen twice.
|
|
44
|
+
*
|
|
45
|
+
* Deliberately does NOT re-merge signals here. The observation inbox is the only
|
|
46
|
+
* source of these items and already aggregates `occurrences` by the FULL identity
|
|
47
|
+
* — `skillName + cwd + sourceKind + signalType + signalSubtype + evidence`
|
|
48
|
+
* (`inbox.ts` `keyFor`). Re-merging on a narrower key (e.g. type+subtype+evidence)
|
|
49
|
+
* would fold *different skills / cwd* with the same failure shape into one entry,
|
|
50
|
+
* mis-attributing the summed occurrences to the first skill and skewing the
|
|
51
|
+
* regenerated distribution. So we trust the upstream dedup and only sort + weight.
|
|
52
|
+
*
|
|
53
|
+
* Does NOT fix the underlying selection bias (traces only capture *failures*),
|
|
54
|
+
* which is why the `omk sample --from-traces` draft warning still stands — this
|
|
55
|
+
* only makes the within-failure distribution representative of frequency.
|
|
56
|
+
*/
|
|
57
|
+
export declare function stratifyTraceSignals(items: TraceSignalItem[]): StratifiedTraceSignal[];
|
|
34
58
|
/**
|
|
35
59
|
* Build the generation prompt for `omk sample --from-traces`. Renders each
|
|
36
60
|
* observation-inbox signal (evidence + message window) into a section and asks
|
|
@@ -465,10 +465,34 @@ async function finalizeSamples(samples, costUSD, skillContent) {
|
|
|
465
465
|
const TRACE_GEN_INSTRUCTIONS = `下面给出的不是 skill,而是从生产会话 trace 中观测到的失败 / 异常信号。请为这些信号生成评测用例(eval samples),使评测能复现并守住这些失败模式——把线上真实发生过的问题沉淀成回归用例。
|
|
466
466
|
|
|
467
467
|
要求:
|
|
468
|
-
-
|
|
468
|
+
- 按各信号标注的「占比」分配用例数:高频信号多生成、低频少生成,让用例集覆盖线上失败的真实频次分布(高占比信号 2-3 条,低占比 1 条即可);信号若是噪声 / 证据不足 / 无法复现,跳过它,不要硬凑。
|
|
469
469
|
- prompt 要还原触发该信号的场景(自然语言任务),不要直接复述证据文本。
|
|
470
470
|
- 断言优先用 mock_hit / tools_called / tools_not_called / tool_input_contains 精确锚定失败步骤,再用 contains 兜底;按「原子型」配比处理(无需工作流编号步骤),除非证据明显是多步流程。
|
|
471
471
|
- 不要在输出里说明判断过程,直接输出 JSON 数组。`;
|
|
472
|
+
/**
|
|
473
|
+
* Rank trace signals by frequency and annotate each with its share of the total
|
|
474
|
+
* occurrences. Lets `omk sample --from-traces` allocate samples *proportional to
|
|
475
|
+
* how often a failure actually happened* instead of a flat "1-2 per signal" — a
|
|
476
|
+
* failure seen 100× deserves more regression coverage than one seen twice.
|
|
477
|
+
*
|
|
478
|
+
* Deliberately does NOT re-merge signals here. The observation inbox is the only
|
|
479
|
+
* source of these items and already aggregates `occurrences` by the FULL identity
|
|
480
|
+
* — `skillName + cwd + sourceKind + signalType + signalSubtype + evidence`
|
|
481
|
+
* (`inbox.ts` `keyFor`). Re-merging on a narrower key (e.g. type+subtype+evidence)
|
|
482
|
+
* would fold *different skills / cwd* with the same failure shape into one entry,
|
|
483
|
+
* mis-attributing the summed occurrences to the first skill and skewing the
|
|
484
|
+
* regenerated distribution. So we trust the upstream dedup and only sort + weight.
|
|
485
|
+
*
|
|
486
|
+
* Does NOT fix the underlying selection bias (traces only capture *failures*),
|
|
487
|
+
* which is why the `omk sample --from-traces` draft warning still stands — this
|
|
488
|
+
* only makes the within-failure distribution representative of frequency.
|
|
489
|
+
*/
|
|
490
|
+
export function stratifyTraceSignals(items) {
|
|
491
|
+
const total = items.reduce((sum, it) => sum + (it.occurrences ?? 0), 0) || 1;
|
|
492
|
+
return items
|
|
493
|
+
.map((it) => ({ ...it, occurrences: it.occurrences ?? 0, weight: (it.occurrences ?? 0) / total }))
|
|
494
|
+
.sort((a, b) => b.occurrences - a.occurrences);
|
|
495
|
+
}
|
|
472
496
|
/**
|
|
473
497
|
* Build the generation prompt for `omk sample --from-traces`. Renders each
|
|
474
498
|
* observation-inbox signal (evidence + message window) into a section and asks
|
|
@@ -477,7 +501,9 @@ const TRACE_GEN_INSTRUCTIONS = `下面给出的不是 skill,而是从生产会
|
|
|
477
501
|
* judge prompt — so judge-prompt isolation is unaffected.
|
|
478
502
|
*/
|
|
479
503
|
export function buildSamplesFromTracesPrompt(items, count) {
|
|
480
|
-
|
|
504
|
+
// 先按频次分层(合并重复 + 算占比 + 降序),让模型按「占比」分配配额,而非每信号一刀切。
|
|
505
|
+
const stratified = stratifyTraceSignals(items);
|
|
506
|
+
const sections = stratified.map((it, i) => {
|
|
481
507
|
const ev = it.evidence ?? {};
|
|
482
508
|
const evLines = [
|
|
483
509
|
ev.tool && `工具: ${ev.tool}`,
|
|
@@ -491,14 +517,15 @@ export function buildSamplesFromTracesPrompt(items, count) {
|
|
|
491
517
|
? '\n上下文消息:\n' + [...it.messageWindow.before, ...it.messageWindow.event, ...it.messageWindow.after]
|
|
492
518
|
.map((m) => ` [${m.role}] ${m.snippet}`).join('\n')
|
|
493
519
|
: '';
|
|
494
|
-
|
|
520
|
+
const pct = (it.weight * 100).toFixed(0);
|
|
521
|
+
return `### 信号 ${i + 1}:${it.signalType} / ${it.signalSubtype}(严重度 ${it.severity},出现 ${it.occurrences} 次 · 占比 ${pct}%,skill: ${it.skillName})\n${evLines}${win}`;
|
|
495
522
|
}).join('\n\n---\n\n');
|
|
496
523
|
const countLine = typeof count === 'number'
|
|
497
|
-
? `共生成约 ${count}
|
|
498
|
-
: '
|
|
524
|
+
? `共生成约 ${count} 条评测用例,按各信号的「占比」分配配额(高频多、低频少),覆盖整体失败分布。`
|
|
525
|
+
: '按各信号「占比」分配:高频信号多生成、低频少生成,覆盖整体失败分布。';
|
|
499
526
|
return `${TRACE_GEN_INSTRUCTIONS}
|
|
500
527
|
|
|
501
|
-
## 观测到的失败信号(共 ${
|
|
528
|
+
## 观测到的失败信号(共 ${stratified.length} 个,已按出现频次降序)
|
|
502
529
|
|
|
503
530
|
${sections}
|
|
504
531
|
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
1
|
+
import { mkdirSync, readFileSync, readdirSync, unlinkSync, writeFileSync } from 'node:fs';
|
|
2
|
+
import { join, resolve } from 'node:path';
|
|
3
3
|
import { Args, Flags } from '@oclif/core';
|
|
4
4
|
import { LANG_FLAG, bilingual } from '../oclif/i18n.js';
|
|
5
5
|
import { BaseCommand } from '../oclif/base-command.js';
|
|
@@ -9,50 +9,11 @@ import { tCli } from '../lib/i18n.js';
|
|
|
9
9
|
import { makeDoctorProgress } from '../lib/progress.js';
|
|
10
10
|
import { DEFAULT_DOCTORS_DIR } from '../../eval-core/default-dirs.js';
|
|
11
11
|
import { indexDoctorWrite, removeDoctorCard } from '../../eval-core/artifact-index.js';
|
|
12
|
+
import { doctorReportFileStem, isReportFileName, reportFilePath } from '../../eval-core/artifact-file-names.js';
|
|
13
|
+
import { migrateLegacyReportFiles } from '../../eval-core/report-file-migration.js';
|
|
12
14
|
import { projectDoctorsDir, globalDoctorsDir } from '../../eval-core/measurement-dirs.js';
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
for (const name of DEFAULT_SAMPLE_FILENAMES) {
|
|
16
|
-
const candidate = join(dir, name);
|
|
17
|
-
if (existsSync(candidate))
|
|
18
|
-
return candidate;
|
|
19
|
-
}
|
|
20
|
-
return null;
|
|
21
|
-
}
|
|
22
|
-
function sampleSearchDirs(target, cwd) {
|
|
23
|
-
const dirs = [];
|
|
24
|
-
const add = (dir) => {
|
|
25
|
-
const abs = resolve(dir);
|
|
26
|
-
if (!dirs.includes(abs))
|
|
27
|
-
dirs.push(abs);
|
|
28
|
-
};
|
|
29
|
-
if (target) {
|
|
30
|
-
const absTarget = resolve(target);
|
|
31
|
-
if (existsSync(absTarget)) {
|
|
32
|
-
const stat = statSync(absTarget);
|
|
33
|
-
if (stat.isDirectory()) {
|
|
34
|
-
add(absTarget);
|
|
35
|
-
add(dirname(absTarget));
|
|
36
|
-
add(dirname(dirname(absTarget)));
|
|
37
|
-
}
|
|
38
|
-
else {
|
|
39
|
-
const parent = dirname(absTarget);
|
|
40
|
-
add(parent);
|
|
41
|
-
add(dirname(parent));
|
|
42
|
-
}
|
|
43
|
-
}
|
|
44
|
-
}
|
|
45
|
-
add(cwd);
|
|
46
|
-
return dirs;
|
|
47
|
-
}
|
|
48
|
-
function findDefaultSamplesPath(target, cwd) {
|
|
49
|
-
for (const dir of sampleSearchDirs(target, cwd)) {
|
|
50
|
-
const samplesPath = findSamplesInDir(dir);
|
|
51
|
-
if (samplesPath)
|
|
52
|
-
return samplesPath;
|
|
53
|
-
}
|
|
54
|
-
return null;
|
|
55
|
-
}
|
|
15
|
+
import { findDoctorDeprecatedSamplesHint, findDoctorSamplesPath } from '../../inputs/sample-locator.js';
|
|
16
|
+
import { persistDoctorGraphSidecars, removeDoctorGraphSidecars } from '../../artifact-graph/doctor.js';
|
|
56
17
|
export default class Doctor extends BaseCommand {
|
|
57
18
|
static description = bilingual({
|
|
58
19
|
zh: '体检 omk 工作目录,检查 skill 配置 / 依赖 / executor 连通性。',
|
|
@@ -186,7 +147,16 @@ export default class Doctor extends BaseCommand {
|
|
|
186
147
|
const timeoutSec = flags.timeout != null ? Number(flags.timeout) : defaultTimeoutSec;
|
|
187
148
|
const timeoutMs = Math.max(1000, Math.floor((Number.isFinite(timeoutSec) ? timeoutSec : defaultTimeoutSec) * 1000));
|
|
188
149
|
const cwd = process.cwd();
|
|
189
|
-
const samplesPath = flags.samples ? resolve(flags.samples) :
|
|
150
|
+
const samplesPath = flags.samples ? resolve(flags.samples) : findDoctorSamplesPath(target, cwd);
|
|
151
|
+
const deprecatedSamplesHint = !flags.samples && !samplesPath
|
|
152
|
+
? findDoctorDeprecatedSamplesHint(target, cwd)
|
|
153
|
+
: null;
|
|
154
|
+
if (deprecatedSamplesHint) {
|
|
155
|
+
process.stderr.write(tCli('cli.common.deprecated_skill_samples_path', lang, {
|
|
156
|
+
oldPath: deprecatedSamplesHint.oldPath,
|
|
157
|
+
newPath: deprecatedSamplesHint.newPath,
|
|
158
|
+
}));
|
|
159
|
+
}
|
|
190
160
|
let samples;
|
|
191
161
|
let requires;
|
|
192
162
|
if (samplesPath) {
|
|
@@ -270,7 +240,7 @@ export default class Doctor extends BaseCommand {
|
|
|
270
240
|
}
|
|
271
241
|
persistDoctorReport(report, flags['output-dir']
|
|
272
242
|
? resolve(flags['output-dir'])
|
|
273
|
-
: (flags.global ? globalDoctorsDir() : projectDoctorsDir()));
|
|
243
|
+
: (flags.global ? globalDoctorsDir() : projectDoctorsDir()), lang);
|
|
274
244
|
if (flags.fix) {
|
|
275
245
|
const existing = report;
|
|
276
246
|
if (existing.outcome !== 'failed') {
|
|
@@ -288,47 +258,76 @@ export default class Doctor extends BaseCommand {
|
|
|
288
258
|
// 每个 skill 最多保留多少份历史 doctor 报告(避免无界增长拖慢 studio 启动 +
|
|
289
259
|
// scanDoctorReports 扫盘成本)。50 = ~每天 1 跑撑 1.5 个月 sparkline,够用。
|
|
290
260
|
const DOCTOR_HISTORY_MAX_PER_SKILL = 50;
|
|
291
|
-
function persistDoctorReport(report, outputDir) {
|
|
261
|
+
function persistDoctorReport(report, outputDir, lang = 'zh') {
|
|
292
262
|
const dir = outputDir ?? DEFAULT_DOCTORS_DIR;
|
|
293
263
|
mkdirSync(dir, { recursive: true });
|
|
294
|
-
|
|
264
|
+
migrateLegacyReportFiles(dir, 'doctor');
|
|
295
265
|
for (const skill of report.skills) {
|
|
296
|
-
const counts = {
|
|
266
|
+
const counts = {
|
|
267
|
+
pass: 0,
|
|
268
|
+
warn: 0,
|
|
269
|
+
fail: 0,
|
|
270
|
+
skipped: 0,
|
|
271
|
+
};
|
|
297
272
|
for (const r of skill.results) {
|
|
298
273
|
const s = r.status;
|
|
299
274
|
if (s in counts)
|
|
300
275
|
counts[s]++;
|
|
301
276
|
}
|
|
277
|
+
const outcome = skill.status === 'fail' ? 'failed' : skill.status === 'warn' ? 'warnings_only' : 'passed';
|
|
302
278
|
const perSkill = {
|
|
303
279
|
...report,
|
|
304
280
|
skills: [skill],
|
|
305
|
-
ruleStats: {
|
|
281
|
+
ruleStats: {
|
|
282
|
+
pass: counts.pass,
|
|
283
|
+
warn: counts.warn,
|
|
284
|
+
fail: counts.fail,
|
|
285
|
+
skipped: counts.skipped,
|
|
286
|
+
total: skill.results.length,
|
|
287
|
+
},
|
|
306
288
|
totals: {
|
|
307
289
|
pass: skill.status === 'pass' ? 1 : 0,
|
|
308
290
|
warn: skill.status === 'warn' ? 1 : 0,
|
|
309
291
|
fail: skill.status === 'fail' ? 1 : 0,
|
|
310
292
|
},
|
|
311
|
-
outcome
|
|
293
|
+
outcome,
|
|
312
294
|
};
|
|
313
|
-
const
|
|
314
|
-
const
|
|
315
|
-
const filePath = join(dir, `${cardId}.json`);
|
|
295
|
+
const cardId = doctorReportFileStem(skill.skillName, report.id);
|
|
296
|
+
const filePath = reportFilePath(dir, cardId);
|
|
316
297
|
writeFileSync(filePath, JSON.stringify(perSkill, null, 2), 'utf8');
|
|
317
298
|
// 产物发现索引:per-skill 报告落项目本地后,best-effort 追加全局轻卡片,让 studio 跨项目聚合。
|
|
318
299
|
indexDoctorWrite({
|
|
319
300
|
id: cardId, path: filePath, skillName: skill.skillName, reportId: report.id, timestamp: report.timestamp,
|
|
320
301
|
status: skill.status, passCount: counts.pass, warnCount: counts.warn, failCount: counts.fail,
|
|
321
302
|
}, dir);
|
|
303
|
+
try {
|
|
304
|
+
persistDoctorGraphSidecars({
|
|
305
|
+
report: perSkill,
|
|
306
|
+
skill,
|
|
307
|
+
sourcePath: filePath,
|
|
308
|
+
outputDir: dir,
|
|
309
|
+
fileStem: cardId,
|
|
310
|
+
lang,
|
|
311
|
+
});
|
|
312
|
+
}
|
|
313
|
+
catch (err) {
|
|
314
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
315
|
+
const warning = lang === 'zh'
|
|
316
|
+
? `⚠️ doctor graph sidecar 写入失败:${message}\n`
|
|
317
|
+
: `⚠️ failed to write doctor graph sidecar: ${message}\n`;
|
|
318
|
+
process.stderr.write(warning);
|
|
319
|
+
}
|
|
322
320
|
pruneDoctorHistory(dir, skill.skillName, DOCTOR_HISTORY_MAX_PER_SKILL);
|
|
323
321
|
}
|
|
324
322
|
}
|
|
325
|
-
// 写入新报告后调用:扫 dir 里属于该 skill 的所有 single-skill doctor
|
|
323
|
+
// 写入新报告后调用:扫 dir 里属于该 skill 的所有 single-skill doctor report,
|
|
326
324
|
// 按 timestamp 倒排,保留 maxKeep 份最近的,其余删。按 content 匹配 skillName 不
|
|
327
|
-
//
|
|
325
|
+
// 看文件名,所以清理逻辑不依赖 readdir 顺序或 stem 推断 skill 名。
|
|
328
326
|
export function pruneDoctorHistory(dir, skillName, maxKeep) {
|
|
327
|
+
migrateLegacyReportFiles(dir, 'doctor');
|
|
329
328
|
const candidates = [];
|
|
330
329
|
for (const file of readdirSync(dir)) {
|
|
331
|
-
if (!file
|
|
330
|
+
if (!isReportFileName(file))
|
|
332
331
|
continue;
|
|
333
332
|
try {
|
|
334
333
|
const data = JSON.parse(readFileSync(join(dir, file), 'utf-8'));
|
|
@@ -337,20 +336,22 @@ export function pruneDoctorHistory(dir, skillName, maxKeep) {
|
|
|
337
336
|
continue;
|
|
338
337
|
if (data.skills[0].skillName !== skillName)
|
|
339
338
|
continue;
|
|
340
|
-
candidates.push({ file, timestamp: data.timestamp });
|
|
339
|
+
candidates.push({ file, graphStem: doctorReportFileStem(skillName, data.id), timestamp: data.timestamp });
|
|
341
340
|
}
|
|
342
341
|
catch { /* skip corrupt / unrelated json */ }
|
|
343
342
|
}
|
|
344
343
|
if (candidates.length <= maxKeep)
|
|
345
344
|
return;
|
|
346
345
|
candidates.sort((a, b) => b.timestamp.localeCompare(a.timestamp));
|
|
347
|
-
for (const { file } of candidates.slice(maxKeep)) {
|
|
346
|
+
for (const { file, graphStem } of candidates.slice(maxKeep)) {
|
|
348
347
|
try {
|
|
349
348
|
unlinkSync(join(dir, file));
|
|
350
349
|
}
|
|
351
350
|
catch { /* ignore */ }
|
|
352
351
|
// 连带删卡片:否则被 prune 掉的报告会经 listDoctorCards 合并在本项目 studio「复活」(正文已删、卡片还在)。
|
|
353
352
|
// 卡片 id = 文件 stem(`{name}-{id}`),与 indexDoctorWrite 写入口径一致。
|
|
354
|
-
|
|
353
|
+
const doctorStem = file.replace(/\.report\.json$/, '');
|
|
354
|
+
removeDoctorCard(doctorStem);
|
|
355
|
+
removeDoctorGraphSidecars(dir, graphStem);
|
|
355
356
|
}
|
|
356
357
|
}
|
|
@@ -38,6 +38,7 @@ export default class Eval extends BaseCommand {
|
|
|
38
38
|
effort: import("@oclif/core/interfaces").OptionFlag<string | undefined, import("@oclif/core/interfaces").CustomOptions>;
|
|
39
39
|
'no-diagnostic': import("@oclif/core/interfaces").BooleanFlag<boolean>;
|
|
40
40
|
repeat: import("@oclif/core/interfaces").OptionFlag<string | undefined, import("@oclif/core/interfaces").CustomOptions>;
|
|
41
|
+
'holdout-ratio': import("@oclif/core/interfaces").OptionFlag<string | undefined, import("@oclif/core/interfaces").CustomOptions>;
|
|
41
42
|
'judge-repeat': import("@oclif/core/interfaces").OptionFlag<string | undefined, import("@oclif/core/interfaces").CustomOptions>;
|
|
42
43
|
bootstrap: import("@oclif/core/interfaces").BooleanFlag<boolean>;
|
|
43
44
|
'bootstrap-samples': import("@oclif/core/interfaces").OptionFlag<string | undefined, import("@oclif/core/interfaces").CustomOptions>;
|
|
@@ -10,6 +10,7 @@ import { computeRunTally } from '../../lib/run-tally.js';
|
|
|
10
10
|
import { DEFAULT_BOOTSTRAP_SAMPLES } from '../../../eval-core/bootstrap.js';
|
|
11
11
|
import { DEFAULT_GATE_THRESHOLD } from '../../../eval-core/verdict.js';
|
|
12
12
|
import { EVALUATION_REPORT_SCHEMA_VERSION } from '../../../eval-core/evaluation-reporting.js';
|
|
13
|
+
import { findSingleTreatmentDeprecatedSamplesHint, hasUsableSamplesPath, } from '../../../inputs/sample-locator.js';
|
|
13
14
|
function isDryRunReport(report) {
|
|
14
15
|
return Boolean(report && typeof report === 'object' && report.dryRun === true);
|
|
15
16
|
}
|
|
@@ -41,10 +42,30 @@ function applyGateExitCode(code, values, lang) {
|
|
|
41
42
|
process.stderr.write(tCli('cli.run.report_only_gate_skipped', lang));
|
|
42
43
|
return 0;
|
|
43
44
|
}
|
|
45
|
+
/**
|
|
46
|
+
* 完整 report JSON 是**机器输出**:重定向 / 管道(`omk eval > r.json`、`| jq`)时吐到 stdout 供下游消费。
|
|
47
|
+
* 交互式 TTY 下报告已存盘、(默认)还起了 report server,再刷上千行 JSON 只会把 verdict 淹没在屏幕外 ——
|
|
48
|
+
* 故只在非 TTY(stdout 被重定向 / 管道)时 dump。dry-run 的 JSON 是用户显式索取的产物,不走此门控。
|
|
49
|
+
*/
|
|
50
|
+
function emitReportJson(report) {
|
|
51
|
+
if (!process.stdout.isTTY) {
|
|
52
|
+
console.log(JSON.stringify(report, null, 2));
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
/**
|
|
56
|
+
* 给人读的 verdict 文案:stdout 是 TTY 时进 stdout(交互终端没有 JSON,verdict 就是答案),
|
|
57
|
+
* 否则进 stderr —— 与 emitReportJson 配对,保证非 TTY 的 stdout 是**纯 report JSON**,
|
|
58
|
+
* `omk eval | jq` / `> report.json` 不会被末尾拼上的人类文案噎住(否则 JSON.parse 直接失败)。
|
|
59
|
+
*/
|
|
60
|
+
function emitVerdictText(text) {
|
|
61
|
+
// 与 console.log 等价(对单个字符串 = write(text + '\n')),只切换目标流,逐字节保留既有文案。
|
|
62
|
+
const stream = process.stdout.isTTY ? process.stdout : process.stderr;
|
|
63
|
+
stream.write(text + '\n');
|
|
64
|
+
}
|
|
44
65
|
async function emitEvaluationVerdict(report, values, lang) {
|
|
45
66
|
const { computeVerdict, formatVerdictText } = await import('../../../eval-core/verdict.js');
|
|
46
67
|
const result = computeVerdict(report, verdictOptions(values));
|
|
47
|
-
|
|
68
|
+
emitVerdictText(formatVerdictText(result, { verbose: true, lang }));
|
|
48
69
|
await recordEvidenceSafely(report, result.level, values, lang);
|
|
49
70
|
return verdictPasses(result.level, result.headline) ? 0 : 1;
|
|
50
71
|
}
|
|
@@ -123,13 +144,13 @@ async function emitBatchVerdict(report, reportsDir, values, lang) {
|
|
|
123
144
|
const status = lang === 'zh'
|
|
124
145
|
? (failed === 0 ? '通过' : '未通过')
|
|
125
146
|
: (failed === 0 ? 'PASS' : 'FAIL');
|
|
126
|
-
|
|
147
|
+
emitVerdictText(tCli('cli.run.batch_verdict_header', lang, {
|
|
127
148
|
status,
|
|
128
149
|
passed,
|
|
129
150
|
total: results.length,
|
|
130
151
|
}));
|
|
131
152
|
for (const result of results) {
|
|
132
|
-
|
|
153
|
+
emitVerdictText(` ${result.verdict.level}: ${result.treatment} — ${result.verdict.headline}`);
|
|
133
154
|
}
|
|
134
155
|
return failed === 0 ? 0 : 1;
|
|
135
156
|
}
|
|
@@ -160,6 +181,23 @@ async function announceSavedReport({ report, filePath, reportsDir, values, lang,
|
|
|
160
181
|
}
|
|
161
182
|
async function runEval(_args, flags, lang) {
|
|
162
183
|
const { values, config, evalConfig } = parseRunConfig({ ...flags });
|
|
184
|
+
if (!values.batch && !hasUsableSamplesPath(config.samplesPath)) {
|
|
185
|
+
const treatmentRaw = typeof values.treatment === 'string' ? values.treatment : '';
|
|
186
|
+
const treatments = treatmentRaw.split(',').map((v) => v.trim()).filter(Boolean);
|
|
187
|
+
const deprecatedSamplesHint = !values.samples && !evalConfig?.samples && treatments.length === 1
|
|
188
|
+
? findSingleTreatmentDeprecatedSamplesHint(treatments[0], config.skillDir, process.cwd())
|
|
189
|
+
: null;
|
|
190
|
+
if (deprecatedSamplesHint) {
|
|
191
|
+
process.stderr.write(tCli('cli.common.deprecated_skill_samples_path', lang, {
|
|
192
|
+
oldPath: deprecatedSamplesHint.oldPath,
|
|
193
|
+
newPath: deprecatedSamplesHint.newPath,
|
|
194
|
+
}));
|
|
195
|
+
}
|
|
196
|
+
console.error(tCli('cli.common.error_prefix', lang, {
|
|
197
|
+
message: tCli('cli.common.samples_not_found', lang, { path: config.samplesPath }),
|
|
198
|
+
}));
|
|
199
|
+
throw new CliExit(1);
|
|
200
|
+
}
|
|
163
201
|
const { runEvaluation, runMultiple, runBatchEvaluation } = await import('../../../eval-workflows/run-evaluation.js');
|
|
164
202
|
config.onProgress = makeOnProgress(lang);
|
|
165
203
|
const repeatRaw = values.repeat;
|
|
@@ -168,6 +206,13 @@ async function runEval(_args, flags, lang) {
|
|
|
168
206
|
process.stderr.write(tCli('cli.run.invalid_repeat', lang, { value: repeatRaw }));
|
|
169
207
|
}
|
|
170
208
|
const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
|
|
209
|
+
const holdoutRatioRaw = values['holdout-ratio'];
|
|
210
|
+
const parsedHoldoutRatio = holdoutRatioRaw !== undefined ? Number(holdoutRatioRaw) : (evalConfig?.holdoutRatio ?? 0);
|
|
211
|
+
if (holdoutRatioRaw !== undefined && (!Number.isFinite(parsedHoldoutRatio) || parsedHoldoutRatio <= 0 || parsedHoldoutRatio >= 1)) {
|
|
212
|
+
process.stderr.write(tCli('cli.run.invalid_holdout_ratio', lang, { value: holdoutRatioRaw }));
|
|
213
|
+
}
|
|
214
|
+
if (parsedHoldoutRatio > 0 && parsedHoldoutRatio < 1)
|
|
215
|
+
config.holdoutRatio = parsedHoldoutRatio;
|
|
171
216
|
const judgeRepeatRaw = values['judge-repeat'];
|
|
172
217
|
const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : (evalConfig?.judgeRepeat ?? 1);
|
|
173
218
|
if (judgeRepeatRaw !== undefined && (!Number.isFinite(parsedJudgeRepeat) || parsedJudgeRepeat < 1)) {
|
|
@@ -224,11 +269,12 @@ async function runEval(_args, flags, lang) {
|
|
|
224
269
|
}
|
|
225
270
|
},
|
|
226
271
|
});
|
|
227
|
-
console.log(JSON.stringify(report, null, 2));
|
|
228
272
|
if (isDryRunBatchReport(report)) {
|
|
273
|
+
console.log(JSON.stringify(report, null, 2));
|
|
229
274
|
console.log(tCli('cli.run.dry_run_no_scores', lang));
|
|
230
275
|
throw new CliExit(0);
|
|
231
276
|
}
|
|
277
|
+
emitReportJson(report);
|
|
232
278
|
if (filePath) {
|
|
233
279
|
await announceSavedReport({ report, filePath, reportsDir: config.outputDir, values, lang });
|
|
234
280
|
}
|
|
@@ -285,7 +331,7 @@ async function runEval(_args, flags, lang) {
|
|
|
285
331
|
}
|
|
286
332
|
}
|
|
287
333
|
}
|
|
288
|
-
|
|
334
|
+
emitReportJson(report);
|
|
289
335
|
if (filePath) {
|
|
290
336
|
await announceSavedReport({ report, filePath, reportsDir: config.outputDir, values, lang });
|
|
291
337
|
}
|
|
@@ -349,8 +395,8 @@ export default class Eval extends BaseCommand {
|
|
|
349
395
|
}),
|
|
350
396
|
samples: Flags.string({
|
|
351
397
|
description: bilingual({
|
|
352
|
-
zh: '
|
|
353
|
-
en: 'Samples
|
|
398
|
+
zh: '用例文件路径。默认项目级 eval-samples.json,也接受 .yaml/.yml;单 treatment 时可自动发现 <skill>/.omk/。',
|
|
399
|
+
en: 'Samples path. Defaults to project-level eval-samples.json (also .yaml/.yml); single-treatment runs can auto-discover <skill>/.omk/.',
|
|
354
400
|
}),
|
|
355
401
|
}),
|
|
356
402
|
'skill-dir': Flags.string({
|
|
@@ -457,6 +503,13 @@ export default class Eval extends BaseCommand {
|
|
|
457
503
|
description: bilingual({ zh: '每个 sample 重复跑 N 次', en: 'Repeat each sample N times' }),
|
|
458
504
|
parse: integerStringParser('--repeat', { min: 1 }),
|
|
459
505
|
}),
|
|
506
|
+
'holdout-ratio': Flags.string({
|
|
507
|
+
description: bilingual({
|
|
508
|
+
zh: '留出比例 0-1(如 0.3);切出 holdout 子集,对比 train/holdout 综合分检测过拟合',
|
|
509
|
+
en: 'Holdout fraction 0-1 (e.g. 0.3); splits a holdout subset, compares train/holdout composite to flag overfitting',
|
|
510
|
+
}),
|
|
511
|
+
parse: numberStringParser('--holdout-ratio', { min: 0, max: 1 }),
|
|
512
|
+
}),
|
|
460
513
|
'judge-repeat': Flags.string({
|
|
461
514
|
description: bilingual({ zh: '每个 dim 评 N 次', en: 'Judge each dim N times' }),
|
|
462
515
|
parse: integerStringParser('--judge-repeat', { min: 1 }),
|
|
@@ -8,21 +8,26 @@ import { tCli } from '../lib/i18n.js';
|
|
|
8
8
|
const INIT_OMK_GITIGNORE = `# omk 测量 bulk + doctor --fix 备份(项目本地)——不入库;前导 / 锚定 .omk/ 顶层,不误伤嵌套同名目录。
|
|
9
9
|
/observe-health/
|
|
10
10
|
/doctors/
|
|
11
|
+
/graphs/
|
|
11
12
|
/observe-inbox/
|
|
12
13
|
/reports/
|
|
13
14
|
/backups/
|
|
14
15
|
`;
|
|
16
|
+
// 脚手架用例必须过 omk 自身的断言合规校验(load-samples.ts Rule A),否则新用户照
|
|
17
|
+
// 快速开始跑的第一条 omk eval 会直接硬报错。约束:contains / not_contains 的 value
|
|
18
|
+
// 只能是单个 ASCII token(长度 [2,40]、无内部空白、无 CJK);多词 / 中文语义匹配一律
|
|
19
|
+
// 走 rubric 交评委判;regex pattern 不能含 CJK。改这里前先跑 `omk eval --dry-run`
|
|
20
|
+
// (非 lenient 合规 oracle)与 test/cli/init-scaffold-conformance 回归测试。
|
|
15
21
|
const INIT_SAMPLES = `[
|
|
16
22
|
{
|
|
17
23
|
"sample_id": "s001",
|
|
18
24
|
"prompt": "审查以下代码",
|
|
19
25
|
"context": "function authenticate(username, password) {\\n const query = \`SELECT * FROM users WHERE name='\${username}' AND pass='\${password}'\`;\\n return db.execute(query);\\n}",
|
|
20
|
-
"rubric": "应识别 SQL
|
|
26
|
+
"rubric": "应识别 SQL 注入风险,建议使用参数化查询;不应把这段代码判为安全无问题。",
|
|
21
27
|
"assertions": [
|
|
22
28
|
{ "type": "contains", "value": "SQL", "weight": 1 },
|
|
23
29
|
{ "type": "contains", "value": "injection", "weight": 1 },
|
|
24
|
-
{ "type": "regex", "pattern": "parameterized|prepared|placeholder|bind", "flags": "i", "weight": 0.5 }
|
|
25
|
-
{ "type": "not_contains", "value": "looks good", "weight": 0.5 }
|
|
30
|
+
{ "type": "regex", "pattern": "parameterized|prepared|placeholder|bind", "flags": "i", "weight": 0.5 }
|
|
26
31
|
],
|
|
27
32
|
"dimensions": {
|
|
28
33
|
"security": "是否准确识别出 SQL 注入漏洞并说明其危害",
|
|
@@ -35,8 +40,7 @@ const INIT_SAMPLES = `[
|
|
|
35
40
|
"context": "async function fetchData(url) {\\n const res = await fetch(url);\\n const data = await res.json();\\n return data;\\n}",
|
|
36
41
|
"rubric": "应指出缺少错误处理(网络异常、非 JSON 响应、HTTP 错误状态码)",
|
|
37
42
|
"assertions": [
|
|
38
|
-
{ "type": "
|
|
39
|
-
{ "type": "regex", "pattern": "try[\\\\s\\\\S]*catch|exception|error", "flags": "i", "weight": 1 },
|
|
43
|
+
{ "type": "regex", "pattern": "try[\\\\s\\\\S]*catch|catch|exception|error", "flags": "i", "weight": 1 },
|
|
40
44
|
{ "type": "contains", "value": "status", "weight": 0.5 }
|
|
41
45
|
],
|
|
42
46
|
"dimensions": {
|
|
@@ -154,9 +158,9 @@ export default class Init extends BaseCommand {
|
|
|
154
158
|
console.log(tCli('cli.init.scaffolded', lang, { dir: targetDir }));
|
|
155
159
|
console.log('');
|
|
156
160
|
console.log(tCli('cli.init.next_steps_title', lang));
|
|
157
|
-
console.log(tCli('cli.init.next_step_edit_samples', lang));
|
|
158
|
-
console.log(tCli('cli.init.next_step_edit_skills', lang));
|
|
159
161
|
console.log(tCli('cli.init.next_step_run', lang));
|
|
162
|
+
console.log(tCli('cli.init.next_step_executor', lang));
|
|
163
|
+
console.log(tCli('cli.init.next_step_customize', lang));
|
|
160
164
|
console.log(tCli('cli.init.note_codex_executor', lang));
|
|
161
165
|
});
|
|
162
166
|
}
|
|
@@ -2,8 +2,8 @@ import { BaseCommand } from '../../oclif/base-command.js';
|
|
|
2
2
|
import type { SkillHealthReport } from '../../../observability/skill-health-analyzer.js';
|
|
3
3
|
/**
|
|
4
4
|
* observe-health 报告落盘:id / 文件名加 4 位随机段,根治「同秒两次 omk observe 直接覆盖、数据丢失」的 bug。
|
|
5
|
-
*
|
|
6
|
-
*
|
|
5
|
+
* 文件名统一为 `{run}.report.json`:目录表达 observe-health 域,文件表达这是可读报告。
|
|
6
|
+
* 落盘后 best-effort 追加全局轻卡片,让 studio 跨项目聚合。
|
|
7
7
|
*/
|
|
8
8
|
export declare function persistObserveHealthReport(report: SkillHealthReport, outDir: string): {
|
|
9
9
|
id: string;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { mkdirSync, writeFileSync } from 'node:fs';
|
|
2
|
-
import { resolve
|
|
2
|
+
import { resolve } from 'node:path';
|
|
3
3
|
import { Args, Flags } from '@oclif/core';
|
|
4
4
|
import { LANG_FLAG, bilingual } from '../../oclif/i18n.js';
|
|
5
5
|
import { BaseCommand } from '../../oclif/base-command.js';
|
|
@@ -8,17 +8,18 @@ import { tCli } from '../../lib/i18n.js';
|
|
|
8
8
|
import { parseLastWindow } from '../../lib/shared.js';
|
|
9
9
|
import { projectObserveHealthDir, globalObserveHealthDir } from '../../../eval-core/measurement-dirs.js';
|
|
10
10
|
import { indexObserveWrite } from '../../../eval-core/artifact-index.js';
|
|
11
|
+
import { reportFilePath, runFileSuffix } from '../../../eval-core/artifact-file-names.js';
|
|
12
|
+
import { migrateLegacyReportFiles } from '../../../eval-core/report-file-migration.js';
|
|
11
13
|
/**
|
|
12
14
|
* observe-health 报告落盘:id / 文件名加 4 位随机段,根治「同秒两次 omk observe 直接覆盖、数据丢失」的 bug。
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
+
* 文件名统一为 `{run}.report.json`:目录表达 observe-health 域,文件表达这是可读报告。
|
|
16
|
+
* 落盘后 best-effort 追加全局轻卡片,让 studio 跨项目聚合。
|
|
15
17
|
*/
|
|
16
18
|
export function persistObserveHealthReport(report, outDir) {
|
|
17
19
|
mkdirSync(outDir, { recursive: true });
|
|
18
|
-
|
|
19
|
-
const
|
|
20
|
-
const
|
|
21
|
-
const jsonPath = join(outDir, `${id}.json`);
|
|
20
|
+
migrateLegacyReportFiles(outDir, 'observe-health');
|
|
21
|
+
const id = runFileSuffix();
|
|
22
|
+
const jsonPath = reportFilePath(outDir, id);
|
|
22
23
|
writeFileSync(jsonPath, JSON.stringify(report, null, 2));
|
|
23
24
|
indexObserveWrite(report, jsonPath, outDir, id);
|
|
24
25
|
return { id, jsonPath };
|