oh-my-knowledge 0.41.0 → 0.43.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -2
- package/README.zh.md +7 -2
- package/dist/analysis/report-diagnostics.d.ts +8 -1
- package/dist/analysis/report-diagnostics.js +82 -1
- package/dist/artifact-graph/doctor.d.ts +21 -0
- package/dist/artifact-graph/doctor.js +569 -0
- package/dist/assets/agent-skills/omk/SKILL.md +9 -9
- package/dist/assets/agent-skills/omk/references/commands.md +2 -1
- package/dist/authoring/evolver.d.ts +3 -14
- package/dist/authoring/evolver.js +1 -52
- package/dist/authoring/generator.d.ts +24 -0
- package/dist/authoring/generator.js +33 -6
- package/dist/cli/commands/doctor.js +62 -61
- package/dist/cli/commands/eval/index.d.ts +1 -0
- package/dist/cli/commands/eval/index.js +60 -7
- package/dist/cli/commands/init.js +11 -7
- package/dist/cli/commands/observe/index.d.ts +2 -2
- package/dist/cli/commands/observe/index.js +8 -7
- package/dist/cli/commands/sample.js +22 -16
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +8 -0
- package/dist/cli/lib/i18n-dict/help.js +12 -10
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +14 -11
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +4 -0
- package/dist/cli/lib/parse-run-config/samples-discovery.d.ts +5 -7
- package/dist/cli/lib/parse-run-config/samples-discovery.js +10 -32
- package/dist/cli/lib/parse-run-config.d.ts +3 -0
- package/dist/cli/lib/parse-run-config.js +4 -4
- package/dist/cli/lib/resolve-skill-input.js +10 -12
- package/dist/doctor/messages.js +2 -2
- package/dist/eval-core/artifact-file-names.d.ts +15 -0
- package/dist/eval-core/artifact-file-names.js +46 -0
- package/dist/eval-core/evaluation-job.d.ts +2 -1
- package/dist/eval-core/evaluation-job.js +2 -1
- package/dist/eval-core/evaluation-reporting.js +9 -4
- package/dist/eval-core/holdout.d.ts +66 -0
- package/dist/eval-core/holdout.js +118 -0
- package/dist/eval-core/measurement-dirs.js +13 -7
- package/dist/eval-core/report-file-migration.d.ts +10 -0
- package/dist/eval-core/report-file-migration.js +90 -0
- package/dist/eval-core/verdict.d.ts +44 -1
- package/dist/eval-core/verdict.js +175 -13
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -1
- package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
- package/dist/eval-workflows/evaluation-pipeline.js +2 -1
- package/dist/eval-workflows/run-evaluation.d.ts +5 -2
- package/dist/eval-workflows/run-evaluation.js +8 -5
- package/dist/inputs/eval-config.js +6 -0
- package/dist/inputs/sample-locator.d.ts +23 -0
- package/dist/inputs/sample-locator.js +195 -0
- package/dist/inputs/skill-loader.js +7 -17
- package/dist/observability/inbox.js +7 -3
- package/dist/renderer/summary.js +36 -3
- package/dist/server/report-server.js +10 -4
- package/dist/server/report-store.js +17 -9
- package/dist/server/skill-index.js +16 -11
- package/dist/types/artifact-graph.d.ts +93 -0
- package/dist/types/artifact-graph.js +1 -0
- package/dist/types/eval.d.ts +7 -0
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/report.d.ts +49 -0
- package/package.json +1 -1
|
@@ -5,6 +5,8 @@
|
|
|
5
5
|
*/
|
|
6
6
|
import { readdir, readFile, writeFile, unlink, access, mkdir, rename, stat } from 'node:fs/promises';
|
|
7
7
|
import { join } from 'node:path';
|
|
8
|
+
import { isReportFileName, reportFilePath, reportFileStem } from '../eval-core/artifact-file-names.js';
|
|
9
|
+
import { migrateLegacyReportFiles } from '../eval-core/report-file-migration.js';
|
|
8
10
|
// Per-id in-memory mutex for safe read-modify-write.
|
|
9
11
|
// Uses a queue to avoid the race window between checking and acquiring the lock.
|
|
10
12
|
const locks = new Map();
|
|
@@ -67,7 +69,7 @@ export function createFileStore(dir) {
|
|
|
67
69
|
function isEvaluationReport(report) {
|
|
68
70
|
return report.kind === 'evaluation';
|
|
69
71
|
}
|
|
70
|
-
// Studio 每个 / 和 /skills/<name> 请求都调 list(),里面对每个 .json 同步 readFile +
|
|
72
|
+
// Studio 每个 / 和 /skills/<name> 请求都调 list(),里面对每个 .report.json 同步 readFile +
|
|
71
73
|
// JSON.parse。报告数上来后这是主性能瓶颈。缓存策略:fingerprint = dir mtime + 文件名
|
|
72
74
|
// 排序串 + 每个文件 mtime;任一变化 invalidate。fingerprint 算 cheap(只 stat),命中后
|
|
73
75
|
// 完全跳过 readFile。
|
|
@@ -76,7 +78,7 @@ export function createFileStore(dir) {
|
|
|
76
78
|
async function computeListFingerprint() {
|
|
77
79
|
try {
|
|
78
80
|
const dirStat = await stat(dir);
|
|
79
|
-
const files = (await readdir(dir)).filter(
|
|
81
|
+
const files = (await readdir(dir)).filter(isReportFileName).sort();
|
|
80
82
|
const parts = await Promise.all(files.map(async (f) => {
|
|
81
83
|
try {
|
|
82
84
|
const s = await stat(join(dir, f));
|
|
@@ -99,18 +101,19 @@ export function createFileStore(dir) {
|
|
|
99
101
|
catch {
|
|
100
102
|
return [];
|
|
101
103
|
}
|
|
104
|
+
migrateLegacyReportFiles(dir, 'report');
|
|
102
105
|
const fp = await computeListFingerprint();
|
|
103
106
|
if (fp != null && fp === cachedFingerprint && cachedRuns)
|
|
104
107
|
return cachedRuns;
|
|
105
108
|
const files = (await readdir(dir))
|
|
106
|
-
.filter(
|
|
109
|
+
.filter(isReportFileName)
|
|
107
110
|
.sort()
|
|
108
111
|
.reverse();
|
|
109
112
|
const runs = [];
|
|
110
113
|
for (const file of files) {
|
|
111
114
|
try {
|
|
112
115
|
const data = JSON.parse(await readFile(join(dir, file), 'utf-8'));
|
|
113
|
-
const report = normalizeReportDocument(data, file
|
|
116
|
+
const report = normalizeReportDocument(data, reportFileStem(file) ?? file);
|
|
114
117
|
if (report)
|
|
115
118
|
runs.push(report);
|
|
116
119
|
}
|
|
@@ -128,8 +131,9 @@ export function createFileStore(dir) {
|
|
|
128
131
|
return runs;
|
|
129
132
|
}
|
|
130
133
|
async function get(id) {
|
|
134
|
+
migrateLegacyReportFiles(dir, 'report');
|
|
131
135
|
try {
|
|
132
|
-
const data = JSON.parse(await readFile(
|
|
136
|
+
const data = JSON.parse(await readFile(reportFilePath(dir, id), 'utf-8'));
|
|
133
137
|
return normalizeReportDocument(data, id);
|
|
134
138
|
}
|
|
135
139
|
catch {
|
|
@@ -138,9 +142,11 @@ export function createFileStore(dir) {
|
|
|
138
142
|
}
|
|
139
143
|
async function save(id, report) {
|
|
140
144
|
await ensureDir();
|
|
141
|
-
|
|
145
|
+
migrateLegacyReportFiles(dir, 'report');
|
|
146
|
+
const targetPath = reportFilePath(dir, id);
|
|
147
|
+
const tmpPath = `${targetPath}.tmp.${Date.now()}.${Math.random().toString(36).slice(2)}`;
|
|
142
148
|
await writeFile(tmpPath, JSON.stringify(report, null, 2));
|
|
143
|
-
await rename(tmpPath,
|
|
149
|
+
await rename(tmpPath, targetPath);
|
|
144
150
|
}
|
|
145
151
|
/**
|
|
146
152
|
* Atomic read-modify-write with in-memory mutex.
|
|
@@ -157,8 +163,9 @@ export function createFileStore(dir) {
|
|
|
157
163
|
});
|
|
158
164
|
}
|
|
159
165
|
async function remove(id) {
|
|
166
|
+
migrateLegacyReportFiles(dir, 'report');
|
|
160
167
|
try {
|
|
161
|
-
await unlink(
|
|
168
|
+
await unlink(reportFilePath(dir, id));
|
|
162
169
|
return true;
|
|
163
170
|
}
|
|
164
171
|
catch (err) {
|
|
@@ -169,8 +176,9 @@ export function createFileStore(dir) {
|
|
|
169
176
|
}
|
|
170
177
|
}
|
|
171
178
|
async function exists(id) {
|
|
179
|
+
migrateLegacyReportFiles(dir, 'report');
|
|
172
180
|
try {
|
|
173
|
-
await access(
|
|
181
|
+
await access(reportFilePath(dir, id));
|
|
174
182
|
return true;
|
|
175
183
|
}
|
|
176
184
|
catch {
|
|
@@ -9,14 +9,14 @@
|
|
|
9
9
|
* 因为同一份 EvaluationReport 同时驱动两个 view tab。
|
|
10
10
|
* - observe 报告(analysesDir, SkillHealthReport):data.bySkill[name] 每个键作为
|
|
11
11
|
* 一个 skill 的 observe snapshot,取最新 generatedAt。
|
|
12
|
-
* - doctor
|
|
13
|
-
* 该字段保留 null,渲染层显示"未独立运行 omk doctor"。后续 commit 加默认
|
|
14
|
-
* ~/.oh-my-knowledge/doctors/ 持久化路径再回填。
|
|
12
|
+
* - doctor:读取 `.omk/doctors/*.report.json`,按 skill 名聚合体检历史。
|
|
15
13
|
*
|
|
16
14
|
* 综合 band:eval / observe 任一红 → 红,任一黄 → 黄,全绿 → 绿,皆未跑 → gray。
|
|
17
15
|
*/
|
|
18
16
|
import { existsSync, readdirSync, readFileSync, statSync } from 'node:fs';
|
|
19
17
|
import { join, isAbsolute, basename, dirname } from 'node:path';
|
|
18
|
+
import { isReportFileName, reportFileStem } from '../eval-core/artifact-file-names.js';
|
|
19
|
+
import { migrateLegacyReportFiles } from '../eval-core/report-file-migration.js';
|
|
20
20
|
import { confidenceOf } from '../observability/skill-health-analyzer.js';
|
|
21
21
|
import { computeVerdict } from '../eval-core/verdict.js';
|
|
22
22
|
import { artifactIndexDir, listLiveDoctorCards, cardToDoctorSnapshot, listLiveObserveCards, cardTargetSentinel } from '../eval-core/artifact-index.js';
|
|
@@ -26,13 +26,13 @@ import { buildStudioDiagnosisSummary, mergeDiagnosisBundles } from '../diagnosis
|
|
|
26
26
|
let _indexCache = null;
|
|
27
27
|
/**
|
|
28
28
|
* Sync 版 dir-content fingerprint helper,仿 `src/server/report-store.ts:80-92`
|
|
29
|
-
* 的 async `computeListFingerprint` — 把目录本身的 mtime 跟目录下每个 `.json`
|
|
29
|
+
* 的 async `computeListFingerprint` — 把目录本身的 mtime 跟目录下每个 `.report.json`
|
|
30
30
|
* 文件的 "filename:mtimeMs:size" 三元组排序拼接成 stable 字符串作为 dir-level
|
|
31
31
|
* 的 content-aware fingerprint。
|
|
32
32
|
*
|
|
33
|
-
* - 目录下任何 .json 文件**新增 / 删除 / 重命名** → 目录 mtime 跟着变,且
|
|
33
|
+
* - 目录下任何 .report.json 文件**新增 / 删除 / 重命名** → 目录 mtime 跟着变,且
|
|
34
34
|
* sorted-filenames 列表变,字符串变 → cache invalidate。
|
|
35
|
-
* - 任何**已有同名 .json 文件被外部进程原地覆写内容** → 该文件自己的 mtimeMs
|
|
35
|
+
* - 任何**已有同名 .report.json 文件被外部进程原地覆写内容** → 该文件自己的 mtimeMs
|
|
36
36
|
* 跟通常情况下 size 都变(byte 长度跟内容相关),字符串里那一 entry 的后缀
|
|
37
37
|
* 变,整体字符串变 → cache invalidate。这是 pre-fix 的 fingerprint(只看
|
|
38
38
|
* dir mtime+文件数)漏掉的信号(reviewer 2026-05-11 P2-a)。
|
|
@@ -48,7 +48,7 @@ function safeDirJsonContentFingerprint(dir) {
|
|
|
48
48
|
let jsonFiles;
|
|
49
49
|
try {
|
|
50
50
|
dirMtimeMs = statSync(dir).mtimeMs;
|
|
51
|
-
jsonFiles = readdirSync(dir).filter(
|
|
51
|
+
jsonFiles = readdirSync(dir).filter(isReportFileName).sort();
|
|
52
52
|
}
|
|
53
53
|
catch {
|
|
54
54
|
return `missing:${dir}`;
|
|
@@ -71,6 +71,9 @@ function buildIndexFingerprint(reports, analysesDir, doctorsDir, observationsDir
|
|
|
71
71
|
// safeDirJsonContentFingerprint 返回的 "{dir-mtime}|{file1}:{m}:{s},..."
|
|
72
72
|
// content-aware 字符串。
|
|
73
73
|
const reportIds = reports.map((r) => `${r.id}:${r.meta?.timestamp ?? ''}:${r.kind === 'evaluation' ? r.meta.evolve?.skillName ?? '' : ''}`).join(',');
|
|
74
|
+
migrateLegacyReportFiles(doctorsDir, 'doctor');
|
|
75
|
+
migrateLegacyReportFiles(analysesDir, 'observe-health');
|
|
76
|
+
migrateLegacyReportFiles(observationsDir, 'observe-inbox');
|
|
74
77
|
const doctorsFp = safeDirJsonContentFingerprint(doctorsDir);
|
|
75
78
|
const analysesFp = safeDirJsonContentFingerprint(analysesDir);
|
|
76
79
|
const observationsFp = safeDirJsonContentFingerprint(observationsDir);
|
|
@@ -224,14 +227,15 @@ function latestEvalSnapshot(list) {
|
|
|
224
227
|
return null;
|
|
225
228
|
return list[list.length - 1];
|
|
226
229
|
}
|
|
227
|
-
/** 扫 doctorsDir/*.json,按 skill 名分桶,**返回该 skill 的所有历史 snapshot**(asc 时序)。
|
|
230
|
+
/** 扫 doctorsDir/*.report.json,按 skill 名分桶,**返回该 skill 的所有历史 snapshot**(asc 时序)。
|
|
228
231
|
* renderer 用最后一项做"当前",前面项画 sparkline。 */
|
|
229
232
|
function scanDoctorReports(dir) {
|
|
230
233
|
const out = {};
|
|
234
|
+
migrateLegacyReportFiles(dir, 'doctor');
|
|
231
235
|
if (!existsSync(dir))
|
|
232
236
|
return out;
|
|
233
237
|
for (const file of readdirSync(dir)) {
|
|
234
|
-
if (!file
|
|
238
|
+
if (!isReportFileName(file))
|
|
235
239
|
continue;
|
|
236
240
|
try {
|
|
237
241
|
const data = JSON.parse(readFileSync(join(dir, file), 'utf-8'));
|
|
@@ -288,16 +292,17 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
|
|
|
288
292
|
list.sort((a, b) => evalSnapshotSortKey(a).localeCompare(evalSnapshotSortKey(b)));
|
|
289
293
|
// ── observe 聚合(历史 list)──────────────────────────────
|
|
290
294
|
const observeBy = {};
|
|
295
|
+
migrateLegacyReportFiles(analysesDir, 'observe-health');
|
|
291
296
|
if (existsSync(analysesDir)) {
|
|
292
297
|
for (const file of readdirSync(analysesDir)) {
|
|
293
|
-
|
|
298
|
+
const id = reportFileStem(file);
|
|
299
|
+
if (!id)
|
|
294
300
|
continue;
|
|
295
301
|
try {
|
|
296
302
|
const data = JSON.parse(readFileSync(join(analysesDir, file), 'utf-8'));
|
|
297
303
|
if (!data?.bySkill || !data.meta)
|
|
298
304
|
continue;
|
|
299
305
|
const generatedAt = data.meta.generatedAt;
|
|
300
|
-
const id = file.replace(/\.json$/, '');
|
|
301
306
|
for (const [skill, h] of Object.entries(data.bySkill)) {
|
|
302
307
|
const snap = {
|
|
303
308
|
analysisId: id, generatedAt,
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
export interface ArtifactGraphDocument {
|
|
2
|
+
documentKind: 'artifact-graph';
|
|
3
|
+
schemaVersion: 1;
|
|
4
|
+
graphId: string;
|
|
5
|
+
generatedAt: string;
|
|
6
|
+
source: ArtifactGraphSource;
|
|
7
|
+
scope: ArtifactGraphScope;
|
|
8
|
+
nodes: ArtifactGraphNode[];
|
|
9
|
+
edges: ArtifactGraphEdge[];
|
|
10
|
+
summaries?: ArtifactGraphSummary[];
|
|
11
|
+
}
|
|
12
|
+
export type ArtifactGraphSourceKind = 'doctor' | 'eval' | 'observe';
|
|
13
|
+
export interface ArtifactGraphSource {
|
|
14
|
+
sourceKind: ArtifactGraphSourceKind;
|
|
15
|
+
sourceId: string;
|
|
16
|
+
sourcePath?: string;
|
|
17
|
+
cliVersion?: string;
|
|
18
|
+
}
|
|
19
|
+
export interface ArtifactGraphScope {
|
|
20
|
+
cwd: string;
|
|
21
|
+
artifactKind?: 'skill' | 'prompt' | 'agent' | 'workflow';
|
|
22
|
+
skillName?: string;
|
|
23
|
+
artifactHash?: string;
|
|
24
|
+
sourceLocator?: string;
|
|
25
|
+
sampleSetHash?: string;
|
|
26
|
+
}
|
|
27
|
+
export type ArtifactGraphLayer = 'definition' | 'measurement' | 'production';
|
|
28
|
+
export type ArtifactGraphNodeRole = 'entity' | 'observation' | 'aggregate';
|
|
29
|
+
export type ArtifactGraphNodeKind = 'skill' | 'skill_file' | 'frontmatter' | 'reference' | 'script' | 'tool' | 'env' | 'preflight' | 'hard_rule' | 'workflow' | 'workflow_node' | 'sample' | 'assertion' | 'variant' | 'doctor_rule_result' | 'eval_result' | 'judge_dimension' | 'diagnostic' | 'trace_session' | 'skill_invocation' | 'tool_call' | 'gap_signal';
|
|
30
|
+
export type ArtifactGraphStatus = 'ok' | 'warning' | 'failed' | 'skipped' | 'unknown' | 'not_measured';
|
|
31
|
+
export interface ArtifactGraphBinding {
|
|
32
|
+
bindingStrength: 'content-hash' | 'source-locator' | 'runtime-trace' | 'name-only' | 'aggregate';
|
|
33
|
+
keys: Record<string, string>;
|
|
34
|
+
}
|
|
35
|
+
export interface ArtifactGraphAttrs {
|
|
36
|
+
display?: Record<string, unknown>;
|
|
37
|
+
producer?: Record<string, unknown>;
|
|
38
|
+
experimental?: Record<string, unknown>;
|
|
39
|
+
}
|
|
40
|
+
export interface ArtifactGraphNode {
|
|
41
|
+
id: string;
|
|
42
|
+
stableKey: string;
|
|
43
|
+
nodeKind: ArtifactGraphNodeKind;
|
|
44
|
+
nodeRole: ArtifactGraphNodeRole;
|
|
45
|
+
layer: ArtifactGraphLayer;
|
|
46
|
+
label: string;
|
|
47
|
+
status?: ArtifactGraphStatus;
|
|
48
|
+
confidence?: number;
|
|
49
|
+
binding?: ArtifactGraphBinding;
|
|
50
|
+
metrics?: Record<string, number>;
|
|
51
|
+
attrs?: ArtifactGraphAttrs;
|
|
52
|
+
evidenceRefs?: ArtifactGraphEvidenceRef[];
|
|
53
|
+
}
|
|
54
|
+
export type ArtifactGraphEdgeKind = 'contains' | 'declares' | 'requires' | 'references' | 'defines_workflow' | 'next_step' | 'covers' | 'evaluates' | 'passes' | 'fails' | 'diagnoses' | 'invokes' | 'calls_tool' | 'observes' | 'signals_gap' | 'derived_from';
|
|
55
|
+
export interface ArtifactGraphEdge {
|
|
56
|
+
id: string;
|
|
57
|
+
fromNodeId: string;
|
|
58
|
+
toNodeId: string;
|
|
59
|
+
edgeKind: ArtifactGraphEdgeKind;
|
|
60
|
+
layer: ArtifactGraphLayer;
|
|
61
|
+
label?: string;
|
|
62
|
+
status?: ArtifactGraphStatus;
|
|
63
|
+
confidence?: number;
|
|
64
|
+
weight?: number;
|
|
65
|
+
binding?: ArtifactGraphBinding;
|
|
66
|
+
metrics?: Record<string, number>;
|
|
67
|
+
attrs?: ArtifactGraphAttrs;
|
|
68
|
+
evidenceRefs?: ArtifactGraphEvidenceRef[];
|
|
69
|
+
}
|
|
70
|
+
export type ArtifactGraphEvidenceSourceKind = 'skill-file' | 'doctor-report' | 'eval-report' | 'observe-report' | 'trace' | 'sample' | 'managed-record';
|
|
71
|
+
export type ArtifactGraphEvidenceSelectorKind = 'json-pointer' | 'line-range' | 'sample-id' | 'rule-id' | 'trace-event-id' | 'node-id';
|
|
72
|
+
export interface ArtifactGraphEvidenceSelector {
|
|
73
|
+
selectorKind: ArtifactGraphEvidenceSelectorKind;
|
|
74
|
+
value: string;
|
|
75
|
+
}
|
|
76
|
+
export interface ArtifactGraphEvidenceRef {
|
|
77
|
+
sourceKind: ArtifactGraphEvidenceSourceKind;
|
|
78
|
+
sourceId?: string;
|
|
79
|
+
path?: string;
|
|
80
|
+
selector?: ArtifactGraphEvidenceSelector;
|
|
81
|
+
contentHash?: string;
|
|
82
|
+
label?: string;
|
|
83
|
+
snippet?: string;
|
|
84
|
+
redaction?: 'none' | 'truncated' | 'redacted';
|
|
85
|
+
}
|
|
86
|
+
export interface ArtifactGraphSummary {
|
|
87
|
+
summaryKind: 'structure' | 'coverage' | 'risk' | 'gap' | 'workflow' | 'collection';
|
|
88
|
+
title: string;
|
|
89
|
+
severity: 'info' | 'low' | 'medium' | 'high';
|
|
90
|
+
nodeIds?: string[];
|
|
91
|
+
edgeIds?: string[];
|
|
92
|
+
evidenceRefs?: ArtifactGraphEvidenceRef[];
|
|
93
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
package/dist/types/eval.d.ts
CHANGED
|
@@ -220,6 +220,9 @@ export interface EvalConfig {
|
|
|
220
220
|
budget?: EvalBudget;
|
|
221
221
|
/** --repeat N. Multi-run variance analysis. */
|
|
222
222
|
repeat?: number;
|
|
223
|
+
/** --holdout-ratio R (0 < R < 1). Hold out a deterministic sample slice and
|
|
224
|
+
* report train vs holdout composite as a generalization / overfitting signal. */
|
|
225
|
+
holdoutRatio?: number;
|
|
223
226
|
/** --judge-repeat N. Each (sample × dimension) judged N times for self-consistency stddev. */
|
|
224
227
|
judgeRepeat?: number;
|
|
225
228
|
/** --bootstrap. Distribution-free CI per variant + pairwise diff. */
|
|
@@ -257,6 +260,10 @@ export interface EvaluationRequest {
|
|
|
257
260
|
dryRun: boolean;
|
|
258
261
|
/** --repeat N; 1 表示单次跑,> 1 走 runMultiple 做 variance 分析 */
|
|
259
262
|
repeat?: number;
|
|
263
|
+
/** --holdout-ratio R; 0 / 缺省表示不切分(默认)。> 0 时 report-finalize 在结果上
|
|
264
|
+
* post-hoc 切出 train / holdout 子集算综合分(`report.analysis.holdout`),供 verdict
|
|
265
|
+
* 的过拟合门控读取。see src/eval-core/holdout.ts */
|
|
266
|
+
holdoutRatio?: number;
|
|
260
267
|
/** --batch; default absent/false. True means skill-batch mode. */
|
|
261
268
|
batch?: boolean;
|
|
262
269
|
/** --judge-repeat N; 每条 sample × dimension 用 LLM judge 跑 N 次, 输出 stddev. 默认 1 (单次). */
|
package/dist/types/index.d.ts
CHANGED
|
@@ -6,6 +6,7 @@ export * from './report.js';
|
|
|
6
6
|
export * from './storage.js';
|
|
7
7
|
export * from './doctor.js';
|
|
8
8
|
export * from './diagnosis.js';
|
|
9
|
+
export * from './artifact-graph.js';
|
|
9
10
|
export * from './dependencies.js';
|
|
10
11
|
export * from './skill-index.js';
|
|
11
12
|
export * from './observability.js';
|
package/dist/types/index.js
CHANGED
|
@@ -6,6 +6,7 @@ export * from './report.js';
|
|
|
6
6
|
export * from './storage.js';
|
|
7
7
|
export * from './doctor.js';
|
|
8
8
|
export * from './diagnosis.js';
|
|
9
|
+
export * from './artifact-graph.js';
|
|
9
10
|
export * from './dependencies.js';
|
|
10
11
|
export * from './skill-index.js';
|
|
11
12
|
export * from './observability.js';
|
package/dist/types/report.d.ts
CHANGED
|
@@ -480,6 +480,34 @@ export interface AnalysisResult {
|
|
|
480
480
|
* (capability / difficulty / construct / provenance); persisted on report
|
|
481
481
|
* for studio to surface coverage gaps. See docs/specs/sample-design-spec.md. */
|
|
482
482
|
sampleQuality?: SampleQualityAggregate;
|
|
483
|
+
/** Opt-in train/holdout generalization breakdown (`omk eval --holdout-ratio`).
|
|
484
|
+
* Absent on default runs; present only when a holdout ratio was requested. */
|
|
485
|
+
holdout?: HoldoutBreakdown;
|
|
486
|
+
}
|
|
487
|
+
/** Train vs holdout composite breakdown for `omk eval --holdout-ratio`.
|
|
488
|
+
* Computed post-hoc from `report.results` by `computeHoldoutBreakdown`
|
|
489
|
+
* (`src/eval-core/holdout.ts`), sharing the same testSetHash watermark as
|
|
490
|
+
* gapReports (gap-spec §7.1). A large train − holdout composite gap is the
|
|
491
|
+
* sample-set-overfitting signal the verdict's overfitting gate reads. */
|
|
492
|
+
export interface HoldoutBreakdown {
|
|
493
|
+
/** Held-out fraction requested via --holdout-ratio. */
|
|
494
|
+
ratio: number;
|
|
495
|
+
/** true when either side fell below the minimum subset size → scored full-set,
|
|
496
|
+
* no usable split. `perVariant` is empty and the verdict gate stays inert. */
|
|
497
|
+
disabled?: boolean;
|
|
498
|
+
/** Per-variant train vs holdout composite (1-5 scale). `*Count` is the authored
|
|
499
|
+
* split size; `*Scorable` is how many of those actually produced a composite (> 0)
|
|
500
|
+
* — they diverge under partial errors, and the overfitting gate trusts `*Scorable`. */
|
|
501
|
+
perVariant: Record<string, {
|
|
502
|
+
trainScore: number;
|
|
503
|
+
holdoutScore: number;
|
|
504
|
+
trainCount: number;
|
|
505
|
+
holdoutCount: number;
|
|
506
|
+
trainScorable: number;
|
|
507
|
+
holdoutScorable: number;
|
|
508
|
+
}>;
|
|
509
|
+
testSetPath?: string | null;
|
|
510
|
+
testSetHash?: string | null;
|
|
483
511
|
}
|
|
484
512
|
/** Aggregated sample design coverage stats. Built by
|
|
485
513
|
* `buildSampleQualityAggregate(samples)` from `Sample.capability` /
|
|
@@ -502,6 +530,27 @@ export interface SampleQualityAggregate {
|
|
|
502
530
|
sampleCountWithDifficulty: number;
|
|
503
531
|
sampleCountWithConstruct: number;
|
|
504
532
|
sampleCountWithProvenance: number;
|
|
533
|
+
/** Relative-balance / skew of the sample set (derived from the distributions
|
|
534
|
+
* above). Flags over-representation — "70% of samples are easy" — without an
|
|
535
|
+
* external denominator. Diagnostic only; never feeds grading / judge / verdict. */
|
|
536
|
+
representativeness?: Representativeness;
|
|
537
|
+
}
|
|
538
|
+
/** Distribution skew over what the sample set declares. Pure relative balance —
|
|
539
|
+
* there is no authored "expected" capability list to measure absolute coverage
|
|
540
|
+
* against (capabilities are free-form strings), so this reports concentration
|
|
541
|
+
* (dominant bucket share, 0-1) and the dominant label per dimension. */
|
|
542
|
+
export interface Representativeness {
|
|
543
|
+
/** Distinct capabilities declared across the set. */
|
|
544
|
+
capabilityCount: number;
|
|
545
|
+
/** Dominant capability's share of all capability tags (0-1); 0 when none declared. */
|
|
546
|
+
capabilityConcentration: number;
|
|
547
|
+
dominantCapability?: string;
|
|
548
|
+
/** Dominant difficulty bucket's share of samples that declared a difficulty (0-1). */
|
|
549
|
+
difficultyConcentration: number;
|
|
550
|
+
dominantDifficulty?: 'easy' | 'medium' | 'hard';
|
|
551
|
+
/** Dominant construct's share of samples that declared a construct (0-1). */
|
|
552
|
+
constructConcentration: number;
|
|
553
|
+
dominantConstruct?: string;
|
|
505
554
|
}
|
|
506
555
|
export interface HedgingVerdict {
|
|
507
556
|
isUncertainty: boolean;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "oh-my-knowledge",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.43.0",
|
|
4
4
|
"packageManager": "yarn@4.16.0",
|
|
5
5
|
"description": "Evaluation framework for LLM knowledge inputs — prompts, RAG corpora, skills, agent workflows. Fix the model, vary the artifact. Built-in statistical rigor: bootstrap CI, Krippendorff α, length-debias, saturation curves.",
|
|
6
6
|
"type": "module",
|