oh-my-knowledge 0.41.0 → 0.43.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/README.md +7 -2
  2. package/README.zh.md +7 -2
  3. package/dist/analysis/report-diagnostics.d.ts +8 -1
  4. package/dist/analysis/report-diagnostics.js +82 -1
  5. package/dist/artifact-graph/doctor.d.ts +21 -0
  6. package/dist/artifact-graph/doctor.js +569 -0
  7. package/dist/assets/agent-skills/omk/SKILL.md +9 -9
  8. package/dist/assets/agent-skills/omk/references/commands.md +2 -1
  9. package/dist/authoring/evolver.d.ts +3 -14
  10. package/dist/authoring/evolver.js +1 -52
  11. package/dist/authoring/generator.d.ts +24 -0
  12. package/dist/authoring/generator.js +33 -6
  13. package/dist/cli/commands/doctor.js +62 -61
  14. package/dist/cli/commands/eval/index.d.ts +1 -0
  15. package/dist/cli/commands/eval/index.js +60 -7
  16. package/dist/cli/commands/init.js +11 -7
  17. package/dist/cli/commands/observe/index.d.ts +2 -2
  18. package/dist/cli/commands/observe/index.js +8 -7
  19. package/dist/cli/commands/sample.js +22 -16
  20. package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
  21. package/dist/cli/lib/i18n-dict/common.js +8 -0
  22. package/dist/cli/lib/i18n-dict/help.js +12 -10
  23. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  24. package/dist/cli/lib/i18n-dict/init.js +14 -11
  25. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  26. package/dist/cli/lib/i18n-dict/run.js +4 -0
  27. package/dist/cli/lib/parse-run-config/samples-discovery.d.ts +5 -7
  28. package/dist/cli/lib/parse-run-config/samples-discovery.js +10 -32
  29. package/dist/cli/lib/parse-run-config.d.ts +3 -0
  30. package/dist/cli/lib/parse-run-config.js +4 -4
  31. package/dist/cli/lib/resolve-skill-input.js +10 -12
  32. package/dist/doctor/messages.js +2 -2
  33. package/dist/eval-core/artifact-file-names.d.ts +15 -0
  34. package/dist/eval-core/artifact-file-names.js +46 -0
  35. package/dist/eval-core/evaluation-job.d.ts +2 -1
  36. package/dist/eval-core/evaluation-job.js +2 -1
  37. package/dist/eval-core/evaluation-reporting.js +9 -4
  38. package/dist/eval-core/holdout.d.ts +66 -0
  39. package/dist/eval-core/holdout.js +118 -0
  40. package/dist/eval-core/measurement-dirs.js +13 -7
  41. package/dist/eval-core/report-file-migration.d.ts +10 -0
  42. package/dist/eval-core/report-file-migration.js +90 -0
  43. package/dist/eval-core/verdict.d.ts +44 -1
  44. package/dist/eval-core/verdict.js +175 -13
  45. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -1
  46. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -1
  47. package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -1
  48. package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
  49. package/dist/eval-workflows/evaluation-pipeline.js +2 -1
  50. package/dist/eval-workflows/run-evaluation.d.ts +5 -2
  51. package/dist/eval-workflows/run-evaluation.js +8 -5
  52. package/dist/inputs/eval-config.js +6 -0
  53. package/dist/inputs/sample-locator.d.ts +23 -0
  54. package/dist/inputs/sample-locator.js +195 -0
  55. package/dist/inputs/skill-loader.js +7 -17
  56. package/dist/observability/inbox.js +7 -3
  57. package/dist/renderer/summary.js +36 -3
  58. package/dist/server/report-server.js +10 -4
  59. package/dist/server/report-store.js +17 -9
  60. package/dist/server/skill-index.js +16 -11
  61. package/dist/types/artifact-graph.d.ts +93 -0
  62. package/dist/types/artifact-graph.js +1 -0
  63. package/dist/types/eval.d.ts +7 -0
  64. package/dist/types/index.d.ts +1 -0
  65. package/dist/types/index.js +1 -0
  66. package/dist/types/report.d.ts +49 -0
  67. package/package.json +1 -1
@@ -5,6 +5,8 @@
5
5
  */
6
6
  import { readdir, readFile, writeFile, unlink, access, mkdir, rename, stat } from 'node:fs/promises';
7
7
  import { join } from 'node:path';
8
+ import { isReportFileName, reportFilePath, reportFileStem } from '../eval-core/artifact-file-names.js';
9
+ import { migrateLegacyReportFiles } from '../eval-core/report-file-migration.js';
8
10
  // Per-id in-memory mutex for safe read-modify-write.
9
11
  // Uses a queue to avoid the race window between checking and acquiring the lock.
10
12
  const locks = new Map();
@@ -67,7 +69,7 @@ export function createFileStore(dir) {
67
69
  function isEvaluationReport(report) {
68
70
  return report.kind === 'evaluation';
69
71
  }
70
- // Studio 每个 / 和 /skills/<name> 请求都调 list(),里面对每个 .json 同步 readFile +
72
+ // Studio 每个 / 和 /skills/<name> 请求都调 list(),里面对每个 .report.json 同步 readFile +
71
73
  // JSON.parse。报告数上来后这是主性能瓶颈。缓存策略:fingerprint = dir mtime + 文件名
72
74
  // 排序串 + 每个文件 mtime;任一变化 invalidate。fingerprint 算 cheap(只 stat),命中后
73
75
  // 完全跳过 readFile。
@@ -76,7 +78,7 @@ export function createFileStore(dir) {
76
78
  async function computeListFingerprint() {
77
79
  try {
78
80
  const dirStat = await stat(dir);
79
- const files = (await readdir(dir)).filter((f) => f.endsWith('.json')).sort();
81
+ const files = (await readdir(dir)).filter(isReportFileName).sort();
80
82
  const parts = await Promise.all(files.map(async (f) => {
81
83
  try {
82
84
  const s = await stat(join(dir, f));
@@ -99,18 +101,19 @@ export function createFileStore(dir) {
99
101
  catch {
100
102
  return [];
101
103
  }
104
+ migrateLegacyReportFiles(dir, 'report');
102
105
  const fp = await computeListFingerprint();
103
106
  if (fp != null && fp === cachedFingerprint && cachedRuns)
104
107
  return cachedRuns;
105
108
  const files = (await readdir(dir))
106
- .filter((f) => f.endsWith('.json'))
109
+ .filter(isReportFileName)
107
110
  .sort()
108
111
  .reverse();
109
112
  const runs = [];
110
113
  for (const file of files) {
111
114
  try {
112
115
  const data = JSON.parse(await readFile(join(dir, file), 'utf-8'));
113
- const report = normalizeReportDocument(data, file.replace(/\.json$/, ''));
116
+ const report = normalizeReportDocument(data, reportFileStem(file) ?? file);
114
117
  if (report)
115
118
  runs.push(report);
116
119
  }
@@ -128,8 +131,9 @@ export function createFileStore(dir) {
128
131
  return runs;
129
132
  }
130
133
  async function get(id) {
134
+ migrateLegacyReportFiles(dir, 'report');
131
135
  try {
132
- const data = JSON.parse(await readFile(join(dir, `${id}.json`), 'utf-8'));
136
+ const data = JSON.parse(await readFile(reportFilePath(dir, id), 'utf-8'));
133
137
  return normalizeReportDocument(data, id);
134
138
  }
135
139
  catch {
@@ -138,9 +142,11 @@ export function createFileStore(dir) {
138
142
  }
139
143
  async function save(id, report) {
140
144
  await ensureDir();
141
- const tmpPath = join(dir, `${id}.json.tmp.${Date.now()}.${Math.random().toString(36).slice(2)}`);
145
+ migrateLegacyReportFiles(dir, 'report');
146
+ const targetPath = reportFilePath(dir, id);
147
+ const tmpPath = `${targetPath}.tmp.${Date.now()}.${Math.random().toString(36).slice(2)}`;
142
148
  await writeFile(tmpPath, JSON.stringify(report, null, 2));
143
- await rename(tmpPath, join(dir, `${id}.json`));
149
+ await rename(tmpPath, targetPath);
144
150
  }
145
151
  /**
146
152
  * Atomic read-modify-write with in-memory mutex.
@@ -157,8 +163,9 @@ export function createFileStore(dir) {
157
163
  });
158
164
  }
159
165
  async function remove(id) {
166
+ migrateLegacyReportFiles(dir, 'report');
160
167
  try {
161
- await unlink(join(dir, `${id}.json`));
168
+ await unlink(reportFilePath(dir, id));
162
169
  return true;
163
170
  }
164
171
  catch (err) {
@@ -169,8 +176,9 @@ export function createFileStore(dir) {
169
176
  }
170
177
  }
171
178
  async function exists(id) {
179
+ migrateLegacyReportFiles(dir, 'report');
172
180
  try {
173
- await access(join(dir, `${id}.json`));
181
+ await access(reportFilePath(dir, id));
174
182
  return true;
175
183
  }
176
184
  catch {
@@ -9,14 +9,14 @@
9
9
  * 因为同一份 EvaluationReport 同时驱动两个 view tab。
10
10
  * - observe 报告(analysesDir, SkillHealthReport):data.bySkill[name] 每个键作为
11
11
  * 一个 skill 的 observe snapshot,取最新 generatedAt。
12
- * - doctor:暂无独立持久化路径(omk doctor --html 写指定路径,默认不存盘),
13
- * 该字段保留 null,渲染层显示"未独立运行 omk doctor"。后续 commit 加默认
14
- * ~/.oh-my-knowledge/doctors/ 持久化路径再回填。
12
+ * - doctor:读取 `.omk/doctors/*.report.json`,按 skill 名聚合体检历史。
15
13
  *
16
14
  * 综合 band:eval / observe 任一红 → 红,任一黄 → 黄,全绿 → 绿,皆未跑 → gray。
17
15
  */
18
16
  import { existsSync, readdirSync, readFileSync, statSync } from 'node:fs';
19
17
  import { join, isAbsolute, basename, dirname } from 'node:path';
18
+ import { isReportFileName, reportFileStem } from '../eval-core/artifact-file-names.js';
19
+ import { migrateLegacyReportFiles } from '../eval-core/report-file-migration.js';
20
20
  import { confidenceOf } from '../observability/skill-health-analyzer.js';
21
21
  import { computeVerdict } from '../eval-core/verdict.js';
22
22
  import { artifactIndexDir, listLiveDoctorCards, cardToDoctorSnapshot, listLiveObserveCards, cardTargetSentinel } from '../eval-core/artifact-index.js';
@@ -26,13 +26,13 @@ import { buildStudioDiagnosisSummary, mergeDiagnosisBundles } from '../diagnosis
26
26
  let _indexCache = null;
27
27
  /**
28
28
  * Sync 版 dir-content fingerprint helper,仿 `src/server/report-store.ts:80-92`
29
- * 的 async `computeListFingerprint` — 把目录本身的 mtime 跟目录下每个 `.json`
29
+ * 的 async `computeListFingerprint` — 把目录本身的 mtime 跟目录下每个 `.report.json`
30
30
  * 文件的 "filename:mtimeMs:size" 三元组排序拼接成 stable 字符串作为 dir-level
31
31
  * 的 content-aware fingerprint。
32
32
  *
33
- * - 目录下任何 .json 文件**新增 / 删除 / 重命名** → 目录 mtime 跟着变,且
33
+ * - 目录下任何 .report.json 文件**新增 / 删除 / 重命名** → 目录 mtime 跟着变,且
34
34
  * sorted-filenames 列表变,字符串变 → cache invalidate。
35
- * - 任何**已有同名 .json 文件被外部进程原地覆写内容** → 该文件自己的 mtimeMs
35
+ * - 任何**已有同名 .report.json 文件被外部进程原地覆写内容** → 该文件自己的 mtimeMs
36
36
  * 跟通常情况下 size 都变(byte 长度跟内容相关),字符串里那一 entry 的后缀
37
37
  * 变,整体字符串变 → cache invalidate。这是 pre-fix 的 fingerprint(只看
38
38
  * dir mtime+文件数)漏掉的信号(reviewer 2026-05-11 P2-a)。
@@ -48,7 +48,7 @@ function safeDirJsonContentFingerprint(dir) {
48
48
  let jsonFiles;
49
49
  try {
50
50
  dirMtimeMs = statSync(dir).mtimeMs;
51
- jsonFiles = readdirSync(dir).filter((f) => f.endsWith('.json')).sort();
51
+ jsonFiles = readdirSync(dir).filter(isReportFileName).sort();
52
52
  }
53
53
  catch {
54
54
  return `missing:${dir}`;
@@ -71,6 +71,9 @@ function buildIndexFingerprint(reports, analysesDir, doctorsDir, observationsDir
71
71
  // safeDirJsonContentFingerprint 返回的 "{dir-mtime}|{file1}:{m}:{s},..."
72
72
  // content-aware 字符串。
73
73
  const reportIds = reports.map((r) => `${r.id}:${r.meta?.timestamp ?? ''}:${r.kind === 'evaluation' ? r.meta.evolve?.skillName ?? '' : ''}`).join(',');
74
+ migrateLegacyReportFiles(doctorsDir, 'doctor');
75
+ migrateLegacyReportFiles(analysesDir, 'observe-health');
76
+ migrateLegacyReportFiles(observationsDir, 'observe-inbox');
74
77
  const doctorsFp = safeDirJsonContentFingerprint(doctorsDir);
75
78
  const analysesFp = safeDirJsonContentFingerprint(analysesDir);
76
79
  const observationsFp = safeDirJsonContentFingerprint(observationsDir);
@@ -224,14 +227,15 @@ function latestEvalSnapshot(list) {
224
227
  return null;
225
228
  return list[list.length - 1];
226
229
  }
227
- /** 扫 doctorsDir/*.json,按 skill 名分桶,**返回该 skill 的所有历史 snapshot**(asc 时序)。
230
+ /** 扫 doctorsDir/*.report.json,按 skill 名分桶,**返回该 skill 的所有历史 snapshot**(asc 时序)。
228
231
  * renderer 用最后一项做"当前",前面项画 sparkline。 */
229
232
  function scanDoctorReports(dir) {
230
233
  const out = {};
234
+ migrateLegacyReportFiles(dir, 'doctor');
231
235
  if (!existsSync(dir))
232
236
  return out;
233
237
  for (const file of readdirSync(dir)) {
234
- if (!file.endsWith('.json'))
238
+ if (!isReportFileName(file))
235
239
  continue;
236
240
  try {
237
241
  const data = JSON.parse(readFileSync(join(dir, file), 'utf-8'));
@@ -288,16 +292,17 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
288
292
  list.sort((a, b) => evalSnapshotSortKey(a).localeCompare(evalSnapshotSortKey(b)));
289
293
  // ── observe 聚合(历史 list)──────────────────────────────
290
294
  const observeBy = {};
295
+ migrateLegacyReportFiles(analysesDir, 'observe-health');
291
296
  if (existsSync(analysesDir)) {
292
297
  for (const file of readdirSync(analysesDir)) {
293
- if (!file.endsWith('.json'))
298
+ const id = reportFileStem(file);
299
+ if (!id)
294
300
  continue;
295
301
  try {
296
302
  const data = JSON.parse(readFileSync(join(analysesDir, file), 'utf-8'));
297
303
  if (!data?.bySkill || !data.meta)
298
304
  continue;
299
305
  const generatedAt = data.meta.generatedAt;
300
- const id = file.replace(/\.json$/, '');
301
306
  for (const [skill, h] of Object.entries(data.bySkill)) {
302
307
  const snap = {
303
308
  analysisId: id, generatedAt,
@@ -0,0 +1,93 @@
1
+ export interface ArtifactGraphDocument {
2
+ documentKind: 'artifact-graph';
3
+ schemaVersion: 1;
4
+ graphId: string;
5
+ generatedAt: string;
6
+ source: ArtifactGraphSource;
7
+ scope: ArtifactGraphScope;
8
+ nodes: ArtifactGraphNode[];
9
+ edges: ArtifactGraphEdge[];
10
+ summaries?: ArtifactGraphSummary[];
11
+ }
12
+ export type ArtifactGraphSourceKind = 'doctor' | 'eval' | 'observe';
13
+ export interface ArtifactGraphSource {
14
+ sourceKind: ArtifactGraphSourceKind;
15
+ sourceId: string;
16
+ sourcePath?: string;
17
+ cliVersion?: string;
18
+ }
19
+ export interface ArtifactGraphScope {
20
+ cwd: string;
21
+ artifactKind?: 'skill' | 'prompt' | 'agent' | 'workflow';
22
+ skillName?: string;
23
+ artifactHash?: string;
24
+ sourceLocator?: string;
25
+ sampleSetHash?: string;
26
+ }
27
+ export type ArtifactGraphLayer = 'definition' | 'measurement' | 'production';
28
+ export type ArtifactGraphNodeRole = 'entity' | 'observation' | 'aggregate';
29
+ export type ArtifactGraphNodeKind = 'skill' | 'skill_file' | 'frontmatter' | 'reference' | 'script' | 'tool' | 'env' | 'preflight' | 'hard_rule' | 'workflow' | 'workflow_node' | 'sample' | 'assertion' | 'variant' | 'doctor_rule_result' | 'eval_result' | 'judge_dimension' | 'diagnostic' | 'trace_session' | 'skill_invocation' | 'tool_call' | 'gap_signal';
30
+ export type ArtifactGraphStatus = 'ok' | 'warning' | 'failed' | 'skipped' | 'unknown' | 'not_measured';
31
+ export interface ArtifactGraphBinding {
32
+ bindingStrength: 'content-hash' | 'source-locator' | 'runtime-trace' | 'name-only' | 'aggregate';
33
+ keys: Record<string, string>;
34
+ }
35
+ export interface ArtifactGraphAttrs {
36
+ display?: Record<string, unknown>;
37
+ producer?: Record<string, unknown>;
38
+ experimental?: Record<string, unknown>;
39
+ }
40
+ export interface ArtifactGraphNode {
41
+ id: string;
42
+ stableKey: string;
43
+ nodeKind: ArtifactGraphNodeKind;
44
+ nodeRole: ArtifactGraphNodeRole;
45
+ layer: ArtifactGraphLayer;
46
+ label: string;
47
+ status?: ArtifactGraphStatus;
48
+ confidence?: number;
49
+ binding?: ArtifactGraphBinding;
50
+ metrics?: Record<string, number>;
51
+ attrs?: ArtifactGraphAttrs;
52
+ evidenceRefs?: ArtifactGraphEvidenceRef[];
53
+ }
54
+ export type ArtifactGraphEdgeKind = 'contains' | 'declares' | 'requires' | 'references' | 'defines_workflow' | 'next_step' | 'covers' | 'evaluates' | 'passes' | 'fails' | 'diagnoses' | 'invokes' | 'calls_tool' | 'observes' | 'signals_gap' | 'derived_from';
55
+ export interface ArtifactGraphEdge {
56
+ id: string;
57
+ fromNodeId: string;
58
+ toNodeId: string;
59
+ edgeKind: ArtifactGraphEdgeKind;
60
+ layer: ArtifactGraphLayer;
61
+ label?: string;
62
+ status?: ArtifactGraphStatus;
63
+ confidence?: number;
64
+ weight?: number;
65
+ binding?: ArtifactGraphBinding;
66
+ metrics?: Record<string, number>;
67
+ attrs?: ArtifactGraphAttrs;
68
+ evidenceRefs?: ArtifactGraphEvidenceRef[];
69
+ }
70
+ export type ArtifactGraphEvidenceSourceKind = 'skill-file' | 'doctor-report' | 'eval-report' | 'observe-report' | 'trace' | 'sample' | 'managed-record';
71
+ export type ArtifactGraphEvidenceSelectorKind = 'json-pointer' | 'line-range' | 'sample-id' | 'rule-id' | 'trace-event-id' | 'node-id';
72
+ export interface ArtifactGraphEvidenceSelector {
73
+ selectorKind: ArtifactGraphEvidenceSelectorKind;
74
+ value: string;
75
+ }
76
+ export interface ArtifactGraphEvidenceRef {
77
+ sourceKind: ArtifactGraphEvidenceSourceKind;
78
+ sourceId?: string;
79
+ path?: string;
80
+ selector?: ArtifactGraphEvidenceSelector;
81
+ contentHash?: string;
82
+ label?: string;
83
+ snippet?: string;
84
+ redaction?: 'none' | 'truncated' | 'redacted';
85
+ }
86
+ export interface ArtifactGraphSummary {
87
+ summaryKind: 'structure' | 'coverage' | 'risk' | 'gap' | 'workflow' | 'collection';
88
+ title: string;
89
+ severity: 'info' | 'low' | 'medium' | 'high';
90
+ nodeIds?: string[];
91
+ edgeIds?: string[];
92
+ evidenceRefs?: ArtifactGraphEvidenceRef[];
93
+ }
@@ -0,0 +1 @@
1
+ export {};
@@ -220,6 +220,9 @@ export interface EvalConfig {
220
220
  budget?: EvalBudget;
221
221
  /** --repeat N. Multi-run variance analysis. */
222
222
  repeat?: number;
223
+ /** --holdout-ratio R (0 < R < 1). Hold out a deterministic sample slice and
224
+ * report train vs holdout composite as a generalization / overfitting signal. */
225
+ holdoutRatio?: number;
223
226
  /** --judge-repeat N. Each (sample × dimension) judged N times for self-consistency stddev. */
224
227
  judgeRepeat?: number;
225
228
  /** --bootstrap. Distribution-free CI per variant + pairwise diff. */
@@ -257,6 +260,10 @@ export interface EvaluationRequest {
257
260
  dryRun: boolean;
258
261
  /** --repeat N; 1 表示单次跑,> 1 走 runMultiple 做 variance 分析 */
259
262
  repeat?: number;
263
+ /** --holdout-ratio R; 0 / 缺省表示不切分(默认)。> 0 时 report-finalize 在结果上
264
+ * post-hoc 切出 train / holdout 子集算综合分(`report.analysis.holdout`),供 verdict
265
+ * 的过拟合门控读取。see src/eval-core/holdout.ts */
266
+ holdoutRatio?: number;
260
267
  /** --batch; default absent/false. True means skill-batch mode. */
261
268
  batch?: boolean;
262
269
  /** --judge-repeat N; 每条 sample × dimension 用 LLM judge 跑 N 次, 输出 stddev. 默认 1 (单次). */
@@ -6,6 +6,7 @@ export * from './report.js';
6
6
  export * from './storage.js';
7
7
  export * from './doctor.js';
8
8
  export * from './diagnosis.js';
9
+ export * from './artifact-graph.js';
9
10
  export * from './dependencies.js';
10
11
  export * from './skill-index.js';
11
12
  export * from './observability.js';
@@ -6,6 +6,7 @@ export * from './report.js';
6
6
  export * from './storage.js';
7
7
  export * from './doctor.js';
8
8
  export * from './diagnosis.js';
9
+ export * from './artifact-graph.js';
9
10
  export * from './dependencies.js';
10
11
  export * from './skill-index.js';
11
12
  export * from './observability.js';
@@ -480,6 +480,34 @@ export interface AnalysisResult {
480
480
  * (capability / difficulty / construct / provenance); persisted on report
481
481
  * for studio to surface coverage gaps. See docs/specs/sample-design-spec.md. */
482
482
  sampleQuality?: SampleQualityAggregate;
483
+ /** Opt-in train/holdout generalization breakdown (`omk eval --holdout-ratio`).
484
+ * Absent on default runs; present only when a holdout ratio was requested. */
485
+ holdout?: HoldoutBreakdown;
486
+ }
487
+ /** Train vs holdout composite breakdown for `omk eval --holdout-ratio`.
488
+ * Computed post-hoc from `report.results` by `computeHoldoutBreakdown`
489
+ * (`src/eval-core/holdout.ts`), sharing the same testSetHash watermark as
490
+ * gapReports (gap-spec §7.1). A large train − holdout composite gap is the
491
+ * sample-set-overfitting signal the verdict's overfitting gate reads. */
492
+ export interface HoldoutBreakdown {
493
+ /** Held-out fraction requested via --holdout-ratio. */
494
+ ratio: number;
495
+ /** true when either side fell below the minimum subset size → scored full-set,
496
+ * no usable split. `perVariant` is empty and the verdict gate stays inert. */
497
+ disabled?: boolean;
498
+ /** Per-variant train vs holdout composite (1-5 scale). `*Count` is the authored
499
+ * split size; `*Scorable` is how many of those actually produced a composite (> 0)
500
+ * — they diverge under partial errors, and the overfitting gate trusts `*Scorable`. */
501
+ perVariant: Record<string, {
502
+ trainScore: number;
503
+ holdoutScore: number;
504
+ trainCount: number;
505
+ holdoutCount: number;
506
+ trainScorable: number;
507
+ holdoutScorable: number;
508
+ }>;
509
+ testSetPath?: string | null;
510
+ testSetHash?: string | null;
483
511
  }
484
512
  /** Aggregated sample design coverage stats. Built by
485
513
  * `buildSampleQualityAggregate(samples)` from `Sample.capability` /
@@ -502,6 +530,27 @@ export interface SampleQualityAggregate {
502
530
  sampleCountWithDifficulty: number;
503
531
  sampleCountWithConstruct: number;
504
532
  sampleCountWithProvenance: number;
533
+ /** Relative-balance / skew of the sample set (derived from the distributions
534
+ * above). Flags over-representation — "70% of samples are easy" — without an
535
+ * external denominator. Diagnostic only; never feeds grading / judge / verdict. */
536
+ representativeness?: Representativeness;
537
+ }
538
+ /** Distribution skew over what the sample set declares. Pure relative balance —
539
+ * there is no authored "expected" capability list to measure absolute coverage
540
+ * against (capabilities are free-form strings), so this reports concentration
541
+ * (dominant bucket share, 0-1) and the dominant label per dimension. */
542
+ export interface Representativeness {
543
+ /** Distinct capabilities declared across the set. */
544
+ capabilityCount: number;
545
+ /** Dominant capability's share of all capability tags (0-1); 0 when none declared. */
546
+ capabilityConcentration: number;
547
+ dominantCapability?: string;
548
+ /** Dominant difficulty bucket's share of samples that declared a difficulty (0-1). */
549
+ difficultyConcentration: number;
550
+ dominantDifficulty?: 'easy' | 'medium' | 'hard';
551
+ /** Dominant construct's share of samples that declared a construct (0-1). */
552
+ constructConcentration: number;
553
+ dominantConstruct?: string;
505
554
  }
506
555
  export interface HedgingVerdict {
507
556
  isUncertainty: boolean;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "oh-my-knowledge",
3
- "version": "0.41.0",
3
+ "version": "0.43.0",
4
4
  "packageManager": "yarn@4.16.0",
5
5
  "description": "Evaluation framework for LLM knowledge inputs — prompts, RAG corpora, skills, agent workflows. Fix the model, vary the artifact. Built-in statistical rigor: bootstrap CI, Krippendorff α, length-debias, saturation curves.",
6
6
  "type": "module",