oh-my-knowledge 0.34.0 → 0.36.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -0
- package/README.zh.md +3 -0
- package/dist/assets/agent-skills/omk/SKILL.md +2 -2
- package/dist/assets/agent-skills/omk/references/commands.md +45 -0
- package/dist/authoring/evolver.js +13 -4
- package/dist/cli/commands/doctor.js +2 -1
- package/dist/cli/commands/eval/index.d.ts +1 -0
- package/dist/cli/commands/eval/index.js +40 -5
- package/dist/cli/commands/install.d.ts +2 -0
- package/dist/cli/commands/install.js +45 -20
- package/dist/cli/commands/list.d.ts +46 -0
- package/dist/cli/commands/list.js +243 -0
- package/dist/cli/commands/sample.d.ts +1 -1
- package/dist/cli/commands/sample.js +25 -12
- package/dist/cli/lib/cmd-flags.d.ts +1 -0
- package/dist/cli/lib/i18n-dict/install.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/install.js +20 -0
- package/dist/cli/lib/i18n-dict/list.d.ts +3 -0
- package/dist/cli/lib/i18n-dict/list.js +32 -0
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +8 -0
- package/dist/cli/lib/i18n-dict.d.ts +2 -1
- package/dist/cli/lib/i18n-dict.js +2 -0
- package/dist/cli/lib/parse-run-config/variant-resolution.js +7 -0
- package/dist/cli/lib/run-tally.js +1 -1
- package/dist/cli/lib/shared.js +1 -1
- package/dist/doctor/index.js +6 -6
- package/dist/eval-core/cache.d.ts +11 -3
- package/dist/eval-core/cache.js +13 -5
- package/dist/eval-core/dependency-checker.js +5 -3
- package/dist/eval-core/evaluation-execution.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +1 -0
- package/dist/eval-core/evaluation-reporting.js +20 -9
- package/dist/eval-core/execution-strategy.js +6 -4
- package/dist/eval-core/task-planner.js +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +3 -2
- package/dist/eval-workflows/evaluation-preparation.js +3 -1
- package/dist/eval-workflows/run-evaluation.js +2 -2
- package/dist/inputs/content-hash.d.ts +28 -0
- package/dist/inputs/content-hash.js +106 -0
- package/dist/inputs/eval-config.js +44 -11
- package/dist/inputs/materialize-copy.d.ts +37 -0
- package/dist/inputs/materialize-copy.js +193 -0
- package/dist/inputs/skill-loader.d.ts +65 -2
- package/dist/inputs/skill-loader.js +314 -16
- package/dist/inputs/source-resolver.d.ts +16 -9
- package/dist/inputs/source-resolver.js +57 -77
- package/dist/managed/evidence.d.ts +22 -0
- package/dist/managed/evidence.js +143 -0
- package/dist/managed/index.d.ts +2 -0
- package/dist/managed/index.js +2 -0
- package/dist/managed/list-view.d.ts +55 -0
- package/dist/managed/list-view.js +66 -0
- package/dist/managed/store.d.ts +16 -20
- package/dist/managed/store.js +85 -84
- package/dist/observability/experience.d.ts +2 -0
- package/dist/observability/experience.js +31 -2
- package/dist/observability/inbox.js +31 -5
- package/dist/observability/review-state.js +22 -11
- package/dist/observability/soft-standards/llm-extractor.js +3 -3
- package/dist/observability/soft-standards/skill-standards-store.d.ts +1 -0
- package/dist/observability/soft-standards/skill-standards-store.js +77 -14
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/renderer/html-renderer.js +3 -3
- package/dist/renderer/layout.js +2 -2
- package/dist/server/report-server.js +1 -1
- package/dist/server/report-store.d.ts +1 -1
- package/dist/server/report-store.js +16 -15
- package/dist/server/skill-index.js +5 -4
- package/dist/types/doctor.d.ts +2 -2
- package/dist/types/doctor.js +1 -1
- package/dist/types/eval.d.ts +12 -1
- package/dist/types/managed.d.ts +29 -4
- package/dist/types/observability.d.ts +6 -6
- package/dist/types/report.d.ts +24 -5
- package/package.json +3 -3
|
@@ -39,32 +39,33 @@ export function createFileStore(dir) {
|
|
|
39
39
|
if (!data || typeof data !== 'object')
|
|
40
40
|
return null;
|
|
41
41
|
const record = data;
|
|
42
|
-
|
|
42
|
+
const kind = record.kind === 'evaluation' || record.kind === 'batch-evaluation'
|
|
43
|
+
? record.kind
|
|
44
|
+
: undefined;
|
|
45
|
+
if (kind === 'evaluation') {
|
|
43
46
|
if (!record.meta || !record.summary || !Array.isArray(record.results))
|
|
44
47
|
return null;
|
|
45
|
-
return {
|
|
48
|
+
return {
|
|
49
|
+
...record,
|
|
50
|
+
kind,
|
|
51
|
+
id: typeof record.id === 'string' && record.id ? record.id : fallbackId,
|
|
52
|
+
};
|
|
46
53
|
}
|
|
47
|
-
if (
|
|
54
|
+
if (kind === 'batch-evaluation') {
|
|
48
55
|
if (!record.meta || !Array.isArray(record.items))
|
|
49
56
|
return null;
|
|
50
|
-
return { ...record, id: typeof record.id === 'string' && record.id ? record.id : fallbackId };
|
|
51
|
-
}
|
|
52
|
-
if (record.kind === undefined
|
|
53
|
-
&& record.overview === undefined
|
|
54
|
-
&& record.artifacts === undefined
|
|
55
|
-
&& record.meta
|
|
56
|
-
&& record.summary
|
|
57
|
-
&& Array.isArray(record.results)) {
|
|
58
57
|
return {
|
|
59
58
|
...record,
|
|
60
|
-
|
|
59
|
+
kind,
|
|
61
60
|
id: typeof record.id === 'string' && record.id ? record.id : fallbackId,
|
|
62
61
|
};
|
|
63
62
|
}
|
|
63
|
+
// 只认 canonical 顶层 `kind`(evaluation / batch-evaluation)。不再为旧格式(顶层无该判别字段的
|
|
64
|
+
// 历史文件)做读兼容 —— 顶层 kind cutover 是硬切换,旧文件直接判脏丢弃。
|
|
64
65
|
return null;
|
|
65
66
|
}
|
|
66
67
|
function isEvaluationReport(report) {
|
|
67
|
-
return report.
|
|
68
|
+
return report.kind === 'evaluation';
|
|
68
69
|
}
|
|
69
70
|
// Studio 每个 / 和 /skills/<name> 请求都调 list(),里面对每个 .json 同步 readFile +
|
|
70
71
|
// JSON.parse。报告数上来后这是主性能瓶颈。缓存策略:fingerprint = dir mtime + 文件名
|
|
@@ -217,9 +218,9 @@ export async function queryJob(jobStore, id) {
|
|
|
217
218
|
export async function queryRunList(reportStore) {
|
|
218
219
|
return (await reportStore.list()).map((report) => ({
|
|
219
220
|
id: report.id,
|
|
220
|
-
|
|
221
|
+
kind: report.kind,
|
|
221
222
|
meta: report.meta,
|
|
222
|
-
...(report.
|
|
223
|
+
...(report.kind === 'evaluation' ? { summary: report.summary } : { items: report.items }),
|
|
223
224
|
}));
|
|
224
225
|
}
|
|
225
226
|
export async function queryRun(reportStore, id) {
|
|
@@ -69,7 +69,7 @@ function buildIndexFingerprint(reports, analysesDir, doctorsDir, observationsDir
|
|
|
69
69
|
// 每一段的 right-hand-side 从旧的 "{dir-mtime}-{file-count}" 双标量升级成
|
|
70
70
|
// safeDirJsonContentFingerprint 返回的 "{dir-mtime}|{file1}:{m}:{s},..."
|
|
71
71
|
// content-aware 字符串。
|
|
72
|
-
const reportIds = reports.map((r) => `${r.id}:${r.meta?.timestamp ?? ''}:${r.
|
|
72
|
+
const reportIds = reports.map((r) => `${r.id}:${r.meta?.timestamp ?? ''}:${r.kind === 'evaluation' ? r.meta.evolve?.skillName ?? '' : ''}`).join(',');
|
|
73
73
|
const doctorsFp = safeDirJsonContentFingerprint(doctorsDir);
|
|
74
74
|
const analysesFp = safeDirJsonContentFingerprint(analysesDir);
|
|
75
75
|
const observationsFp = safeDirJsonContentFingerprint(observationsDir);
|
|
@@ -223,7 +223,8 @@ function scanDoctorReports(dir) {
|
|
|
223
223
|
continue;
|
|
224
224
|
try {
|
|
225
225
|
const data = JSON.parse(readFileSync(join(dir, file), 'utf-8'));
|
|
226
|
-
|
|
226
|
+
const kind = data?.kind === 'doctor' ? data.kind : null;
|
|
227
|
+
if (!kind || !Array.isArray(data.skills))
|
|
227
228
|
continue;
|
|
228
229
|
const ts = data.timestamp;
|
|
229
230
|
for (const sr of data.skills) {
|
|
@@ -260,7 +261,7 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
|
|
|
260
261
|
// ── eval 聚合(历史 list)─────────────────────────────────
|
|
261
262
|
const evalBy = {};
|
|
262
263
|
for (const r of reports) {
|
|
263
|
-
if (r.
|
|
264
|
+
if (r.kind !== 'evaluation')
|
|
264
265
|
continue;
|
|
265
266
|
const variants = r.meta.variants || [];
|
|
266
267
|
for (const v of variants) {
|
|
@@ -347,7 +348,7 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
|
|
|
347
348
|
// list 页对每个 entry 跑 detectInsights 的 CPU 开销迁移到这里,只 miss 时算一次。
|
|
348
349
|
const insightsBySkill = new Map();
|
|
349
350
|
for (const ent of entries) {
|
|
350
|
-
const evalReport = ent.eval ? reports.find((r) => r.id === ent.eval.reportId && r.
|
|
351
|
+
const evalReport = ent.eval ? reports.find((r) => r.id === ent.eval.reportId && r.kind === 'evaluation') : undefined;
|
|
351
352
|
insightsBySkill.set(ent.skillName, detectInsights(ent, evalReport ?? null, {
|
|
352
353
|
diagnostics: diagnosisBundle.bySkill[ent.skillName] ?? [],
|
|
353
354
|
}));
|
package/dist/types/doctor.d.ts
CHANGED
|
@@ -10,7 +10,7 @@ export type DoctorSkillStatus = 'pass' | 'warn' | 'fail';
|
|
|
10
10
|
export type DoctorOutcome = 'passed' | 'warnings_only' | 'failed';
|
|
11
11
|
/** Bumped whenever DoctorReport schema changes in a way CI consumers should
|
|
12
12
|
* be able to detect. CI can pin/check this when parsing the JSON. */
|
|
13
|
-
export declare const DOCTOR_REPORT_SCHEMA_VERSION = "
|
|
13
|
+
export declare const DOCTOR_REPORT_SCHEMA_VERSION = "3.0.0";
|
|
14
14
|
export interface DoctorRuleResult {
|
|
15
15
|
ruleId: string;
|
|
16
16
|
severity: DoctorSeverity;
|
|
@@ -94,7 +94,7 @@ export interface DoctorSkillReport {
|
|
|
94
94
|
status: DoctorSkillStatus;
|
|
95
95
|
}
|
|
96
96
|
export interface DoctorReport {
|
|
97
|
-
|
|
97
|
+
kind: 'doctor';
|
|
98
98
|
/** Schema version the JSON consumer can pin/check. Bumped on any
|
|
99
99
|
* user-visible change to this report's shape. See DOCTOR_REPORT_SCHEMA_VERSION. */
|
|
100
100
|
schemaVersion: string;
|
package/dist/types/doctor.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
/** Bumped whenever DoctorReport schema changes in a way CI consumers should
|
|
2
2
|
* be able to detect. CI can pin/check this when parsing the JSON. */
|
|
3
|
-
export const DOCTOR_REPORT_SCHEMA_VERSION = '
|
|
3
|
+
export const DOCTOR_REPORT_SCHEMA_VERSION = '3.0.0';
|
|
4
4
|
export function isComposerRule(r) {
|
|
5
5
|
return r.ruleKind === 'composer';
|
|
6
6
|
}
|
package/dist/types/eval.d.ts
CHANGED
|
@@ -136,10 +136,12 @@ export interface Artifact {
|
|
|
136
136
|
kind: ArtifactKind;
|
|
137
137
|
source: 'baseline' | 'variant-name' | 'file-path' | 'git' | 'inline' | 'custom';
|
|
138
138
|
content: string | null;
|
|
139
|
+
contentHash?: string;
|
|
139
140
|
locator?: string;
|
|
140
141
|
ref?: string;
|
|
141
142
|
cwd?: string;
|
|
142
143
|
skillRoot?: string;
|
|
144
|
+
execRoot?: string;
|
|
143
145
|
experimentRole?: ExperimentRole;
|
|
144
146
|
allowedSkills?: string[];
|
|
145
147
|
metadata?: Record<string, unknown>;
|
|
@@ -160,17 +162,26 @@ export interface VariantConfig {
|
|
|
160
162
|
ref?: string;
|
|
161
163
|
allowedSkills?: string[];
|
|
162
164
|
}
|
|
165
|
+
/** 远端 git 源的结构化引用 —— url/ref/spec 分字段,永不拼成单串再 split(避开 parseGitInput 的 `:`
|
|
166
|
+
* 与 parseVariantCwd 的 `@`)。eval 经 eval.yaml 结构化携带,install 经 --git-url/--git-ref。 */
|
|
167
|
+
export interface RemoteGitRef {
|
|
168
|
+
url: string;
|
|
169
|
+
ref?: string;
|
|
170
|
+
spec: string;
|
|
171
|
+
}
|
|
163
172
|
export interface VariantSpec {
|
|
164
173
|
name: string;
|
|
165
174
|
role: ExperimentRole;
|
|
166
175
|
expr: string;
|
|
176
|
+
git?: RemoteGitRef;
|
|
167
177
|
cwd?: string;
|
|
168
178
|
allowedSkills?: string[];
|
|
169
179
|
}
|
|
170
180
|
export interface EvalConfigVariant {
|
|
171
181
|
name: string;
|
|
172
182
|
role: ExperimentRole;
|
|
173
|
-
artifact
|
|
183
|
+
artifact?: string;
|
|
184
|
+
git?: RemoteGitRef;
|
|
174
185
|
cwd?: string;
|
|
175
186
|
allowedSkills?: string[];
|
|
176
187
|
}
|
package/dist/types/managed.d.ts
CHANGED
|
@@ -21,13 +21,35 @@ export interface ManagedDistributionTarget {
|
|
|
21
21
|
contentHash: string;
|
|
22
22
|
copiedAt: string;
|
|
23
23
|
}
|
|
24
|
-
/** 指向一份 Report
|
|
24
|
+
/** 指向一份 Report 的引用 + 该 report 的最小可比性快照——不是 verdict 本体。install 时为空,
|
|
25
|
+
* eval 完成后追加(见 `src/managed/evidence.ts`)。
|
|
26
|
+
*
|
|
27
|
+
* `reportId` / `contentHash` 是承重的两根(读时门控只认这俩,validator 也只硬查这俩);其余三项是
|
|
28
|
+
* `evidence-gated-management.md` §5 的 mandatory bundle —— **denormalize** 进记录(而非读时回 report
|
|
29
|
+
* 解析),让受管记录自解释、可 grep、不依赖 report 文件仍在盘。旧记录(eval 写入前)无这三项,按
|
|
30
|
+
* optional 读;deriveManagedState 不依赖它们,故缺失不影响生命周期推导。 */
|
|
25
31
|
export interface ManagedEvidenceRef {
|
|
26
32
|
reportId: string;
|
|
27
33
|
/** 该 report 测的是哪份内容(artifact contentHash)。读时只把与记录当前 contentHash 匹配的
|
|
28
34
|
* evidence 算作当前有效证据——重装到新内容后旧证据保留供回滚,但不让新内容显得已测。 */
|
|
29
35
|
contentHash: string;
|
|
30
36
|
recordedAt: string;
|
|
37
|
+
/** §5 mandatory:report 时计算的 verdict 等级(PROGRESS / CAUTIOUS / REGRESS / NOISE /
|
|
38
|
+
* UNDERPOWERED / SOLO)。存字符串而非 import VerdictLevel —— 保 types 层为叶子、不依赖 eval-core。
|
|
39
|
+
* measurable 不看 verdict(任何评测都算"已测");verdict 是 promote 门控的事。 */
|
|
40
|
+
verdict?: string;
|
|
41
|
+
/** §5 mandatory:样本集覆盖。`count`=被测样本数,`hash`=report 的 sampleHashes 排序后摘要
|
|
42
|
+
* (同一样本集 ⇒ 同 hash),供 promote / list 不加载重 report 即可判覆盖与"同一用例集"。 */
|
|
43
|
+
sampleCoverage?: {
|
|
44
|
+
count: number;
|
|
45
|
+
hash: string;
|
|
46
|
+
};
|
|
47
|
+
/** §5 mandatory:可比性 marker。跨 report 比 verdict / Δ 前必须三者一致,否则不可比。 */
|
|
48
|
+
comparability?: {
|
|
49
|
+
cliVersion: string;
|
|
50
|
+
judgePromptHash?: string;
|
|
51
|
+
debiasMode?: Array<'length' | 'position'>;
|
|
52
|
+
};
|
|
31
53
|
}
|
|
32
54
|
export type ManagedDecisionKind = 'promote' | 'reject' | 'rollback';
|
|
33
55
|
/** 一次人工管理决定。install 时为空,promote/reject/rollback 追加。 */
|
|
@@ -38,14 +60,17 @@ export interface ManagedDecision {
|
|
|
38
60
|
reason?: string;
|
|
39
61
|
}
|
|
40
62
|
export interface ManagedArtifactSource {
|
|
41
|
-
/** 源类型(限定判别字,非裸 kind)。`file`=本地路径;`git
|
|
63
|
+
/** 源类型(限定判别字,非裸 kind)。`file`=本地路径;`git`=当前仓库某 ref 或远端(带 url)。 */
|
|
42
64
|
sourceKind: 'file' | 'git';
|
|
43
65
|
/** 源身份,按 sourceKind 分义:
|
|
44
66
|
* - file:本地重哈根(目录-skill 为根目录、文件-skill 为 .md),`hashArtifactSource(locator, isDirectorySkill)` 直接 round-trip;
|
|
45
|
-
* - git:`git:<ref>:<name>`(
|
|
67
|
+
* - 本地 git:`git:<ref>:<name>`;远端 git:`git+<url>@<sha>:<name>`(均非临时物化路径,远端串仅作身份、不回喂 parseGitInput)。
|
|
68
|
+
* drift 由 resolver 重物化重哈 —— mutable ref 给真实漂移、SHA 给不可变。 */
|
|
46
69
|
locator: string;
|
|
47
|
-
/** git 来源的 ref(file 源无)。 */
|
|
70
|
+
/** git 来源的 ref(file 源无);远端为 fetch 后 pin 的实际 SHA。 */
|
|
48
71
|
ref?: string;
|
|
72
|
+
/** 远端 git 的 URL(本地源 / 本地 git 无)。结构化存,供 drift 重取,不必从 locator 反 parse。 */
|
|
73
|
+
url?: string;
|
|
49
74
|
/** 目录-skill(SKILL.md + assets)还是裸 .md 文件-skill。 */
|
|
50
75
|
isDirectorySkill: boolean;
|
|
51
76
|
}
|
|
@@ -43,8 +43,8 @@ export interface ObservationReviewStateEntry {
|
|
|
43
43
|
snippet?: string;
|
|
44
44
|
}
|
|
45
45
|
export interface ObservationReviewState {
|
|
46
|
-
|
|
47
|
-
schemaVersion:
|
|
46
|
+
kind: 'observe-review-state';
|
|
47
|
+
schemaVersion: 2;
|
|
48
48
|
updatedAt: string;
|
|
49
49
|
entries: Record<string, ObservationReviewStateEntry>;
|
|
50
50
|
}
|
|
@@ -184,8 +184,8 @@ export interface ObservationSessionTimeRange {
|
|
|
184
184
|
durationMs?: number;
|
|
185
185
|
}
|
|
186
186
|
export interface ObservationInboxReport {
|
|
187
|
-
|
|
188
|
-
schemaVersion:
|
|
187
|
+
kind: 'observe-inbox';
|
|
188
|
+
schemaVersion: 2;
|
|
189
189
|
meta: {
|
|
190
190
|
tracePath: string;
|
|
191
191
|
generatedAt: string;
|
|
@@ -735,8 +735,8 @@ export interface ExperienceSkillSummary {
|
|
|
735
735
|
relatedObservationIds: string[];
|
|
736
736
|
}
|
|
737
737
|
export interface ObservationExperienceReport {
|
|
738
|
-
|
|
739
|
-
schemaVersion:
|
|
738
|
+
kind: 'observe-experience';
|
|
739
|
+
schemaVersion: 2;
|
|
740
740
|
scope: 'evidence-only';
|
|
741
741
|
generatedAt: string;
|
|
742
742
|
meta: {
|
package/dist/types/report.d.ts
CHANGED
|
@@ -237,10 +237,26 @@ export interface ReportMeta {
|
|
|
237
237
|
timestamp: string;
|
|
238
238
|
cliVersion: string;
|
|
239
239
|
nodeVersion: string;
|
|
240
|
+
/** Per-artifact content fingerprint (variantName → SHA256-12, or 'no-skill' for baseline).
|
|
241
|
+
* schemaVersion >= 3: hashes exactly the executor-visible input, and EVERY dir-skill (local OR
|
|
242
|
+
* git) is materialized into an isolated content-addressed copy whose whole distributable tree
|
|
243
|
+
* (SKILL.md + references/ assets, excluding .omk/.git/node_modules/evolve) is exposed to the
|
|
244
|
+
* executor via cwd — so the fingerprint covers the whole tree and lives in the same space as the
|
|
245
|
+
* install managed-record contentHash (evidence binds for all dir-skills). File-skill (local or
|
|
246
|
+
* git) → the single .md bytes (also binds). schemaVersion 2 was a transitional era where local
|
|
247
|
+
* dir-skills were tree-hashed but git dir-skills hashed SKILL.md bytes only (did not bind);
|
|
248
|
+
* v2 git-dir-skill hashes are NOT comparable to v3. schemaVersion < 2 / absent: legacy
|
|
249
|
+
* SKILL.md-body-text hash — NOT comparable to either. */
|
|
240
250
|
artifactHashes: Record<string, string>;
|
|
241
|
-
/**
|
|
242
|
-
*
|
|
243
|
-
*
|
|
251
|
+
/** Report JSON schema version. Reports without this field are treated as v0 (legacy field
|
|
252
|
+
* semantics: pre-v0.21 `gapRate`/`weightedGapRate` map to `evalGapRate`/`evalWeightedGapRate`
|
|
253
|
+
* for eval-side reports). Value 1 was specified for v0.21+ but never emitted in practice.
|
|
254
|
+
* Value 2 marked the artifactHashes tree-hash era for local dir-skills (git dir-skills still
|
|
255
|
+
* SKILL.md-only). Value 3 extends whole-tree hashing to git dir-skills via isolated copies
|
|
256
|
+
* (all dir-skills bind). Value 4 keeps the v3 hash/binding semantics but marks the canonical
|
|
257
|
+
* top-level discriminant era, so external consumers can version-gate the JSON shape. Drift /
|
|
258
|
+
* lineage consumers gate on `>= 2` (tree-hash era); git-dir-skill binding additionally
|
|
259
|
+
* requires `>= 3`. */
|
|
244
260
|
schemaVersion?: number;
|
|
245
261
|
/** SHA256-12 of every sample's content (sample_id → hash). Same hash = same sample. */
|
|
246
262
|
sampleHashes?: Record<string, string>;
|
|
@@ -341,7 +357,7 @@ export interface SampleSnapshot {
|
|
|
341
357
|
tripwire?: boolean;
|
|
342
358
|
}
|
|
343
359
|
export interface EvaluationReport {
|
|
344
|
-
|
|
360
|
+
kind: 'evaluation';
|
|
345
361
|
id: string;
|
|
346
362
|
meta: ReportMeta;
|
|
347
363
|
summary: Record<string, VariantSummary>;
|
|
@@ -355,6 +371,9 @@ export interface EvaluationReport {
|
|
|
355
371
|
export type Report = EvaluationReport;
|
|
356
372
|
export interface BatchEvaluationMeta {
|
|
357
373
|
mode: 'skill';
|
|
374
|
+
/** Batch report JSON schema version. Kept in lockstep with child EvaluationReport top-level
|
|
375
|
+
* shape so external consumers can identify the canonical discriminant era. */
|
|
376
|
+
schemaVersion: number;
|
|
358
377
|
model: string;
|
|
359
378
|
executor: string;
|
|
360
379
|
skillDir: string;
|
|
@@ -394,7 +413,7 @@ export interface BatchEvaluationItem {
|
|
|
394
413
|
variance?: VarianceData;
|
|
395
414
|
}
|
|
396
415
|
export interface BatchEvaluationReport {
|
|
397
|
-
|
|
416
|
+
kind: 'batch-evaluation';
|
|
398
417
|
id: string;
|
|
399
418
|
mode: 'skill';
|
|
400
419
|
meta: BatchEvaluationMeta;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "oh-my-knowledge",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.36.0",
|
|
4
4
|
"packageManager": "yarn@4.16.0",
|
|
5
5
|
"description": "Evaluation framework for LLM knowledge inputs — prompts, RAG corpora, skills, agent workflows. Fix the model, vary the artifact. Built-in statistical rigor: bootstrap CI, Krippendorff α, length-debias, saturation curves.",
|
|
6
6
|
"type": "module",
|
|
@@ -92,11 +92,11 @@
|
|
|
92
92
|
"license": "MIT",
|
|
93
93
|
"dependencies": {
|
|
94
94
|
"@anthropic-ai/claude-agent-sdk": "^0.3.143",
|
|
95
|
-
"@anthropic-ai/sdk": "^0.
|
|
95
|
+
"@anthropic-ai/sdk": "^0.102.0",
|
|
96
96
|
"@inquirer/prompts": "^8.4.3",
|
|
97
97
|
"@modelcontextprotocol/sdk": "^1.29.0",
|
|
98
98
|
"@oclif/core": "^4",
|
|
99
|
-
"@openai/codex-sdk": "0.
|
|
99
|
+
"@openai/codex-sdk": "0.137.0",
|
|
100
100
|
"ajv": "^8.18.0",
|
|
101
101
|
"chart.js": "^4.5.1",
|
|
102
102
|
"js-yaml": "^4.1.1",
|