oh-my-knowledge 0.35.0 → 0.37.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -3
- package/README.zh.md +5 -3
- package/dist/assets/agent-skills/omk/SKILL.md +4 -2
- package/dist/assets/agent-skills/omk/references/commands.md +121 -2
- package/dist/authoring/evolver.js +1 -1
- package/dist/cli/commands/doctor.js +21 -5
- package/dist/cli/commands/eval/index.js +7 -6
- package/dist/cli/commands/evolve.d.ts +11 -0
- package/dist/cli/commands/evolve.js +105 -5
- package/dist/cli/commands/list.d.ts +25 -0
- package/dist/cli/commands/list.js +118 -0
- package/dist/cli/commands/promote.d.ts +24 -0
- package/dist/cli/commands/promote.js +158 -0
- package/dist/cli/commands/rollback.d.ts +20 -0
- package/dist/cli/commands/rollback.js +95 -0
- package/dist/cli/commands/sample.d.ts +1 -0
- package/dist/cli/commands/sample.js +10 -4
- package/dist/cli/lib/cell-format.d.ts +17 -0
- package/dist/cli/lib/cell-format.js +19 -0
- package/dist/cli/lib/cmd-flags.d.ts +1 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +8 -0
- package/dist/cli/lib/i18n-dict/list.d.ts +3 -0
- package/dist/cli/lib/i18n-dict/list.js +36 -0
- package/dist/cli/lib/i18n-dict/promote.d.ts +3 -0
- package/dist/cli/lib/i18n-dict/promote.js +54 -0
- package/dist/cli/lib/i18n-dict/rollback.d.ts +3 -0
- package/dist/cli/lib/i18n-dict/rollback.js +22 -0
- package/dist/cli/lib/i18n-dict.d.ts +4 -1
- package/dist/cli/lib/i18n-dict.js +6 -0
- package/dist/cli/lib/progress.d.ts +3 -0
- package/dist/cli/lib/progress.js +22 -0
- package/dist/cli/lib/run-tally.js +1 -1
- package/dist/cli/lib/shared.js +1 -1
- package/dist/cli/lib/source-probe.d.ts +15 -0
- package/dist/cli/lib/source-probe.js +129 -0
- package/dist/doctor/endpoint-rule.d.ts +60 -0
- package/dist/doctor/endpoint-rule.js +381 -0
- package/dist/doctor/health/load-custom-dimensions.d.ts +6 -0
- package/dist/doctor/health/load-custom-dimensions.js +43 -3
- package/dist/doctor/index.js +9 -2
- package/dist/eval-core/evaluation-reporting.d.ts +1 -0
- package/dist/eval-core/evaluation-reporting.js +6 -4
- package/dist/eval-workflows/batch-evaluation-workflow.js +3 -2
- package/dist/eval-workflows/run-evaluation.js +2 -2
- package/dist/executors/codex-cli.js +5 -1
- package/dist/inputs/skill-loader.js +6 -3
- package/dist/managed/index.d.ts +2 -0
- package/dist/managed/index.js +2 -0
- package/dist/managed/list-view.d.ts +58 -0
- package/dist/managed/list-view.js +70 -0
- package/dist/managed/promote-gate.d.ts +40 -0
- package/dist/managed/promote-gate.js +37 -0
- package/dist/managed/store.d.ts +25 -2
- package/dist/managed/store.js +137 -11
- package/dist/observability/experience.d.ts +2 -0
- package/dist/observability/experience.js +31 -2
- package/dist/observability/inbox.js +31 -5
- package/dist/observability/review-state.js +22 -11
- package/dist/observability/soft-standards/llm-extractor.js +3 -3
- package/dist/observability/soft-standards/skill-standards-store.d.ts +1 -0
- package/dist/observability/soft-standards/skill-standards-store.js +77 -14
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/renderer/doctor-detail-renderer.d.ts +9 -0
- package/dist/renderer/doctor-detail-renderer.js +114 -0
- package/dist/renderer/html-renderer.d.ts +3 -2
- package/dist/renderer/html-renderer.js +102 -102
- package/dist/renderer/icons.d.ts +29 -0
- package/dist/renderer/icons.js +66 -0
- package/dist/renderer/layout.js +81 -68
- package/dist/renderer/observation-inbox/styles.d.ts +1 -1
- package/dist/renderer/observation-inbox/styles.js +53 -53
- package/dist/renderer/report-shell.d.ts +77 -0
- package/dist/renderer/report-shell.js +223 -0
- package/dist/renderer/skill-detail-renderer.js +147 -352
- package/dist/renderer/skill-health-renderer.js +50 -73
- package/dist/renderer/skill-list-renderer.js +358 -306
- package/dist/renderer/summary.js +272 -162
- package/dist/renderer/test-view.d.ts +4 -3
- package/dist/renderer/test-view.js +386 -135
- package/dist/server/report-server.js +142 -28
- package/dist/server/report-store.d.ts +1 -1
- package/dist/server/report-store.js +16 -15
- package/dist/server/skill-index.js +5 -4
- package/dist/types/doctor.d.ts +24 -2
- package/dist/types/doctor.js +1 -1
- package/dist/types/managed.d.ts +17 -3
- package/dist/types/observability.d.ts +6 -6
- package/dist/types/report.d.ts +9 -4
- package/package.json +1 -1
|
@@ -5,8 +5,9 @@ import { join } from 'node:path';
|
|
|
5
5
|
import { homedir } from 'node:os';
|
|
6
6
|
import { renderReportDocumentDetail, renderTrendsPage, renderRunList } from '../renderer/html-renderer.js';
|
|
7
7
|
import { renderSkillList } from '../renderer/skill-list-renderer.js';
|
|
8
|
-
import { renderSkillDetail } from '../renderer/skill-detail-renderer.js';
|
|
9
8
|
import { renderSkillHealthReport } from '../renderer/skill-health-renderer.js';
|
|
9
|
+
import { renderDoctorDetail } from '../renderer/doctor-detail-renderer.js';
|
|
10
|
+
import { assessHealth } from '../renderer/skill-detail-renderer.js';
|
|
10
11
|
import { renderObservationInboxPage } from '../renderer/observation-inbox-renderer.js';
|
|
11
12
|
import { DEFAULT_LANG, t, layout } from '../renderer/layout.js';
|
|
12
13
|
import { buildSkillIndex } from './skill-index.js';
|
|
@@ -80,6 +81,28 @@ function loadAnalysis(dir, id) {
|
|
|
80
81
|
return null;
|
|
81
82
|
}
|
|
82
83
|
}
|
|
84
|
+
/** 扫 doctorsDir 找 id 匹配的 doctor 报告(文件名不一定等于 report id)。
|
|
85
|
+
* 批量 doctor 会按 skill 拆成多份共享同一 id 的 per-skill 文件,传 skillName 时
|
|
86
|
+
* 优先返回含该 skill 的那份;都不含时回退首个 id 命中(单 skill / 无参行为不变)。 */
|
|
87
|
+
function loadDoctorReport(dir, id, skillName) {
|
|
88
|
+
if (!existsSync(dir))
|
|
89
|
+
return null;
|
|
90
|
+
let fallback = null;
|
|
91
|
+
for (const file of readdirSync(dir)) {
|
|
92
|
+
if (!file.endsWith('.json'))
|
|
93
|
+
continue;
|
|
94
|
+
try {
|
|
95
|
+
const data = JSON.parse(readFileSync(join(dir, file), 'utf-8'));
|
|
96
|
+
if (data?.kind !== 'doctor' || data.id !== id)
|
|
97
|
+
continue;
|
|
98
|
+
if (!skillName || data.skills?.some((s) => s.skillName === skillName))
|
|
99
|
+
return data;
|
|
100
|
+
fallback ??= data;
|
|
101
|
+
}
|
|
102
|
+
catch { /* skip */ }
|
|
103
|
+
}
|
|
104
|
+
return fallback;
|
|
105
|
+
}
|
|
83
106
|
/**
|
|
84
107
|
* 扫 analyses/ 所有 JSON,按 skillName 过滤,按时间排序成 trend points。
|
|
85
108
|
*/
|
|
@@ -204,6 +227,85 @@ function readJsonBody(req, maxBytes = 1024 * 1024) {
|
|
|
204
227
|
req.on('error', reject);
|
|
205
228
|
});
|
|
206
229
|
}
|
|
230
|
+
function fmtHistDate(ts, lang) {
|
|
231
|
+
if (!ts)
|
|
232
|
+
return '-';
|
|
233
|
+
try {
|
|
234
|
+
const d = new Date(ts);
|
|
235
|
+
return d.toLocaleString(lang === 'zh' ? 'zh-CN' : 'en-US', { month: '2-digit', day: '2-digit', hour: '2-digit', minute: '2-digit' });
|
|
236
|
+
}
|
|
237
|
+
catch {
|
|
238
|
+
return ts;
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
// 报告详情页顶部「skill 上下文」:doctor/eval/observe 维度切换器 + 「全部历史」弹框(该 skill 历次 run)。
|
|
242
|
+
// observe 详情是 fleet 级日报,无 per-skill 页,observe chip 只当状态点不可点。
|
|
243
|
+
function buildSkillContext(entry, currentDim, currentReportId, insights, lang) {
|
|
244
|
+
const zh = lang === 'zh';
|
|
245
|
+
const langQ = lang === DEFAULT_LANG ? '' : `?lang=${lang}`;
|
|
246
|
+
const amp = lang === DEFAULT_LANG ? '' : `&lang=${lang}`;
|
|
247
|
+
// 综合健康分(= 跨维度聚合,与首页列表同一口径)— 报告页只显单维度分数,这里把「总分」也带上。
|
|
248
|
+
const health = assessHealth(entry, insights, lang);
|
|
249
|
+
const d = entry.doctor;
|
|
250
|
+
const dTotal = d ? d.passCount + d.warnCount + d.failCount : 0;
|
|
251
|
+
const dScore = d && dTotal > 0 ? Math.round(((d.passCount + d.warnCount * 0.5) / dTotal) * 100) : null;
|
|
252
|
+
const dBand = d ? (d.failCount > 0 ? 'red' : d.warnCount > 0 ? 'yellow' : 'green') : 'gray';
|
|
253
|
+
const ev = entry.eval;
|
|
254
|
+
const evScore = ev && ev.compositeScore != null ? Math.round((ev.compositeScore / 5) * 100) : null;
|
|
255
|
+
const evBand = ev && ev.compositeScore != null
|
|
256
|
+
? (ev.compositeScore >= 4 ? 'green' : ev.compositeScore >= 3 ? 'yellow' : 'red') : 'gray';
|
|
257
|
+
const ob = entry.observe;
|
|
258
|
+
const obScore = ob ? Math.round((1 - ob.gapRate) * 100) : null;
|
|
259
|
+
const obBand = ob ? ob.healthBand : 'gray';
|
|
260
|
+
// 历史:eval + doctor 历次 run,各自新→旧。点行跳到那次报告。
|
|
261
|
+
// evalHistory 是「每轮一条」,但一份 evolve 报告(多轮)是同一个 reportId —— 按 reportId 去重,
|
|
262
|
+
// 一份报告只算一条(取末轮 treatment 分数,与报告头部一致),否则同一报告的多轮会都标「当前」。
|
|
263
|
+
const evalRoundCount = new Map();
|
|
264
|
+
const evalByReport = new Map();
|
|
265
|
+
for (const h of entry.evalHistory) {
|
|
266
|
+
evalRoundCount.set(h.reportId, (evalRoundCount.get(h.reportId) ?? 0) + 1);
|
|
267
|
+
evalByReport.set(h.reportId, h); // 末轮(最高 round)胜出 —— history 已按 timestamp#round 升序
|
|
268
|
+
}
|
|
269
|
+
const evalHist = [...evalByReport.values()].reverse().map((h) => {
|
|
270
|
+
const rounds = evalRoundCount.get(h.reportId) ?? 1;
|
|
271
|
+
return {
|
|
272
|
+
dim: 'eval',
|
|
273
|
+
dateText: fmtHistDate(h.timestamp, lang),
|
|
274
|
+
scoreText: h.compositeScore != null ? `${h.compositeScore.toFixed(2)} / 5` : '—',
|
|
275
|
+
band: (h.compositeScore == null ? 'gray' : h.compositeScore >= 4 ? 'green' : h.compositeScore >= 3 ? 'yellow' : 'red'),
|
|
276
|
+
metaText: `${h.passCount}/${h.totalSamples} ${zh ? '通过' : 'pass'}${h.failCount > 0 ? ` · ${h.failCount} ${zh ? '失败' : 'fail'}` : ''}${rounds > 1 ? ` · ${rounds} ${zh ? '轮' : 'rounds'}` : ''}`,
|
|
277
|
+
href: `/reports/${encodeURIComponent(h.reportId)}${langQ}`,
|
|
278
|
+
current: h.reportId === currentReportId,
|
|
279
|
+
};
|
|
280
|
+
});
|
|
281
|
+
const doctorHist = [...entry.doctorHistory].reverse().map((h) => {
|
|
282
|
+
const tot = h.passCount + h.warnCount + h.failCount;
|
|
283
|
+
return {
|
|
284
|
+
dim: 'doctor',
|
|
285
|
+
dateText: fmtHistDate(h.timestamp, lang),
|
|
286
|
+
scoreText: tot > 0 ? `${Math.round(((h.passCount + h.warnCount * 0.5) / tot) * 100)}` : '—',
|
|
287
|
+
band: (h.failCount > 0 ? 'red' : h.warnCount > 0 ? 'yellow' : 'green'),
|
|
288
|
+
metaText: `${h.passCount}✓ ${h.warnCount}⚠ ${h.failCount}✗`,
|
|
289
|
+
href: `/doctors/${encodeURIComponent(h.reportId)}?skill=${encodeURIComponent(entry.skillName)}${amp}`,
|
|
290
|
+
current: h.reportId === currentReportId,
|
|
291
|
+
};
|
|
292
|
+
});
|
|
293
|
+
return {
|
|
294
|
+
skillName: entry.skillName,
|
|
295
|
+
overall: { score: health.score, band: health.color },
|
|
296
|
+
// 当前维度 chip 不显分数(hero ring 已显示「本报告」分数,避免与「最新」分数冲突);非当前维度显「最新」分数 + 链接。
|
|
297
|
+
chips: [
|
|
298
|
+
{ dim: 'doctor', label: zh ? '体检' : 'Doctor', score: currentDim === 'doctor' ? null : dScore, band: dBand,
|
|
299
|
+
href: currentDim === 'doctor' || !d ? null : `/doctors/${encodeURIComponent(d.reportId)}?skill=${encodeURIComponent(entry.skillName)}${amp}`,
|
|
300
|
+
active: currentDim === 'doctor' },
|
|
301
|
+
{ dim: 'eval', label: zh ? '评测' : 'Eval', score: currentDim === 'eval' ? null : evScore, band: evBand,
|
|
302
|
+
href: currentDim === 'eval' || !ev ? null : `/reports/${encodeURIComponent(ev.reportId)}${langQ}`,
|
|
303
|
+
active: currentDim === 'eval' },
|
|
304
|
+
{ dim: 'observe', label: zh ? '观察' : 'Observe', score: obScore, band: obBand, href: null, active: false },
|
|
305
|
+
],
|
|
306
|
+
history: [...evalHist, ...doctorHist],
|
|
307
|
+
};
|
|
308
|
+
}
|
|
207
309
|
function renderSkillDiffPage(diff, lang = DEFAULT_LANG) {
|
|
208
310
|
const { fromId, toId, fromAt, toAt, rows } = diff;
|
|
209
311
|
const langQ = lang === DEFAULT_LANG ? '' : `?lang=${lang}`;
|
|
@@ -561,6 +663,29 @@ export function createReportServer({ port, host: hostOption, reportsDir = DEFAUL
|
|
|
561
663
|
res.end(JSON.stringify({ error: 'method not allowed' }));
|
|
562
664
|
return;
|
|
563
665
|
}
|
|
666
|
+
const doctorDetailMatch = path.match(/^\/doctors\/(.+)$/);
|
|
667
|
+
if (doctorDetailMatch) {
|
|
668
|
+
const id = decodeURIComponent(doctorDetailMatch[1]);
|
|
669
|
+
const skillName = parsed.searchParams.get('skill') ?? '';
|
|
670
|
+
const report = loadDoctorReport(doctorsDir, id, skillName || undefined);
|
|
671
|
+
if (!report) {
|
|
672
|
+
res.writeHead(404, { 'Content-Type': 'text/plain; charset=utf-8' });
|
|
673
|
+
res.end(lang === 'en' ? 'doctor report not found' : '体检报告不存在');
|
|
674
|
+
return;
|
|
675
|
+
}
|
|
676
|
+
let ctx;
|
|
677
|
+
if (skillName) {
|
|
678
|
+
const runs = await reportStore.list();
|
|
679
|
+
const idx = buildSkillIndex(runs, analysesDir, doctorsDir, observationsDir);
|
|
680
|
+
const entry = idx.entries.find((en) => en.skillName === skillName);
|
|
681
|
+
if (entry)
|
|
682
|
+
ctx = buildSkillContext(entry, 'doctor', id, idx.insightsBySkill.get(entry.skillName) ?? [], lang);
|
|
683
|
+
}
|
|
684
|
+
res.writeHead(200, { 'Content-Type': 'text/html; charset=utf-8' });
|
|
685
|
+
const langQ = lang === DEFAULT_LANG ? '' : `?lang=${lang}`;
|
|
686
|
+
res.end(renderDoctorDetail(report, skillName, langQ, lang, ctx));
|
|
687
|
+
return;
|
|
688
|
+
}
|
|
564
689
|
const analysisDetailMatch = path.match(/^\/analyses\/(.+)$/);
|
|
565
690
|
if (analysisDetailMatch) {
|
|
566
691
|
const id = decodeURIComponent(analysisDetailMatch[1]);
|
|
@@ -708,9 +833,19 @@ export function createReportServer({ port, host: hostOption, reportsDir = DEFAUL
|
|
|
708
833
|
}
|
|
709
834
|
const reportPageMatch = path.match(/^\/reports\/(.+)$/);
|
|
710
835
|
if (reportPageMatch) {
|
|
711
|
-
const
|
|
836
|
+
const reportId = decodeURIComponent(reportPageMatch[1]);
|
|
837
|
+
const report = await queryRun(reportStore, reportId);
|
|
838
|
+
let ctx;
|
|
839
|
+
if (report && report.kind === 'evaluation') {
|
|
840
|
+
const runs = await reportStore.list();
|
|
841
|
+
const idx = buildSkillIndex(runs, analysesDir, doctorsDir, observationsDir);
|
|
842
|
+
// 按 evalHistory 匹配(非仅最新),历史 eval 报告也能定位到所属 skill。
|
|
843
|
+
const entry = idx.entries.find((en) => en.evalHistory.some((h) => h.reportId === reportId));
|
|
844
|
+
if (entry)
|
|
845
|
+
ctx = buildSkillContext(entry, 'eval', reportId, idx.insightsBySkill.get(entry.skillName) ?? [], lang);
|
|
846
|
+
}
|
|
712
847
|
res.writeHead(report ? 200 : 404, { 'Content-Type': 'text/html; charset=utf-8' });
|
|
713
|
-
res.end(renderReportDocumentDetail(report, lang));
|
|
848
|
+
res.end(renderReportDocumentDetail(report, lang, ctx));
|
|
714
849
|
return;
|
|
715
850
|
}
|
|
716
851
|
// 老 run 列表 — 兼容老书签。/skills/ 切换为默认后,run 列表挪这里。
|
|
@@ -745,31 +880,10 @@ export function createReportServer({ port, host: hostOption, reportsDir = DEFAUL
|
|
|
745
880
|
}));
|
|
746
881
|
return;
|
|
747
882
|
}
|
|
748
|
-
//
|
|
749
|
-
//
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
const skillName = decodeURIComponent(skillDetailMatch[1]);
|
|
753
|
-
const runs = await reportStore.list();
|
|
754
|
-
const idx = buildSkillIndex(runs, analysesDir, doctorsDir, observationsDir);
|
|
755
|
-
const entry = idx.entries.find((e) => e.skillName === skillName);
|
|
756
|
-
if (!entry) {
|
|
757
|
-
res.writeHead(404, { 'Content-Type': 'text/plain; charset=utf-8' });
|
|
758
|
-
res.end(`Skill not found: ${skillName}`);
|
|
759
|
-
return;
|
|
760
|
-
}
|
|
761
|
-
// 加载完整 eval 报告(为详情页提供 layered scores / coverage / per-sample diagnostic)
|
|
762
|
-
let evalReport = null;
|
|
763
|
-
if (entry.eval) {
|
|
764
|
-
const r = await reportStore.get(entry.eval.reportId);
|
|
765
|
-
if (r && r.reportKind === 'evaluation')
|
|
766
|
-
evalReport = r;
|
|
767
|
-
}
|
|
768
|
-
res.writeHead(200, { 'Content-Type': 'text/html; charset=utf-8' });
|
|
769
|
-
const insights = idx.insightsBySkill.get(skillName) ?? [];
|
|
770
|
-
res.end(renderSkillDetail(entry, evalReport, lang, insights));
|
|
771
|
-
return;
|
|
772
|
-
}
|
|
883
|
+
// 注:/skills/<name> 详情页(hub)已下线 — 首页点行直接进 eval/doctor 报告,跨维度靠
|
|
884
|
+
// 报告页 chip 条 +「全部历史」弹框,不再有中间汇总页。该路由不做兼容重定向:Studio
|
|
885
|
+
// 是本地临时服务器、端口每次启动都变,跨会话书签 localhost:<port> 本就失效,且 PR 内
|
|
886
|
+
// 已无任何链接指向它。未匹配的 /skills/<name> 直接落到下方通用 404。
|
|
773
887
|
const skillDiagnosticsApiMatch = path.match(/^\/api\/skills\/(.+)\/diagnostics$/);
|
|
774
888
|
if (skillDiagnosticsApiMatch) {
|
|
775
889
|
const skillName = decodeURIComponent(skillDiagnosticsApiMatch[1]);
|
|
@@ -20,7 +20,7 @@ export declare function queryJobList(jobStore: JobStore, query?: JobQuery): Prom
|
|
|
20
20
|
export declare function queryJob(jobStore: JobStore, id: string): Promise<EvaluationJob | null>;
|
|
21
21
|
export interface RunListItem {
|
|
22
22
|
id: string;
|
|
23
|
-
|
|
23
|
+
kind: ReportDocument['kind'];
|
|
24
24
|
meta: ReportDocument['meta'];
|
|
25
25
|
summary?: EvaluationReport['summary'];
|
|
26
26
|
items?: BatchEvaluationReport['items'];
|
|
@@ -39,32 +39,33 @@ export function createFileStore(dir) {
|
|
|
39
39
|
if (!data || typeof data !== 'object')
|
|
40
40
|
return null;
|
|
41
41
|
const record = data;
|
|
42
|
-
|
|
42
|
+
const kind = record.kind === 'evaluation' || record.kind === 'batch-evaluation'
|
|
43
|
+
? record.kind
|
|
44
|
+
: undefined;
|
|
45
|
+
if (kind === 'evaluation') {
|
|
43
46
|
if (!record.meta || !record.summary || !Array.isArray(record.results))
|
|
44
47
|
return null;
|
|
45
|
-
return {
|
|
48
|
+
return {
|
|
49
|
+
...record,
|
|
50
|
+
kind,
|
|
51
|
+
id: typeof record.id === 'string' && record.id ? record.id : fallbackId,
|
|
52
|
+
};
|
|
46
53
|
}
|
|
47
|
-
if (
|
|
54
|
+
if (kind === 'batch-evaluation') {
|
|
48
55
|
if (!record.meta || !Array.isArray(record.items))
|
|
49
56
|
return null;
|
|
50
|
-
return { ...record, id: typeof record.id === 'string' && record.id ? record.id : fallbackId };
|
|
51
|
-
}
|
|
52
|
-
if (record.kind === undefined
|
|
53
|
-
&& record.overview === undefined
|
|
54
|
-
&& record.artifacts === undefined
|
|
55
|
-
&& record.meta
|
|
56
|
-
&& record.summary
|
|
57
|
-
&& Array.isArray(record.results)) {
|
|
58
57
|
return {
|
|
59
58
|
...record,
|
|
60
|
-
|
|
59
|
+
kind,
|
|
61
60
|
id: typeof record.id === 'string' && record.id ? record.id : fallbackId,
|
|
62
61
|
};
|
|
63
62
|
}
|
|
63
|
+
// 只认 canonical 顶层 `kind`(evaluation / batch-evaluation)。不再为旧格式(顶层无该判别字段的
|
|
64
|
+
// 历史文件)做读兼容 —— 顶层 kind cutover 是硬切换,旧文件直接判脏丢弃。
|
|
64
65
|
return null;
|
|
65
66
|
}
|
|
66
67
|
function isEvaluationReport(report) {
|
|
67
|
-
return report.
|
|
68
|
+
return report.kind === 'evaluation';
|
|
68
69
|
}
|
|
69
70
|
// Studio 每个 / 和 /skills/<name> 请求都调 list(),里面对每个 .json 同步 readFile +
|
|
70
71
|
// JSON.parse。报告数上来后这是主性能瓶颈。缓存策略:fingerprint = dir mtime + 文件名
|
|
@@ -217,9 +218,9 @@ export async function queryJob(jobStore, id) {
|
|
|
217
218
|
export async function queryRunList(reportStore) {
|
|
218
219
|
return (await reportStore.list()).map((report) => ({
|
|
219
220
|
id: report.id,
|
|
220
|
-
|
|
221
|
+
kind: report.kind,
|
|
221
222
|
meta: report.meta,
|
|
222
|
-
...(report.
|
|
223
|
+
...(report.kind === 'evaluation' ? { summary: report.summary } : { items: report.items }),
|
|
223
224
|
}));
|
|
224
225
|
}
|
|
225
226
|
export async function queryRun(reportStore, id) {
|
|
@@ -69,7 +69,7 @@ function buildIndexFingerprint(reports, analysesDir, doctorsDir, observationsDir
|
|
|
69
69
|
// 每一段的 right-hand-side 从旧的 "{dir-mtime}-{file-count}" 双标量升级成
|
|
70
70
|
// safeDirJsonContentFingerprint 返回的 "{dir-mtime}|{file1}:{m}:{s},..."
|
|
71
71
|
// content-aware 字符串。
|
|
72
|
-
const reportIds = reports.map((r) => `${r.id}:${r.meta?.timestamp ?? ''}:${r.
|
|
72
|
+
const reportIds = reports.map((r) => `${r.id}:${r.meta?.timestamp ?? ''}:${r.kind === 'evaluation' ? r.meta.evolve?.skillName ?? '' : ''}`).join(',');
|
|
73
73
|
const doctorsFp = safeDirJsonContentFingerprint(doctorsDir);
|
|
74
74
|
const analysesFp = safeDirJsonContentFingerprint(analysesDir);
|
|
75
75
|
const observationsFp = safeDirJsonContentFingerprint(observationsDir);
|
|
@@ -223,7 +223,8 @@ function scanDoctorReports(dir) {
|
|
|
223
223
|
continue;
|
|
224
224
|
try {
|
|
225
225
|
const data = JSON.parse(readFileSync(join(dir, file), 'utf-8'));
|
|
226
|
-
|
|
226
|
+
const kind = data?.kind === 'doctor' ? data.kind : null;
|
|
227
|
+
if (!kind || !Array.isArray(data.skills))
|
|
227
228
|
continue;
|
|
228
229
|
const ts = data.timestamp;
|
|
229
230
|
for (const sr of data.skills) {
|
|
@@ -260,7 +261,7 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
|
|
|
260
261
|
// ── eval 聚合(历史 list)─────────────────────────────────
|
|
261
262
|
const evalBy = {};
|
|
262
263
|
for (const r of reports) {
|
|
263
|
-
if (r.
|
|
264
|
+
if (r.kind !== 'evaluation')
|
|
264
265
|
continue;
|
|
265
266
|
const variants = r.meta.variants || [];
|
|
266
267
|
for (const v of variants) {
|
|
@@ -347,7 +348,7 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
|
|
|
347
348
|
// list 页对每个 entry 跑 detectInsights 的 CPU 开销迁移到这里,只 miss 时算一次。
|
|
348
349
|
const insightsBySkill = new Map();
|
|
349
350
|
for (const ent of entries) {
|
|
350
|
-
const evalReport = ent.eval ? reports.find((r) => r.id === ent.eval.reportId && r.
|
|
351
|
+
const evalReport = ent.eval ? reports.find((r) => r.id === ent.eval.reportId && r.kind === 'evaluation') : undefined;
|
|
351
352
|
insightsBySkill.set(ent.skillName, detectInsights(ent, evalReport ?? null, {
|
|
352
353
|
diagnostics: diagnosisBundle.bySkill[ent.skillName] ?? [],
|
|
353
354
|
}));
|
package/dist/types/doctor.d.ts
CHANGED
|
@@ -10,7 +10,7 @@ export type DoctorSkillStatus = 'pass' | 'warn' | 'fail';
|
|
|
10
10
|
export type DoctorOutcome = 'passed' | 'warnings_only' | 'failed';
|
|
11
11
|
/** Bumped whenever DoctorReport schema changes in a way CI consumers should
|
|
12
12
|
* be able to detect. CI can pin/check this when parsing the JSON. */
|
|
13
|
-
export declare const DOCTOR_REPORT_SCHEMA_VERSION = "
|
|
13
|
+
export declare const DOCTOR_REPORT_SCHEMA_VERSION = "3.0.0";
|
|
14
14
|
export interface DoctorRuleResult {
|
|
15
15
|
ruleId: string;
|
|
16
16
|
severity: DoctorSeverity;
|
|
@@ -85,6 +85,10 @@ export interface DoctorRule {
|
|
|
85
85
|
severity: DoctorSeverity;
|
|
86
86
|
/** i18n key,terminal 渲染时用作 rule 标题。 */
|
|
87
87
|
labelKey: string;
|
|
88
|
+
/** true = 需要外部 I/O(网络 / LLM)的"在线"检查,跟 skill_health composer 同档:
|
|
89
|
+
* 默认 `omk doctor` 会跑,`--static-only` 离线模式跳过。endpoint 自定义维度置 true。
|
|
90
|
+
* 缺省(undefined/false)= 纯静态低成本检查(内置 4 条),静态模式才跑。 */
|
|
91
|
+
external?: boolean;
|
|
88
92
|
check(ctx: DoctorContext): Promise<DoctorRuleCheckOutcome>;
|
|
89
93
|
}
|
|
90
94
|
export interface DoctorSkillReport {
|
|
@@ -94,7 +98,7 @@ export interface DoctorSkillReport {
|
|
|
94
98
|
status: DoctorSkillStatus;
|
|
95
99
|
}
|
|
96
100
|
export interface DoctorReport {
|
|
97
|
-
|
|
101
|
+
kind: 'doctor';
|
|
98
102
|
/** Schema version the JSON consumer can pin/check. Bumped on any
|
|
99
103
|
* user-visible change to this report's shape. See DOCTOR_REPORT_SCHEMA_VERSION. */
|
|
100
104
|
schemaVersion: string;
|
|
@@ -126,6 +130,21 @@ export interface DoctorReport {
|
|
|
126
130
|
total: number;
|
|
127
131
|
};
|
|
128
132
|
}
|
|
133
|
+
/** 批量体检的 per-skill 进度事件。runDoctor 在遍历 artifacts 时,每个 skill
|
|
134
|
+
* 开始(skill_start)和结束(skill_done)各发一次,对齐 eval 的 per-sample 进度。 */
|
|
135
|
+
export interface DoctorProgressInfo {
|
|
136
|
+
phase: 'skill_start' | 'skill_done';
|
|
137
|
+
/** 第几个 skill(1-based)。 */
|
|
138
|
+
index: number;
|
|
139
|
+
/** skill 总数。 */
|
|
140
|
+
total: number;
|
|
141
|
+
skillName: string;
|
|
142
|
+
/** 仅 skill_done:该 skill 的最终状态。 */
|
|
143
|
+
status?: DoctorSkillStatus;
|
|
144
|
+
/** 仅 skill_done:该 skill 全部 rule 的执行耗时(ms)。 */
|
|
145
|
+
durationMs?: number;
|
|
146
|
+
}
|
|
147
|
+
export type DoctorProgressCallback = (info: DoctorProgressInfo) => void;
|
|
129
148
|
export interface DoctorRunOptions {
|
|
130
149
|
/** 单 skill 文件 / 目录 / null(=cwd 当前目录批量)。当 artifacts 显式提供时, target 被忽略。 */
|
|
131
150
|
target?: string | null;
|
|
@@ -153,4 +172,7 @@ export interface DoctorRunOptions {
|
|
|
153
172
|
* programmatic API 默认 false。 */
|
|
154
173
|
runHealthCheck?: boolean;
|
|
155
174
|
effort?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
|
|
175
|
+
/** 批量体检进度回调(per-skill)。CLI 非 gate 模式注入、写 stderr;eval 内嵌
|
|
176
|
+
* 调用不传(eval 有自己的进度体系,不应冒出 doctor 进度)。 */
|
|
177
|
+
onProgress?: DoctorProgressCallback;
|
|
156
178
|
}
|
package/dist/types/doctor.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
/** Bumped whenever DoctorReport schema changes in a way CI consumers should
|
|
2
2
|
* be able to detect. CI can pin/check this when parsing the JSON. */
|
|
3
|
-
export const DOCTOR_REPORT_SCHEMA_VERSION = '
|
|
3
|
+
export const DOCTOR_REPORT_SCHEMA_VERSION = '3.0.0';
|
|
4
4
|
export function isComposerRule(r) {
|
|
5
5
|
return r.ruleKind === 'composer';
|
|
6
6
|
}
|
package/dist/types/managed.d.ts
CHANGED
|
@@ -52,12 +52,25 @@ export interface ManagedEvidenceRef {
|
|
|
52
52
|
};
|
|
53
53
|
}
|
|
54
54
|
export type ManagedDecisionKind = 'promote' | 'reject' | 'rollback';
|
|
55
|
-
/** 一次人工管理决定。install 时为空,promote/reject/rollback
|
|
55
|
+
/** 一次人工管理决定。install 时为空,promote/reject/rollback 追加(append-only 事件流)。 */
|
|
56
56
|
export interface ManagedDecision {
|
|
57
57
|
decisionKind: ManagedDecisionKind;
|
|
58
58
|
actor: string;
|
|
59
59
|
decidedAt: string;
|
|
60
60
|
reason?: string;
|
|
61
|
+
/** 被该决定接受 / 回滚到的内容 hash —— 锚定「决定的是哪份内容」。promote 取决定时的 record.contentHash;
|
|
62
|
+
* 读时只把与当前 contentHash 匹配的 promote 决定算作「当前版本已 promoted」(旧内容的决定不冒充当前)。 */
|
|
63
|
+
contentHash?: string;
|
|
64
|
+
/** 该决定锚定的证据 report(promote 取 latestCurrentEvidence 那条的 reportId)——可回溯「凭什么 ship」。 */
|
|
65
|
+
reportId?: string;
|
|
66
|
+
/** 越门记录:门禁本应拦下,经 --force 显式越过时记下供审计(spec §7「Overrides must be explicit and
|
|
67
|
+
* recorded」)。无此字段 = 正常通过门禁。`verdict` 是越门时证据的 verdict(上下文);`overriddenBlocks`
|
|
68
|
+
* 是真正被越过的判据(drifted / incomparable / verdict_blocked),让审计能回答「越过了什么」—— 只看
|
|
69
|
+
* verdict 会误导(仅因 drift 越门时 verdict 可能仍是 PROGRESS)。 */
|
|
70
|
+
override?: {
|
|
71
|
+
verdict: string;
|
|
72
|
+
overriddenBlocks?: string[];
|
|
73
|
+
};
|
|
61
74
|
}
|
|
62
75
|
export interface ManagedArtifactSource {
|
|
63
76
|
/** 源类型(限定判别字,非裸 kind)。`file`=本地路径;`git`=当前仓库某 ref 或远端(带 url)。 */
|
|
@@ -92,8 +105,9 @@ export interface ManagedArtifactRecord {
|
|
|
92
105
|
evidence: ManagedEvidenceRef[];
|
|
93
106
|
decisions: ManagedDecision[];
|
|
94
107
|
}
|
|
95
|
-
/** 读时推导的生命周期标签——不持久(见文件头说明)
|
|
96
|
-
|
|
108
|
+
/** 读时推导的生命周期标签——不持久(见文件头说明)。`promoted` 由当前内容(contentHash 匹配)有一条
|
|
109
|
+
* promote 决定推出:它是 measurable 之上的「已人工接受当前版本」,故优先级高于 measurable。 */
|
|
110
|
+
export type ManagedLifecycleLabel = 'discovered' | 'installed' | 'measurable' | 'stale' | 'promoted';
|
|
97
111
|
export interface DerivedManagedState {
|
|
98
112
|
label: ManagedLifecycleLabel;
|
|
99
113
|
/** 当前源 hash 与记录的 contentHash 不一致(或源已不在)。 */
|
|
@@ -43,8 +43,8 @@ export interface ObservationReviewStateEntry {
|
|
|
43
43
|
snippet?: string;
|
|
44
44
|
}
|
|
45
45
|
export interface ObservationReviewState {
|
|
46
|
-
|
|
47
|
-
schemaVersion:
|
|
46
|
+
kind: 'observe-review-state';
|
|
47
|
+
schemaVersion: 2;
|
|
48
48
|
updatedAt: string;
|
|
49
49
|
entries: Record<string, ObservationReviewStateEntry>;
|
|
50
50
|
}
|
|
@@ -184,8 +184,8 @@ export interface ObservationSessionTimeRange {
|
|
|
184
184
|
durationMs?: number;
|
|
185
185
|
}
|
|
186
186
|
export interface ObservationInboxReport {
|
|
187
|
-
|
|
188
|
-
schemaVersion:
|
|
187
|
+
kind: 'observe-inbox';
|
|
188
|
+
schemaVersion: 2;
|
|
189
189
|
meta: {
|
|
190
190
|
tracePath: string;
|
|
191
191
|
generatedAt: string;
|
|
@@ -735,8 +735,8 @@ export interface ExperienceSkillSummary {
|
|
|
735
735
|
relatedObservationIds: string[];
|
|
736
736
|
}
|
|
737
737
|
export interface ObservationExperienceReport {
|
|
738
|
-
|
|
739
|
-
schemaVersion:
|
|
738
|
+
kind: 'observe-experience';
|
|
739
|
+
schemaVersion: 2;
|
|
740
740
|
scope: 'evidence-only';
|
|
741
741
|
generatedAt: string;
|
|
742
742
|
meta: {
|
package/dist/types/report.d.ts
CHANGED
|
@@ -253,8 +253,10 @@ export interface ReportMeta {
|
|
|
253
253
|
* for eval-side reports). Value 1 was specified for v0.21+ but never emitted in practice.
|
|
254
254
|
* Value 2 marked the artifactHashes tree-hash era for local dir-skills (git dir-skills still
|
|
255
255
|
* SKILL.md-only). Value 3 extends whole-tree hashing to git dir-skills via isolated copies
|
|
256
|
-
* (all dir-skills bind).
|
|
257
|
-
*
|
|
256
|
+
* (all dir-skills bind). Value 4 keeps the v3 hash/binding semantics but marks the canonical
|
|
257
|
+
* top-level discriminant era, so external consumers can version-gate the JSON shape. Drift /
|
|
258
|
+
* lineage consumers gate on `>= 2` (tree-hash era); git-dir-skill binding additionally
|
|
259
|
+
* requires `>= 3`. */
|
|
258
260
|
schemaVersion?: number;
|
|
259
261
|
/** SHA256-12 of every sample's content (sample_id → hash). Same hash = same sample. */
|
|
260
262
|
sampleHashes?: Record<string, string>;
|
|
@@ -355,7 +357,7 @@ export interface SampleSnapshot {
|
|
|
355
357
|
tripwire?: boolean;
|
|
356
358
|
}
|
|
357
359
|
export interface EvaluationReport {
|
|
358
|
-
|
|
360
|
+
kind: 'evaluation';
|
|
359
361
|
id: string;
|
|
360
362
|
meta: ReportMeta;
|
|
361
363
|
summary: Record<string, VariantSummary>;
|
|
@@ -369,6 +371,9 @@ export interface EvaluationReport {
|
|
|
369
371
|
export type Report = EvaluationReport;
|
|
370
372
|
export interface BatchEvaluationMeta {
|
|
371
373
|
mode: 'skill';
|
|
374
|
+
/** Batch report JSON schema version. Kept in lockstep with child EvaluationReport top-level
|
|
375
|
+
* shape so external consumers can identify the canonical discriminant era. */
|
|
376
|
+
schemaVersion: number;
|
|
372
377
|
model: string;
|
|
373
378
|
executor: string;
|
|
374
379
|
skillDir: string;
|
|
@@ -408,7 +413,7 @@ export interface BatchEvaluationItem {
|
|
|
408
413
|
variance?: VarianceData;
|
|
409
414
|
}
|
|
410
415
|
export interface BatchEvaluationReport {
|
|
411
|
-
|
|
416
|
+
kind: 'batch-evaluation';
|
|
412
417
|
id: string;
|
|
413
418
|
mode: 'skill';
|
|
414
419
|
meta: BatchEvaluationMeta;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "oh-my-knowledge",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.37.0",
|
|
4
4
|
"packageManager": "yarn@4.16.0",
|
|
5
5
|
"description": "Evaluation framework for LLM knowledge inputs — prompts, RAG corpora, skills, agent workflows. Fix the model, vary the artifact. Built-in statistical rigor: bootstrap CI, Krippendorff α, length-debias, saturation curves.",
|
|
6
6
|
"type": "module",
|