oh-my-knowledge 0.41.0 → 0.43.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -2
- package/README.zh.md +7 -2
- package/dist/analysis/report-diagnostics.d.ts +8 -1
- package/dist/analysis/report-diagnostics.js +82 -1
- package/dist/artifact-graph/doctor.d.ts +21 -0
- package/dist/artifact-graph/doctor.js +569 -0
- package/dist/assets/agent-skills/omk/SKILL.md +9 -9
- package/dist/assets/agent-skills/omk/references/commands.md +2 -1
- package/dist/authoring/evolver.d.ts +3 -14
- package/dist/authoring/evolver.js +1 -52
- package/dist/authoring/generator.d.ts +24 -0
- package/dist/authoring/generator.js +33 -6
- package/dist/cli/commands/doctor.js +62 -61
- package/dist/cli/commands/eval/index.d.ts +1 -0
- package/dist/cli/commands/eval/index.js +60 -7
- package/dist/cli/commands/init.js +11 -7
- package/dist/cli/commands/observe/index.d.ts +2 -2
- package/dist/cli/commands/observe/index.js +8 -7
- package/dist/cli/commands/sample.js +22 -16
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +8 -0
- package/dist/cli/lib/i18n-dict/help.js +12 -10
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +14 -11
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +4 -0
- package/dist/cli/lib/parse-run-config/samples-discovery.d.ts +5 -7
- package/dist/cli/lib/parse-run-config/samples-discovery.js +10 -32
- package/dist/cli/lib/parse-run-config.d.ts +3 -0
- package/dist/cli/lib/parse-run-config.js +4 -4
- package/dist/cli/lib/resolve-skill-input.js +10 -12
- package/dist/doctor/messages.js +2 -2
- package/dist/eval-core/artifact-file-names.d.ts +15 -0
- package/dist/eval-core/artifact-file-names.js +46 -0
- package/dist/eval-core/evaluation-job.d.ts +2 -1
- package/dist/eval-core/evaluation-job.js +2 -1
- package/dist/eval-core/evaluation-reporting.js +9 -4
- package/dist/eval-core/holdout.d.ts +66 -0
- package/dist/eval-core/holdout.js +118 -0
- package/dist/eval-core/measurement-dirs.js +13 -7
- package/dist/eval-core/report-file-migration.d.ts +10 -0
- package/dist/eval-core/report-file-migration.js +90 -0
- package/dist/eval-core/verdict.d.ts +44 -1
- package/dist/eval-core/verdict.js +175 -13
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -1
- package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
- package/dist/eval-workflows/evaluation-pipeline.js +2 -1
- package/dist/eval-workflows/run-evaluation.d.ts +5 -2
- package/dist/eval-workflows/run-evaluation.js +8 -5
- package/dist/inputs/eval-config.js +6 -0
- package/dist/inputs/sample-locator.d.ts +23 -0
- package/dist/inputs/sample-locator.js +195 -0
- package/dist/inputs/skill-loader.js +7 -17
- package/dist/observability/inbox.js +7 -3
- package/dist/renderer/summary.js +36 -3
- package/dist/server/report-server.js +10 -4
- package/dist/server/report-store.js +17 -9
- package/dist/server/skill-index.js +16 -11
- package/dist/types/artifact-graph.d.ts +93 -0
- package/dist/types/artifact-graph.js +1 -0
- package/dist/types/eval.d.ts +7 -0
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/report.d.ts +49 -0
- package/package.json +1 -1
|
@@ -22,7 +22,7 @@ export interface EvaluationRunState {
|
|
|
22
22
|
runningJob: EvaluationJob;
|
|
23
23
|
resolvedJobStore: JobStore | null;
|
|
24
24
|
}
|
|
25
|
-
export declare function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
|
|
25
|
+
export declare function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
|
|
26
26
|
samplesPath: string;
|
|
27
27
|
skillDir: string;
|
|
28
28
|
artifacts: Artifact[];
|
|
@@ -41,6 +41,7 @@ export declare function initializeEvaluationRunState({ samplesPath, skillDir, ar
|
|
|
41
41
|
jobStore?: JobStore | null;
|
|
42
42
|
persistJob?: boolean;
|
|
43
43
|
repeat?: number;
|
|
44
|
+
holdoutRatio?: number;
|
|
44
45
|
batch?: boolean;
|
|
45
46
|
judgeRepeat?: number;
|
|
46
47
|
judgeModels?: import('../../types/index.js').JudgeConfig[];
|
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
import { buildEvaluationRequest, createFailedJob, createEvaluationRun, createQueuedJob, createSucceededJob, finalizeEvaluationRun, markJobRunning, failEvaluationRun, } from '../../eval-core/evaluation-job.js';
|
|
15
15
|
import { createFileJobStore } from '../../server/job-store.js';
|
|
16
16
|
import { DEFAULT_JOBS_DIR } from '../../eval-core/default-dirs.js';
|
|
17
|
-
export async function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
|
|
17
|
+
export async function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
|
|
18
18
|
const effectiveJudges = judgeModels && judgeModels.length > 0
|
|
19
19
|
? judgeModels
|
|
20
20
|
: [{ executor: judgeExecutorName, model: judgeModel }];
|
|
@@ -33,6 +33,7 @@ export async function initializeEvaluationRunState({ samplesPath, skillDir, arti
|
|
|
33
33
|
owner,
|
|
34
34
|
tags,
|
|
35
35
|
repeat,
|
|
36
|
+
holdoutRatio,
|
|
36
37
|
batch,
|
|
37
38
|
judgeRepeat,
|
|
38
39
|
judgeModels: effectiveJudges,
|
|
@@ -61,6 +61,8 @@ export interface EvaluationPipelineOptions {
|
|
|
61
61
|
layeredStats?: boolean;
|
|
62
62
|
/** 透传到 meta.request.repeat */
|
|
63
63
|
repeat?: number;
|
|
64
|
+
/** 透传到 meta.request.holdoutRatio;> 0 时 report-finalize 算 train/holdout 子集综合分。 */
|
|
65
|
+
holdoutRatio?: number;
|
|
64
66
|
/** 透传到 meta.request.batch */
|
|
65
67
|
batch?: boolean;
|
|
66
68
|
/** 透传到 meta.request.judgeRepeat 与 grade(),每条 sample × dimension judge N 次 */
|
|
@@ -89,7 +91,7 @@ export interface EvaluationPipelineOptions {
|
|
|
89
91
|
noDiagnostic?: boolean;
|
|
90
92
|
}
|
|
91
93
|
type VariantResult = import('../types/index.js').VariantResult;
|
|
92
|
-
export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
|
|
94
|
+
export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
|
|
93
95
|
report: Report;
|
|
94
96
|
filePath: string | null;
|
|
95
97
|
}>;
|
|
@@ -33,7 +33,7 @@ export { _computeTestSetHashForTest } from './evaluation-pipeline/test-set-hash.
|
|
|
33
33
|
export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, concurrency = 1, timeoutMs, noCache = false, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, verbose = false, retry = 0, existingResults,
|
|
34
34
|
// requires 现在由 runEvaluation 上游传给 doctor 处理; eval-pipeline 不再用
|
|
35
35
|
// 但保留接口字段,避免破 programmatic API (类型层面接收, 内部忽略)
|
|
36
|
-
requires: _requires, layeredStats = false, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
|
|
36
|
+
requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
|
|
37
37
|
const variantNames = artifacts.map((artifact) => artifact.name);
|
|
38
38
|
const runState = await initializeEvaluationRunState({
|
|
39
39
|
samplesPath,
|
|
@@ -54,6 +54,7 @@ requires: _requires, layeredStats = false, repeat, batch, judgeRepeat, judgeMode
|
|
|
54
54
|
jobStore,
|
|
55
55
|
persistJob,
|
|
56
56
|
repeat,
|
|
57
|
+
holdoutRatio,
|
|
57
58
|
batch,
|
|
58
59
|
judgeRepeat,
|
|
59
60
|
judgeModels,
|
|
@@ -32,6 +32,9 @@ interface CommonEvaluationOptions {
|
|
|
32
32
|
/** --repeat N. 1 表示单次(默认); > 1 时在 runMultiple 层聚合 variance。
|
|
33
33
|
* 记入 report.meta.request.repeat 让 meta 如实反映用户输入。 */
|
|
34
34
|
repeat?: number;
|
|
35
|
+
/** --holdout-ratio R. 0 / 缺省 = 不切分(默认)。> 0 时 report-finalize 算 train/holdout
|
|
36
|
+
* 子集综合分(report.analysis.holdout),供 verdict 过拟合门控读取。 */
|
|
37
|
+
holdoutRatio?: number;
|
|
35
38
|
/** --batch 模式标记, true 表示当前评测是 skill batch 流程。
|
|
36
39
|
* 记入 report.meta.request.batch。 */
|
|
37
40
|
batch?: boolean;
|
|
@@ -117,7 +120,7 @@ export interface DryRunReport extends DryRunBase {
|
|
|
117
120
|
samplesPath: string;
|
|
118
121
|
tasks: DryRunTask[];
|
|
119
122
|
}
|
|
120
|
-
export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
|
|
123
|
+
export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
|
|
121
124
|
report: Report | DryRunReport;
|
|
122
125
|
filePath: string | null;
|
|
123
126
|
}>;
|
|
@@ -133,7 +136,7 @@ export interface DryRunBatchReport extends DryRunBase {
|
|
|
133
136
|
artifacts: DryRunBatchSkill[];
|
|
134
137
|
}
|
|
135
138
|
export declare function buildVarianceData(runs: Report[], bootstrapSamples?: number, seed?: number): VarianceData | null;
|
|
136
|
-
export declare function runBatchEvaluation({ skillDir, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, executorName, jobStore, persistJob, onProgress, onSkillProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, repeat, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache, strictBaseline, variantAllowedSkills, }: RunBatchEvaluationOptions): Promise<{
|
|
139
|
+
export declare function runBatchEvaluation({ skillDir, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, executorName, jobStore, persistJob, onProgress, onSkillProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, repeat, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache, strictBaseline, variantAllowedSkills, }: RunBatchEvaluationOptions): Promise<{
|
|
137
140
|
report: BatchEvaluationReport | DryRunBatchReport;
|
|
138
141
|
filePath: string | null;
|
|
139
142
|
}>;
|
|
@@ -8,7 +8,7 @@ import { buildDryRunTaskReport, prepareEvaluationRun, } from './evaluation-prepa
|
|
|
8
8
|
import { executeEvaluationPipeline } from './evaluation-pipeline.js';
|
|
9
9
|
import { findSaturationPoint } from '../analysis/saturation.js';
|
|
10
10
|
import { bootstrapMeanCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
|
|
11
|
-
export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
|
|
11
|
+
export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
|
|
12
12
|
// Unified judgeModels → derive single-judge fields for downstream pipeline / grading
|
|
13
13
|
// (which still operate on string `judgeModel` + `judgeExecutorName` fields per call).
|
|
14
14
|
const effectiveJudgeModels = judgeModels && judgeModels.length > 0
|
|
@@ -165,6 +165,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
165
165
|
requires,
|
|
166
166
|
layeredStats,
|
|
167
167
|
repeat,
|
|
168
|
+
holdoutRatio,
|
|
168
169
|
batch,
|
|
169
170
|
judgeRepeat,
|
|
170
171
|
judgeModels,
|
|
@@ -366,7 +367,7 @@ export function buildVarianceData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAM
|
|
|
366
367
|
...(saturation ? { saturation } : {}),
|
|
367
368
|
};
|
|
368
369
|
}
|
|
369
|
-
export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, onSkillProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, repeat, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache = false, strictBaseline, variantAllowedSkills, }) {
|
|
370
|
+
export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, onSkillProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, repeat, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache = false, strictBaseline, variantAllowedSkills, }) {
|
|
370
371
|
// Same unified judge derivation as runEvaluation (downstream pipeline / report build
|
|
371
372
|
// still uses single judgeModel + judgeExecutorName per call).
|
|
372
373
|
const effectiveJudgeModels = judgeModels && judgeModels.length > 0
|
|
@@ -485,12 +486,14 @@ export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outp
|
|
|
485
486
|
strictBaseline,
|
|
486
487
|
variantAllowedSkills,
|
|
487
488
|
runSingleEvaluation: async (options) => {
|
|
488
|
-
// repeat > 1 时走 runMultiple 做 variance; batch=true 标记让 meta.request
|
|
489
|
+
// repeat > 1 时走 runMultiple 做 variance; batch=true 标记让 meta.request 如实反映。
|
|
490
|
+
// holdoutRatio 显式转发(executeBatchEvaluationRuns 不一定把它塞进 options),否则
|
|
491
|
+
// batch 子报告 meta.request.holdoutRatio 丢失、过拟合门控对 batch 静默失效。
|
|
489
492
|
if (repeat && repeat > 1) {
|
|
490
|
-
const multi = await runMultiple({ ...options, repeat, batch: true, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
|
|
493
|
+
const multi = await runMultiple({ ...options, repeat, holdoutRatio, batch: true, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
|
|
491
494
|
return { report: multi.report, filePath: multi.filePath };
|
|
492
495
|
}
|
|
493
|
-
const result = await runEvaluation({ ...options, batch: true, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
|
|
496
|
+
const result = await runEvaluation({ ...options, batch: true, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
|
|
494
497
|
return { report: result.report, filePath: result.filePath };
|
|
495
498
|
},
|
|
496
499
|
});
|
|
@@ -171,6 +171,11 @@ function validateEvalConfig(parsed, configPath) {
|
|
|
171
171
|
};
|
|
172
172
|
assertPositiveIntOpt('repeat');
|
|
173
173
|
assertPositiveIntOpt('judgeRepeat');
|
|
174
|
+
if (obj.holdoutRatio !== undefined) {
|
|
175
|
+
if (typeof obj.holdoutRatio !== 'number' || !Number.isFinite(obj.holdoutRatio) || obj.holdoutRatio <= 0 || obj.holdoutRatio >= 1) {
|
|
176
|
+
throw new Error(`${configPath}: holdoutRatio must be a number in (0, 1)`);
|
|
177
|
+
}
|
|
178
|
+
}
|
|
174
179
|
if (obj.bootstrapSamples !== undefined) {
|
|
175
180
|
if (typeof obj.bootstrapSamples !== 'number' || !Number.isFinite(obj.bootstrapSamples) || obj.bootstrapSamples < 100) {
|
|
176
181
|
throw new Error(`${configPath}: bootstrapSamples must be a number ≥ 100`);
|
|
@@ -256,6 +261,7 @@ function validateEvalConfig(parsed, configPath) {
|
|
|
256
261
|
variants,
|
|
257
262
|
budget,
|
|
258
263
|
repeat: obj.repeat,
|
|
264
|
+
holdoutRatio: obj.holdoutRatio,
|
|
259
265
|
judgeRepeat: obj.judgeRepeat,
|
|
260
266
|
bootstrap: obj.bootstrap,
|
|
261
267
|
bootstrapSamples: obj.bootstrapSamples,
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
export declare const PROJECT_SAMPLE_FILENAMES: readonly ["eval-samples.json", "eval-samples.yaml", "eval-samples.yml"];
|
|
2
|
+
export declare const SKILL_LOCAL_SAMPLE_FILENAMES: readonly ["samples.json", "samples.yaml", "samples.yml"];
|
|
3
|
+
export interface DeprecatedSkillSamplesHint {
|
|
4
|
+
oldPath: string;
|
|
5
|
+
newPath: string;
|
|
6
|
+
}
|
|
7
|
+
export declare function isExistingDirectory(path: string): boolean;
|
|
8
|
+
export declare function hasLoadableSampleFile(dir: string): boolean;
|
|
9
|
+
export declare function findProjectSamplesFile(dir: string): string | null;
|
|
10
|
+
export declare function skillLocalSamplesDir(skillRoot: string): string;
|
|
11
|
+
export declare function defaultSkillLocalSamplesFile(skillRoot: string): string;
|
|
12
|
+
export declare function hasUsableSamplesPath(path: string): boolean;
|
|
13
|
+
export declare function findSkillLocalSamplesDir(skillRoot: string): string | null;
|
|
14
|
+
export declare function findSkillSamplesPath(skillRoot: string): string | null;
|
|
15
|
+
export declare function findDeprecatedSkillSamplesHint(skillRoot: string): DeprecatedSkillSamplesHint | null;
|
|
16
|
+
export declare function findFlatSkillSamplesPath(skillDir: string, skillName: string): string | null;
|
|
17
|
+
export declare function defaultFlatSkillSamplesFile(skillDir: string, skillName: string): string;
|
|
18
|
+
export declare function isDirectorySkillRoot(path: string): boolean;
|
|
19
|
+
export declare function findNamedSkillSamplesPath(skillDir: string, skillName: string): string | null;
|
|
20
|
+
export declare function findSingleTreatmentSamplesPath(treatmentExpr: string, skillDir: string, cwd?: string): string | null;
|
|
21
|
+
export declare function findSingleTreatmentDeprecatedSamplesHint(treatmentExpr: string, skillDir: string, cwd?: string): DeprecatedSkillSamplesHint | null;
|
|
22
|
+
export declare function findDoctorSamplesPath(target: string | null, cwd: string): string | null;
|
|
23
|
+
export declare function findDoctorDeprecatedSamplesHint(target: string | null, cwd: string): DeprecatedSkillSamplesHint | null;
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
import { existsSync, readdirSync, statSync } from 'node:fs';
|
|
2
|
+
import { basename, dirname, join, resolve } from 'node:path';
|
|
3
|
+
export const PROJECT_SAMPLE_FILENAMES = ['eval-samples.json', 'eval-samples.yaml', 'eval-samples.yml'];
|
|
4
|
+
export const SKILL_LOCAL_SAMPLE_FILENAMES = ['samples.json', 'samples.yaml', 'samples.yml'];
|
|
5
|
+
const SAMPLE_FILE_RE = /\.(json|ya?ml)$/i;
|
|
6
|
+
const RESERVED_SAMPLE_FILE_RE = /^(report|health|_)/i;
|
|
7
|
+
function isExistingFile(path) {
|
|
8
|
+
try {
|
|
9
|
+
return existsSync(path) && statSync(path).isFile();
|
|
10
|
+
}
|
|
11
|
+
catch {
|
|
12
|
+
return false;
|
|
13
|
+
}
|
|
14
|
+
}
|
|
15
|
+
export function isExistingDirectory(path) {
|
|
16
|
+
try {
|
|
17
|
+
return existsSync(path) && statSync(path).isDirectory();
|
|
18
|
+
}
|
|
19
|
+
catch {
|
|
20
|
+
return false;
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
export function hasLoadableSampleFile(dir) {
|
|
24
|
+
if (!isExistingDirectory(dir))
|
|
25
|
+
return false;
|
|
26
|
+
try {
|
|
27
|
+
return readdirSync(dir).some((file) => SAMPLE_FILE_RE.test(file) && !RESERVED_SAMPLE_FILE_RE.test(file));
|
|
28
|
+
}
|
|
29
|
+
catch {
|
|
30
|
+
return false;
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
export function findProjectSamplesFile(dir) {
|
|
34
|
+
for (const name of PROJECT_SAMPLE_FILENAMES) {
|
|
35
|
+
const candidate = join(dir, name);
|
|
36
|
+
if (isExistingFile(candidate))
|
|
37
|
+
return candidate;
|
|
38
|
+
}
|
|
39
|
+
return null;
|
|
40
|
+
}
|
|
41
|
+
export function skillLocalSamplesDir(skillRoot) {
|
|
42
|
+
return join(skillRoot, '.omk');
|
|
43
|
+
}
|
|
44
|
+
export function defaultSkillLocalSamplesFile(skillRoot) {
|
|
45
|
+
return join(skillLocalSamplesDir(skillRoot), SKILL_LOCAL_SAMPLE_FILENAMES[0]);
|
|
46
|
+
}
|
|
47
|
+
export function hasUsableSamplesPath(path) {
|
|
48
|
+
if (isExistingFile(path))
|
|
49
|
+
return true;
|
|
50
|
+
return hasLoadableSampleFile(path);
|
|
51
|
+
}
|
|
52
|
+
export function findSkillLocalSamplesDir(skillRoot) {
|
|
53
|
+
const dir = skillLocalSamplesDir(skillRoot);
|
|
54
|
+
return hasLoadableSampleFile(dir) ? dir : null;
|
|
55
|
+
}
|
|
56
|
+
export function findSkillSamplesPath(skillRoot) {
|
|
57
|
+
return findSkillLocalSamplesDir(skillRoot);
|
|
58
|
+
}
|
|
59
|
+
export function findDeprecatedSkillSamplesHint(skillRoot) {
|
|
60
|
+
if (!isDirectorySkillRoot(skillRoot))
|
|
61
|
+
return null;
|
|
62
|
+
for (const name of PROJECT_SAMPLE_FILENAMES) {
|
|
63
|
+
const oldPath = join(skillRoot, name);
|
|
64
|
+
if (isExistingFile(oldPath)) {
|
|
65
|
+
return { oldPath, newPath: defaultSkillLocalSamplesFile(skillRoot) };
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
return null;
|
|
69
|
+
}
|
|
70
|
+
export function findFlatSkillSamplesPath(skillDir, skillName) {
|
|
71
|
+
for (const ext of ['json', 'yaml', 'yml']) {
|
|
72
|
+
const candidate = join(skillDir, `${skillName}.eval-samples.${ext}`);
|
|
73
|
+
if (isExistingFile(candidate))
|
|
74
|
+
return candidate;
|
|
75
|
+
}
|
|
76
|
+
return null;
|
|
77
|
+
}
|
|
78
|
+
export function defaultFlatSkillSamplesFile(skillDir, skillName) {
|
|
79
|
+
return join(skillDir, `${skillName}.eval-samples.json`);
|
|
80
|
+
}
|
|
81
|
+
export function isDirectorySkillRoot(path) {
|
|
82
|
+
return isExistingDirectory(path) && isExistingFile(join(path, 'SKILL.md'));
|
|
83
|
+
}
|
|
84
|
+
export function findNamedSkillSamplesPath(skillDir, skillName) {
|
|
85
|
+
const flatSkillPath = join(skillDir, `${skillName}.md`);
|
|
86
|
+
if (isExistingFile(flatSkillPath)) {
|
|
87
|
+
return findFlatSkillSamplesPath(skillDir, skillName);
|
|
88
|
+
}
|
|
89
|
+
const dirSkillRoot = join(skillDir, skillName);
|
|
90
|
+
if (isDirectorySkillRoot(dirSkillRoot)) {
|
|
91
|
+
return findSkillSamplesPath(dirSkillRoot);
|
|
92
|
+
}
|
|
93
|
+
return null;
|
|
94
|
+
}
|
|
95
|
+
function findSamplesForExistingSkillPath(path) {
|
|
96
|
+
if (isExistingDirectory(path)) {
|
|
97
|
+
return findSkillSamplesPath(path);
|
|
98
|
+
}
|
|
99
|
+
const parent = dirname(path);
|
|
100
|
+
if (basename(path) === 'SKILL.md') {
|
|
101
|
+
return findSkillSamplesPath(parent);
|
|
102
|
+
}
|
|
103
|
+
if (/\.md$/i.test(path)) {
|
|
104
|
+
const skillName = basename(path).replace(/\.md$/i, '');
|
|
105
|
+
return findFlatSkillSamplesPath(parent, skillName);
|
|
106
|
+
}
|
|
107
|
+
return findProjectSamplesFile(parent);
|
|
108
|
+
}
|
|
109
|
+
function findDeprecatedSamplesForExistingSkillPath(path) {
|
|
110
|
+
if (isExistingDirectory(path)) {
|
|
111
|
+
return findDeprecatedSkillSamplesHint(path);
|
|
112
|
+
}
|
|
113
|
+
if (basename(path) === 'SKILL.md') {
|
|
114
|
+
return findDeprecatedSkillSamplesHint(dirname(path));
|
|
115
|
+
}
|
|
116
|
+
return null;
|
|
117
|
+
}
|
|
118
|
+
export function findSingleTreatmentSamplesPath(treatmentExpr, skillDir, cwd = process.cwd()) {
|
|
119
|
+
const resolved = resolve(cwd, treatmentExpr);
|
|
120
|
+
if (existsSync(resolved)) {
|
|
121
|
+
const samplesPath = findSamplesForExistingSkillPath(resolved);
|
|
122
|
+
if (samplesPath)
|
|
123
|
+
return samplesPath;
|
|
124
|
+
}
|
|
125
|
+
return findNamedSkillSamplesPath(skillDir, treatmentExpr);
|
|
126
|
+
}
|
|
127
|
+
export function findSingleTreatmentDeprecatedSamplesHint(treatmentExpr, skillDir, cwd = process.cwd()) {
|
|
128
|
+
const resolved = resolve(cwd, treatmentExpr);
|
|
129
|
+
if (existsSync(resolved)) {
|
|
130
|
+
return findDeprecatedSamplesForExistingSkillPath(resolved);
|
|
131
|
+
}
|
|
132
|
+
const flatSkillPath = join(skillDir, `${treatmentExpr}.md`);
|
|
133
|
+
if (isExistingFile(flatSkillPath))
|
|
134
|
+
return null;
|
|
135
|
+
return findDeprecatedSkillSamplesHint(join(skillDir, treatmentExpr));
|
|
136
|
+
}
|
|
137
|
+
function projectSampleSearchDirs(target, cwd) {
|
|
138
|
+
const dirs = [];
|
|
139
|
+
const add = (dir) => {
|
|
140
|
+
const abs = resolve(dir);
|
|
141
|
+
if (!dirs.includes(abs))
|
|
142
|
+
dirs.push(abs);
|
|
143
|
+
};
|
|
144
|
+
if (target) {
|
|
145
|
+
const absTarget = resolve(target);
|
|
146
|
+
if (existsSync(absTarget)) {
|
|
147
|
+
if (isExistingDirectory(absTarget)) {
|
|
148
|
+
if (isDirectorySkillRoot(absTarget)) {
|
|
149
|
+
add(dirname(dirname(absTarget)));
|
|
150
|
+
}
|
|
151
|
+
else {
|
|
152
|
+
add(absTarget);
|
|
153
|
+
add(dirname(absTarget));
|
|
154
|
+
add(dirname(dirname(absTarget)));
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
else {
|
|
158
|
+
const parent = dirname(absTarget);
|
|
159
|
+
if (basename(absTarget) === 'SKILL.md') {
|
|
160
|
+
add(dirname(dirname(parent)));
|
|
161
|
+
}
|
|
162
|
+
else {
|
|
163
|
+
add(parent);
|
|
164
|
+
add(dirname(parent));
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
add(cwd);
|
|
170
|
+
return dirs;
|
|
171
|
+
}
|
|
172
|
+
export function findDoctorSamplesPath(target, cwd) {
|
|
173
|
+
if (target) {
|
|
174
|
+
const absTarget = resolve(target);
|
|
175
|
+
if (existsSync(absTarget)) {
|
|
176
|
+
const targetSamples = findSamplesForExistingSkillPath(absTarget);
|
|
177
|
+
if (targetSamples)
|
|
178
|
+
return targetSamples;
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
for (const dir of projectSampleSearchDirs(target, cwd)) {
|
|
182
|
+
const samplesPath = findProjectSamplesFile(dir);
|
|
183
|
+
if (samplesPath)
|
|
184
|
+
return samplesPath;
|
|
185
|
+
}
|
|
186
|
+
return null;
|
|
187
|
+
}
|
|
188
|
+
export function findDoctorDeprecatedSamplesHint(target, cwd) {
|
|
189
|
+
if (!target)
|
|
190
|
+
return null;
|
|
191
|
+
const absTarget = resolve(cwd, target);
|
|
192
|
+
if (!existsSync(absTarget))
|
|
193
|
+
return null;
|
|
194
|
+
return findDeprecatedSamplesForExistingSkillPath(absTarget);
|
|
195
|
+
}
|
|
@@ -5,6 +5,7 @@ import { execFileSync } from 'node:child_process';
|
|
|
5
5
|
import { extractSkillHardRules, extractSkillWorkflows } from '../shared/hard-rules.js';
|
|
6
6
|
import { hashArtifactSource, hashBytes, isDistributablePath } from './content-hash.js';
|
|
7
7
|
import { materializeIsolatedCopy } from './materialize-copy.js';
|
|
8
|
+
import { findFlatSkillSamplesPath, findSkillSamplesPath } from './sample-locator.js';
|
|
8
9
|
function parseFrontmatterPreflight(content) {
|
|
9
10
|
const match = content.match(/^---\r?\n([\s\S]*?)\r?\n---/);
|
|
10
11
|
if (!match)
|
|
@@ -388,14 +389,9 @@ export function discoverBatchSkills(skillDir) {
|
|
|
388
389
|
const mdMatch = entry.endsWith('.md') && !entry.endsWith('.eval-samples.json');
|
|
389
390
|
if (mdMatch) {
|
|
390
391
|
const name = entry.slice(0, -3);
|
|
391
|
-
const
|
|
392
|
-
|
|
393
|
-
join(skillDir,
|
|
394
|
-
join(skillDir, `${name}.eval-samples.yaml`),
|
|
395
|
-
join(skillDir, `${name}.eval-samples.yml`),
|
|
396
|
-
].filter(existsSync);
|
|
397
|
-
if (candidates.length > 0) {
|
|
398
|
-
skills.push({ name, skillPath: join(skillDir, entry), samplesPath: candidates[0] });
|
|
392
|
+
const samplesPath = findFlatSkillSamplesPath(skillDir, name);
|
|
393
|
+
if (samplesPath) {
|
|
394
|
+
skills.push({ name, skillPath: join(skillDir, entry), samplesPath });
|
|
399
395
|
}
|
|
400
396
|
else {
|
|
401
397
|
warned.push(name);
|
|
@@ -406,15 +402,9 @@ export function discoverBatchSkills(skillDir) {
|
|
|
406
402
|
const skillMd = join(entryPath, 'SKILL.md');
|
|
407
403
|
if (!existsSync(skillMd))
|
|
408
404
|
continue;
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
const
|
|
412
|
-
...(existsSync(omkDir) ? [omkDir] : []),
|
|
413
|
-
join(entryPath, 'eval-samples.json'),
|
|
414
|
-
join(entryPath, 'eval-samples.yaml'),
|
|
415
|
-
join(entryPath, 'eval-samples.yml'),
|
|
416
|
-
];
|
|
417
|
-
const samplesPath = candidates.find(existsSync);
|
|
405
|
+
if (existsSync(join(skillDir, `${entry}.md`)))
|
|
406
|
+
continue;
|
|
407
|
+
const samplesPath = findSkillSamplesPath(entryPath);
|
|
418
408
|
if (samplesPath) {
|
|
419
409
|
skills.push({ name: entry, skillPath: skillMd, samplesPath });
|
|
420
410
|
}
|
|
@@ -2,13 +2,15 @@ import { createHash } from 'node:crypto';
|
|
|
2
2
|
import { existsSync, mkdirSync, readdirSync, readFileSync, writeFileSync } from 'node:fs';
|
|
3
3
|
import { join } from 'node:path';
|
|
4
4
|
import { OMK_HOME } from '../eval-core/default-dirs.js';
|
|
5
|
+
import { isReportFileName, reportFilePath } from '../eval-core/artifact-file-names.js';
|
|
6
|
+
import { migrateLegacyReportFiles } from '../eval-core/report-file-migration.js';
|
|
5
7
|
import { extractGapSignalsFromTrace } from '../analysis/gap-analyzer.js';
|
|
6
8
|
import { ccTracesToResultEntries } from './trace-adapter.js';
|
|
7
9
|
import { isSearchToolCall, toolCallQuery } from '../shared/tool-search.js';
|
|
8
10
|
import { durationMsBetween } from '../shared/time.js';
|
|
9
11
|
import { buildObservationExperienceReport, normalizeObservationExperienceReport, } from './experience.js';
|
|
10
12
|
// observe inbox(观测收件箱)产物根目录。导出名沿用 *_OBSERVATIONS_DIR 以少动 importer,
|
|
11
|
-
// 但落盘目录已统一到 observe-inbox 词根(命令 omk observe inbox /
|
|
13
|
+
// 但落盘目录已统一到 observe-inbox 词根(命令 omk observe inbox / kind observe-inbox)。
|
|
12
14
|
// 项目级 .omk/observe-inbox 优先、全局兜底 —— 这套 project/global 归属是既有正常行为,本次只改名不改归属。
|
|
13
15
|
export const DEFAULT_PROJECT_OBSERVATIONS_DIR = join(process.cwd(), '.omk', 'observe-inbox');
|
|
14
16
|
export const DEFAULT_GLOBAL_OBSERVATIONS_DIR = join(OMK_HOME, 'observe-inbox');
|
|
@@ -529,10 +531,11 @@ function compareInboxItems(a, b) {
|
|
|
529
531
|
}
|
|
530
532
|
export function saveObservationInboxReport(report, outDir = DEFAULT_OBSERVATIONS_DIR) {
|
|
531
533
|
mkdirSync(outDir, { recursive: true });
|
|
534
|
+
migrateLegacyReportFiles(outDir, 'observe-inbox');
|
|
532
535
|
// 保留毫秒;同秒不同毫秒生成的两份 report 不应静默互相覆盖。
|
|
533
536
|
// 例: '2026-05-07T12:00:00.999Z' → '2026-05-07T12-00-00-999'
|
|
534
537
|
const stamp = report.meta.generatedAt.replace(/[:.]/g, '-').replace(/Z$/, '');
|
|
535
|
-
const path =
|
|
538
|
+
const path = reportFilePath(outDir, stamp);
|
|
536
539
|
writeFileSync(path, JSON.stringify(report, null, 2));
|
|
537
540
|
return path;
|
|
538
541
|
}
|
|
@@ -543,8 +546,9 @@ export function loadObservationInboxReports(dir = DEFAULT_OBSERVATIONS_DIR) {
|
|
|
543
546
|
}
|
|
544
547
|
return [];
|
|
545
548
|
}
|
|
549
|
+
migrateLegacyReportFiles(dir, 'observe-inbox');
|
|
546
550
|
return readdirSync(dir)
|
|
547
|
-
.filter(
|
|
551
|
+
.filter(isReportFileName)
|
|
548
552
|
.map((file) => {
|
|
549
553
|
try {
|
|
550
554
|
const report = normalizeObservationInboxReport(JSON.parse(readFileSync(join(dir, file), 'utf-8')));
|
package/dist/renderer/summary.js
CHANGED
|
@@ -101,6 +101,30 @@ function computeMedianCVPercent(report) {
|
|
|
101
101
|
const stab = medianStabilityCV(report);
|
|
102
102
|
return stab ? stab.cv * 100 : null;
|
|
103
103
|
}
|
|
104
|
+
// Verdict caveats(过拟合 / 知识缺口)渲染进 pill —— 让 HTML 报告和 CLI 说同一件事:
|
|
105
|
+
// CLI 在 verbose rationale 里给这两条,HTML 之前只剩一个泛化后的 level、看不到触发原因。
|
|
106
|
+
// 用 result.caveats 的结构化数据 i18n,而不是重解析 zh rationale 串。
|
|
107
|
+
function renderVerdictCaveats(caveats, lang) {
|
|
108
|
+
if (!caveats)
|
|
109
|
+
return '';
|
|
110
|
+
const lines = [];
|
|
111
|
+
if (caveats.overfitting) {
|
|
112
|
+
const c = caveats.overfitting;
|
|
113
|
+
lines.push(lang === 'zh'
|
|
114
|
+
? `⚠ 过拟合敞口:${c.variant} 训练 ${c.trainScore.toFixed(2)} / 留出 ${c.holdoutScore.toFixed(2)}(差 ${c.gap.toFixed(2)}),提升可能不泛化`
|
|
115
|
+
: `⚠ Overfitting: ${c.variant} train ${c.trainScore.toFixed(2)} / holdout ${c.holdoutScore.toFixed(2)} (gap ${c.gap.toFixed(2)}) — gain may not generalize`);
|
|
116
|
+
}
|
|
117
|
+
if (caveats.gapSignal) {
|
|
118
|
+
const g = caveats.gapSignal;
|
|
119
|
+
const wm = g.testSetHash ? g.testSetHash.slice(0, 8) : (g.testSetPath ?? '');
|
|
120
|
+
lines.push(lang === 'zh'
|
|
121
|
+
? `知识缺口率 ${g.gapRatePct}%(test set ${wm},informational)`
|
|
122
|
+
: `Knowledge gap ${g.gapRatePct}% (test set ${wm}, informational)`);
|
|
123
|
+
}
|
|
124
|
+
if (lines.length === 0)
|
|
125
|
+
return '';
|
|
126
|
+
return `<div class="page-verdict-caveats">${lines.map((l) => `<span class="page-verdict-caveat">${e(l)}</span>`).join('')}</div>`;
|
|
127
|
+
}
|
|
104
128
|
export function renderVerdictPill(report, lang) {
|
|
105
129
|
let result;
|
|
106
130
|
try {
|
|
@@ -110,8 +134,16 @@ export function renderVerdictPill(report, lang) {
|
|
|
110
134
|
return '';
|
|
111
135
|
}
|
|
112
136
|
const level = result.level;
|
|
113
|
-
|
|
137
|
+
// representative = top-level worst pair(与 CLI 同口径),不是第一对。多 treatment 报告里
|
|
138
|
+
// worst pair 不一定是 perPair[0],用它才不会把错的 treatment 名写进结论。fallback 兼容旧路径。
|
|
139
|
+
const pair = result.representative ?? result.perPair?.[0];
|
|
114
140
|
const oneLine = verdictOneLine(level, lang, pair?.treatment, pair?.control);
|
|
141
|
+
// Δ/CI 证据必须跟文案指同一对:按 representative 匹配对应的 pairComparison(alpha 也走这对),
|
|
142
|
+
// 否则多 treatment 报告会出现「文案 t2、数字 t1」的混搭。匹配不到 / 无 representative 时 fallback [0]。
|
|
143
|
+
const pairComparisons = report.meta?.pairComparisons;
|
|
144
|
+
const activeComparison = (pair
|
|
145
|
+
? pairComparisons?.find((p) => p.treatment === pair.treatment && p.control === pair.control)
|
|
146
|
+
: undefined) ?? pairComparisons?.[0];
|
|
115
147
|
const tooltip = levelTooltip(level, lang);
|
|
116
148
|
const prefix = lang === 'zh' ? '测评结论' : 'Verdict';
|
|
117
149
|
// 机器可读 enum 永远是 level token; 显示给用户的文字按 lang i18n.
|
|
@@ -120,7 +152,7 @@ export function renderVerdictPill(report, lang) {
|
|
|
120
152
|
// hero 只放「答案」: 分差是 verdict 的核心证据数字, 单独一枚 chip。
|
|
121
153
|
// 评测规模 (用例数 × 轮次) 走「实验配置」section 的 subtitle 那条 canonical 路径,
|
|
122
154
|
// 不在 hero 里重复; CV / CI 走 chip tooltip + 方法学审计 / 波动表。
|
|
123
|
-
const ci =
|
|
155
|
+
const ci = activeComparison?.diffBootstrapCI;
|
|
124
156
|
const cvPct = computeMedianCVPercent(report);
|
|
125
157
|
const metrics = [];
|
|
126
158
|
if (ci) {
|
|
@@ -128,7 +160,7 @@ export function renderVerdictPill(report, lang) {
|
|
|
128
160
|
const cvSuffix = cvPct != null
|
|
129
161
|
? (lang === 'zh' ? `;多轮稳定性 CV=${cvPct.toFixed(1)}% (${cvPct < 5 ? '稳' : cvPct <= STABILITY_UNSTABLE_CV * 100 ? '中' : '不稳'})` : `; CV=${cvPct.toFixed(1)}% (${cvPct < 5 ? 'stable' : cvPct <= STABILITY_UNSTABLE_CV * 100 ? 'moderate' : 'unstable'})`)
|
|
130
162
|
: '';
|
|
131
|
-
const pctLabel = ciLevelLabel(
|
|
163
|
+
const pctLabel = ciLevelLabel(activeComparison?.alpha);
|
|
132
164
|
const ciTipBase = lang === 'zh'
|
|
133
165
|
? `实验组与对照组综合分均值差(Δ)。bootstrap ${pctLabel} 可信区间 [${ci.low}, ${ci.high}],${ci.significant ? '不含 0 = 差异显著' : '跨过 0 = 差异不显著'}${cvSuffix}`
|
|
134
166
|
: `Treatment minus control mean composite score (Δ). Bootstrap ${pctLabel} CI [${ci.low}, ${ci.high}], ${ci.significant ? 'excludes 0 ⇒ significant' : 'spans 0 ⇒ not significant'}${cvSuffix}`;
|
|
@@ -146,6 +178,7 @@ export function renderVerdictPill(report, lang) {
|
|
|
146
178
|
<span class="page-verdict-badge"><span class="page-verdict-badge-dot" aria-hidden="true">●</span>${e(levelDisplay)}</span>
|
|
147
179
|
<span class="page-verdict-text">${e(oneLine)}</span>
|
|
148
180
|
</div>
|
|
181
|
+
${renderVerdictCaveats(result.caveats, lang)}
|
|
149
182
|
${metricChips ? `<div class="page-verdict-metrics">${metricChips}</div>` : ''}
|
|
150
183
|
</section>`;
|
|
151
184
|
}
|
|
@@ -14,6 +14,8 @@ import { renderManagedList, renderManagedHistory } from '../renderer/managed-his
|
|
|
14
14
|
import { DEFAULT_JOBS_DIR } from '../eval-core/default-dirs.js';
|
|
15
15
|
import { resolveObserveHealthDir, projectObserveHealthDir, resolveDoctorsDir, projectDoctorsDir, projectReportsDir, globalReportsDir } from '../eval-core/measurement-dirs.js';
|
|
16
16
|
import { listObserveCards, listDoctorCards, listLiveObserveCards } from '../eval-core/artifact-index.js';
|
|
17
|
+
import { isReportFileName, reportFilePath, reportFileStem } from '../eval-core/artifact-file-names.js';
|
|
18
|
+
import { migrateLegacyReportFiles } from '../eval-core/report-file-migration.js';
|
|
17
19
|
import { buildSkillIndex } from './skill-index.js';
|
|
18
20
|
import { createFileJobStore } from './job-store.js';
|
|
19
21
|
import { createFileStore, queryJob, queryJobList, queryRun, queryRunList, queryTrend } from './report-store.js';
|
|
@@ -46,18 +48,20 @@ function loadChartJsBundle() {
|
|
|
46
48
|
}
|
|
47
49
|
function listAnalyses(dir, includeCards = false) {
|
|
48
50
|
const items = [];
|
|
51
|
+
migrateLegacyReportFiles(dir, 'observe-health');
|
|
49
52
|
// live 扫描 dir 存在才做;dir 不存在(默认机器级模式下当前项目还没 .omk/observe-health、全局也空)时 live 为空,
|
|
50
53
|
// 但**不能早退** —— 后面仍要按 includeCards 合并别项目卡片,否则 observe 列表会与合卡片的 /api/skills 口径分裂。
|
|
51
54
|
if (existsSync(dir)) {
|
|
52
55
|
for (const file of readdirSync(dir)) {
|
|
53
|
-
|
|
56
|
+
const id = reportFileStem(file);
|
|
57
|
+
if (!id)
|
|
54
58
|
continue;
|
|
55
59
|
try {
|
|
56
60
|
const data = JSON.parse(readFileSync(join(dir, file), 'utf-8'));
|
|
57
61
|
if (!data.meta || !data.overall)
|
|
58
62
|
continue;
|
|
59
63
|
items.push({
|
|
60
|
-
id
|
|
64
|
+
id,
|
|
61
65
|
generatedAt: data.meta.generatedAt,
|
|
62
66
|
sessionCount: data.meta.sessionCount,
|
|
63
67
|
segmentCount: data.meta.segmentCount,
|
|
@@ -89,7 +93,8 @@ function listAnalyses(dir, includeCards = false) {
|
|
|
89
93
|
return items;
|
|
90
94
|
}
|
|
91
95
|
function loadAnalysis(dir, id, includeCards = false) {
|
|
92
|
-
|
|
96
|
+
migrateLegacyReportFiles(dir, 'observe-health');
|
|
97
|
+
const path = reportFilePath(dir, id);
|
|
93
98
|
if (existsSync(path)) {
|
|
94
99
|
try {
|
|
95
100
|
return JSON.parse(readFileSync(path, 'utf-8'));
|
|
@@ -114,9 +119,10 @@ function loadAnalysis(dir, id, includeCards = false) {
|
|
|
114
119
|
* 优先返回含该 skill 的那份;都不含时回退首个 id 命中(单 skill / 无参行为不变)。 */
|
|
115
120
|
function loadDoctorReport(dir, id, skillName, includeCards = false) {
|
|
116
121
|
let fallback = null;
|
|
122
|
+
migrateLegacyReportFiles(dir, 'doctor');
|
|
117
123
|
if (existsSync(dir)) {
|
|
118
124
|
for (const file of readdirSync(dir)) {
|
|
119
|
-
if (!file
|
|
125
|
+
if (!isReportFileName(file))
|
|
120
126
|
continue;
|
|
121
127
|
try {
|
|
122
128
|
const data = JSON.parse(readFileSync(join(dir, file), 'utf-8'));
|