oh-my-knowledge 0.48.0 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -18
- package/README.zh.md +55 -23
- package/dist/analysis/coverage-analyzer.d.ts +1 -0
- package/dist/analysis/coverage-analyzer.js +125 -62
- package/dist/analysis/failure-clusterer.js +2 -1
- package/dist/analysis/gap-analyzer.d.ts +2 -2
- package/dist/analysis/gap-analyzer.js +13 -3
- package/dist/analysis/hedging-classifier.d.ts +2 -2
- package/dist/analysis/hedging-classifier.js +3 -4
- package/dist/analysis/report-diagnostics.js +9 -7
- package/dist/analysis/sample-diagnostics.js +6 -6
- package/dist/artifact-graph/doctor.js +15 -7
- package/dist/assets/agent-skills/omk/SKILL.md +27 -7
- package/dist/assets/agent-skills/omk/references/commands.md +18 -17
- package/dist/authoring/evolver.d.ts +10 -6
- package/dist/authoring/evolver.js +496 -83
- package/dist/authoring/generator.d.ts +3 -3
- package/dist/authoring/generator.js +5 -10
- package/dist/authoring/sample-fixer.d.ts +8 -6
- package/dist/authoring/sample-fixer.js +76 -5
- package/dist/cli/commands/doctor.js +31 -14
- package/dist/cli/commands/eval/index.d.ts +3 -0
- package/dist/cli/commands/eval/index.js +163 -17
- package/dist/cli/commands/evolve.d.ts +4 -4
- package/dist/cli/commands/evolve.js +27 -13
- package/dist/cli/commands/init.js +16 -3
- package/dist/cli/commands/observe/inbox.js +28 -21
- package/dist/cli/commands/observe/index.js +20 -11
- package/dist/cli/commands/observe/ingest.d.ts +3 -0
- package/dist/cli/commands/observe/ingest.js +30 -2
- package/dist/cli/commands/sample.d.ts +6 -3
- package/dist/cli/commands/sample.js +72 -68
- package/dist/cli/lib/codex-model-hint.d.ts +9 -0
- package/dist/cli/lib/codex-model-hint.js +45 -0
- package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
- package/dist/cli/lib/generation-failure-hint.js +61 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +4 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +38 -6
- package/dist/cli/lib/i18n-dict/help.js +6 -6
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +13 -9
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +34 -2
- package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
- package/dist/cli/lib/llm-failure-classifier.js +8 -0
- package/dist/cli/lib/parse-run-config.d.ts +6 -5
- package/dist/cli/lib/parse-run-config.js +16 -9
- package/dist/cli/lib/runtime-defaults.d.ts +21 -0
- package/dist/cli/lib/runtime-defaults.js +79 -0
- package/dist/diagnosis/observe-mapper.js +14 -15
- package/dist/diagnosis/observe-producer.js +3 -1
- package/dist/diagnosis/studio-projection.js +14 -7
- package/dist/diagnosis/types.d.ts +2 -0
- package/dist/diagnosis/types.js +12 -0
- package/dist/doctor/endpoint-rule.js +2 -1
- package/dist/eval-core/artifact-file-names.js +18 -1
- package/dist/eval-core/artifact-index.d.ts +7 -11
- package/dist/eval-core/artifact-index.js +139 -80
- package/dist/eval-core/cache.d.ts +12 -3
- package/dist/eval-core/cache.js +89 -29
- package/dist/eval-core/comparability.js +10 -6
- package/dist/eval-core/evaluation-execution.d.ts +2 -1
- package/dist/eval-core/evaluation-execution.js +122 -37
- package/dist/eval-core/evaluation-job.d.ts +4 -1
- package/dist/eval-core/evaluation-job.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +15 -13
- package/dist/eval-core/evaluation-reporting.js +54 -52
- package/dist/eval-core/execution-strategy.d.ts +2 -0
- package/dist/eval-core/execution-strategy.js +11 -9
- package/dist/eval-core/fact-checker.js +15 -7
- package/dist/eval-core/holdout.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +2 -2
- package/dist/eval-core/mock-hook.cjs +23 -6
- package/dist/eval-core/mocks-runtime.js +30 -8
- package/dist/eval-core/report-document.d.ts +12 -0
- package/dist/eval-core/report-document.js +1151 -0
- package/dist/eval-core/report-extensions.d.ts +4 -0
- package/dist/eval-core/report-extensions.js +500 -0
- package/dist/eval-core/report-file-migration.js +7 -2
- package/dist/eval-core/resume-compatibility.d.ts +31 -0
- package/dist/eval-core/resume-compatibility.js +141 -0
- package/dist/eval-core/sample-fingerprint.d.ts +12 -0
- package/dist/eval-core/sample-fingerprint.js +193 -0
- package/dist/eval-core/schema.js +86 -31
- package/dist/eval-core/verdict.d.ts +8 -4
- package/dist/eval-core/verdict.js +24 -10
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
- package/dist/eval-workflows/evaluation-pipeline.js +12 -7
- package/dist/eval-workflows/run-evaluation.d.ts +9 -7
- package/dist/eval-workflows/run-evaluation.js +79 -51
- package/dist/executors/anthropic-api.js +65 -9
- package/dist/executors/claude-cli.js +16 -79
- package/dist/executors/claude-protocol.d.ts +28 -0
- package/dist/executors/claude-protocol.js +180 -0
- package/dist/executors/claude-sdk-trace.js +56 -28
- package/dist/executors/claude-sdk.d.ts +1 -0
- package/dist/executors/claude-sdk.js +39 -93
- package/dist/executors/codex-cli-trace.js +166 -31
- package/dist/executors/codex-cli.d.ts +6 -8
- package/dist/executors/codex-cli.js +49 -151
- package/dist/executors/codex-protocol.d.ts +24 -0
- package/dist/executors/codex-protocol.js +234 -0
- package/dist/executors/codex-sdk.js +68 -120
- package/dist/executors/gemini.js +88 -13
- package/dist/executors/index.d.ts +2 -3
- package/dist/executors/index.js +5 -3
- package/dist/executors/openai-api.js +70 -9
- package/dist/executors/runtime-fingerprint.js +88 -11
- package/dist/executors/script-command.d.ts +8 -0
- package/dist/executors/script-command.js +87 -0
- package/dist/executors/script.js +202 -29
- package/dist/executors/shared.d.ts +35 -3
- package/dist/executors/shared.js +113 -15
- package/dist/grading/assertions.d.ts +1 -1
- package/dist/grading/assertions.js +19 -9
- package/dist/grading/diagnostic.d.ts +9 -2
- package/dist/grading/diagnostic.js +25 -2
- package/dist/grading/index.js +10 -4
- package/dist/grading/judge.js +19 -6
- package/dist/grading/layered-scores.d.ts +2 -3
- package/dist/grading/layered-scores.js +2 -3
- package/dist/inputs/load-samples.d.ts +1 -2
- package/dist/inputs/load-samples.js +23 -1
- package/dist/inputs/mcp-resolver.js +6 -3
- package/dist/inputs/sample-document.d.ts +11 -0
- package/dist/inputs/sample-document.js +96 -0
- package/dist/managed/evidence.d.ts +1 -0
- package/dist/managed/evidence.js +1 -1
- package/dist/managed/store.js +200 -91
- package/dist/observability/codex-trace-adapter.d.ts +5 -0
- package/dist/observability/codex-trace-adapter.js +850 -0
- package/dist/observability/experience.d.ts +32 -6
- package/dist/observability/experience.js +2695 -459
- package/dist/observability/feedback-matchers.js +16 -1
- package/dist/observability/inbox-view-model.d.ts +1 -1
- package/dist/observability/inbox-view-model.js +19 -14
- package/dist/observability/inbox.d.ts +7 -1
- package/dist/observability/inbox.js +632 -124
- package/dist/observability/problem-patterns.js +2 -0
- package/dist/observability/review-state.d.ts +6 -0
- package/dist/observability/review-state.js +235 -63
- package/dist/observability/skill-chain-advisories.js +1 -1
- package/dist/observability/skill-chain.js +17 -4
- package/dist/observability/skill-health-analyzer.d.ts +32 -7
- package/dist/observability/skill-health-analyzer.js +194 -121
- package/dist/observability/skill-health-report.d.ts +10 -0
- package/dist/observability/skill-health-report.js +620 -0
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +1 -1
- package/dist/observability/soft-standards/index.js +1 -1
- package/dist/observability/soft-standards/llm-extractor.js +8 -10
- package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
- package/dist/observability/soft-standards/skill-standards-store.js +59 -18
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/observability/trace-adapter.d.ts +12 -7
- package/dist/observability/trace-adapter.js +11 -9
- package/dist/observability/trace-attribution.d.ts +13 -5
- package/dist/observability/trace-attribution.js +315 -21
- package/dist/observability/trace-ingestion.d.ts +9 -0
- package/dist/observability/trace-ingestion.js +80 -0
- package/dist/observability/trace-ir.d.ts +113 -0
- package/dist/observability/trace-ir.js +87 -0
- package/dist/observability/trace-segmenter.d.ts +19 -6
- package/dist/observability/trace-segmenter.js +377 -196
- package/dist/observability/trace-session-index.d.ts +19 -0
- package/dist/observability/trace-session-index.js +68 -0
- package/dist/observability/trace-source.d.ts +12 -4
- package/dist/observability/trace-source.js +939 -215
- package/dist/renderer/html-renderer.js +37 -6
- package/dist/renderer/icons.js +3 -0
- package/dist/renderer/observation-inbox-renderer.js +208 -90
- package/dist/renderer/skill-detail-renderer.js +452 -109
- package/dist/renderer/skill-health-renderer.js +69 -12
- package/dist/renderer/summary.js +28 -7
- package/dist/renderer/table.js +21 -4
- package/dist/renderer/test-view.d.ts +1 -0
- package/dist/renderer/test-view.js +44 -9
- package/dist/server/indexed-report-store.js +14 -18
- package/dist/server/job-store.js +64 -26
- package/dist/server/report-server.js +190 -78
- package/dist/server/report-store.js +57 -80
- package/dist/server/skill-index.js +143 -49
- package/dist/server/skill-insights.js +44 -5
- package/dist/shared/artifact-graph.d.ts +3 -0
- package/dist/shared/artifact-graph.js +224 -0
- package/dist/shared/assertion-types.d.ts +8 -0
- package/dist/shared/assertion-types.js +46 -0
- package/dist/shared/atomic-json.d.ts +8 -0
- package/dist/shared/atomic-json.js +33 -0
- package/dist/shared/diagnosis-schema.d.ts +9 -0
- package/dist/shared/diagnosis-schema.js +181 -0
- package/dist/shared/doctor-report.d.ts +3 -0
- package/dist/shared/doctor-report.js +103 -0
- package/dist/shared/evaluation-job.d.ts +6 -0
- package/dist/shared/evaluation-job.js +217 -0
- package/dist/shared/executor-result.d.ts +17 -0
- package/dist/shared/executor-result.js +221 -0
- package/dist/shared/file-lock.d.ts +12 -0
- package/dist/shared/file-lock.js +129 -0
- package/dist/shared/json-value.d.ts +5 -0
- package/dist/shared/json-value.js +36 -0
- package/dist/shared/keyed-mutex.d.ts +7 -0
- package/dist/shared/keyed-mutex.js +24 -0
- package/dist/shared/record-count.d.ts +8 -0
- package/dist/shared/record-count.js +43 -0
- package/dist/shared/sample-contract.d.ts +3 -0
- package/dist/shared/sample-contract.js +332 -0
- package/dist/shared/timestamp.d.ts +6 -0
- package/dist/shared/timestamp.js +64 -0
- package/dist/shared/token-usage.d.ts +19 -0
- package/dist/shared/token-usage.js +50 -0
- package/dist/shared/tool-call-status.d.ts +8 -0
- package/dist/shared/tool-call-status.js +28 -0
- package/dist/shared/tool-identity.d.ts +21 -0
- package/dist/shared/tool-identity.js +84 -0
- package/dist/shared/tool-search.js +73 -16
- package/dist/shared/trace-projection.d.ts +5 -0
- package/dist/shared/trace-projection.js +20 -0
- package/dist/shared/trace-source-kind.d.ts +3 -0
- package/dist/shared/trace-source-kind.js +12 -0
- package/dist/types/diagnosis.d.ts +2 -0
- package/dist/types/eval.d.ts +4 -0
- package/dist/types/executor.d.ts +32 -5
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/judge.d.ts +2 -0
- package/dist/types/observability.d.ts +116 -9
- package/dist/types/report.d.ts +58 -6
- package/dist/types/skill-index.d.ts +7 -0
- package/dist/types/trace.d.ts +2 -0
- package/dist/types/trace.js +1 -0
- package/package.json +9 -5
|
@@ -3,32 +3,18 @@
|
|
|
3
3
|
* Default implementation: local file system.
|
|
4
4
|
* Can be replaced with database, S3, etc.
|
|
5
5
|
*/
|
|
6
|
-
import { readdir, readFile,
|
|
6
|
+
import { readdir, readFile, unlink, access, mkdir, stat } from 'node:fs/promises';
|
|
7
7
|
import { join } from 'node:path';
|
|
8
|
-
import { isReportFileName, reportFilePath, reportFileStem } from '../eval-core/artifact-file-names.js';
|
|
8
|
+
import { isReportFileName, reportFileName, reportFilePath, reportFileStem, safeArtifactFileStem, } from '../eval-core/artifact-file-names.js';
|
|
9
9
|
import { migrateLegacyReportFiles } from '../eval-core/report-file-migration.js';
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
async function withLock(id, fn) {
|
|
14
|
-
// Chain onto any existing lock for this id, so requests are serialized
|
|
15
|
-
const prev = locks.get(id) ?? Promise.resolve();
|
|
16
|
-
let releaseLock;
|
|
17
|
-
const next = new Promise((r) => { releaseLock = r; });
|
|
18
|
-
locks.set(id, next);
|
|
19
|
-
await prev;
|
|
20
|
-
try {
|
|
21
|
-
return await fn();
|
|
22
|
-
}
|
|
23
|
-
finally {
|
|
24
|
-
locks.delete(id);
|
|
25
|
-
releaseLock();
|
|
26
|
-
}
|
|
27
|
-
}
|
|
10
|
+
import { parseReportDocument } from '../eval-core/report-document.js';
|
|
11
|
+
import { writeJsonFileAtomicAsync } from '../shared/atomic-json.js';
|
|
12
|
+
import { KeyedMutex } from '../shared/keyed-mutex.js';
|
|
28
13
|
/**
|
|
29
14
|
* Create a file-system-based report store.
|
|
30
15
|
*/
|
|
31
16
|
export function createFileStore(dir) {
|
|
17
|
+
const mutations = new KeyedMutex();
|
|
32
18
|
async function ensureDir() {
|
|
33
19
|
try {
|
|
34
20
|
await access(dir);
|
|
@@ -37,35 +23,6 @@ export function createFileStore(dir) {
|
|
|
37
23
|
await mkdir(dir, { recursive: true });
|
|
38
24
|
}
|
|
39
25
|
}
|
|
40
|
-
function normalizeReportDocument(data, fallbackId) {
|
|
41
|
-
if (!data || typeof data !== 'object')
|
|
42
|
-
return null;
|
|
43
|
-
const record = data;
|
|
44
|
-
const kind = record.kind === 'evaluation' || record.kind === 'batch-evaluation'
|
|
45
|
-
? record.kind
|
|
46
|
-
: undefined;
|
|
47
|
-
if (kind === 'evaluation') {
|
|
48
|
-
if (!record.meta || !record.summary || !Array.isArray(record.results))
|
|
49
|
-
return null;
|
|
50
|
-
return {
|
|
51
|
-
...record,
|
|
52
|
-
kind,
|
|
53
|
-
id: typeof record.id === 'string' && record.id ? record.id : fallbackId,
|
|
54
|
-
};
|
|
55
|
-
}
|
|
56
|
-
if (kind === 'batch-evaluation') {
|
|
57
|
-
if (!record.meta || !Array.isArray(record.items))
|
|
58
|
-
return null;
|
|
59
|
-
return {
|
|
60
|
-
...record,
|
|
61
|
-
kind,
|
|
62
|
-
id: typeof record.id === 'string' && record.id ? record.id : fallbackId,
|
|
63
|
-
};
|
|
64
|
-
}
|
|
65
|
-
// 只认 canonical 顶层 `kind`(evaluation / batch-evaluation)。不再为旧格式(顶层无该判别字段的
|
|
66
|
-
// 历史文件)做读兼容 —— 顶层 kind cutover 是硬切换,旧文件直接判脏丢弃。
|
|
67
|
-
return null;
|
|
68
|
-
}
|
|
69
26
|
function isEvaluationReport(report) {
|
|
70
27
|
return report.kind === 'evaluation';
|
|
71
28
|
}
|
|
@@ -75,6 +32,13 @@ export function createFileStore(dir) {
|
|
|
75
32
|
// 完全跳过 readFile。
|
|
76
33
|
let cachedFingerprint = '';
|
|
77
34
|
let cachedRuns = null;
|
|
35
|
+
function invalidateListCache() {
|
|
36
|
+
cachedFingerprint = '';
|
|
37
|
+
cachedRuns = null;
|
|
38
|
+
}
|
|
39
|
+
function cloneReports(reports) {
|
|
40
|
+
return structuredClone(reports);
|
|
41
|
+
}
|
|
78
42
|
async function computeListFingerprint() {
|
|
79
43
|
try {
|
|
80
44
|
const dirStat = await stat(dir);
|
|
@@ -104,7 +68,7 @@ export function createFileStore(dir) {
|
|
|
104
68
|
migrateLegacyReportFiles(dir, 'report');
|
|
105
69
|
const fp = await computeListFingerprint();
|
|
106
70
|
if (fp != null && fp === cachedFingerprint && cachedRuns)
|
|
107
|
-
return cachedRuns;
|
|
71
|
+
return cloneReports(cachedRuns);
|
|
108
72
|
const files = (await readdir(dir))
|
|
109
73
|
.filter(isReportFileName)
|
|
110
74
|
.sort()
|
|
@@ -113,8 +77,8 @@ export function createFileStore(dir) {
|
|
|
113
77
|
for (const file of files) {
|
|
114
78
|
try {
|
|
115
79
|
const data = JSON.parse(await readFile(join(dir, file), 'utf-8'));
|
|
116
|
-
const report =
|
|
117
|
-
if (report)
|
|
80
|
+
const report = parseReportDocument(data, reportFileStem(file) ?? file);
|
|
81
|
+
if (report && reportFileName(report.id) === file)
|
|
118
82
|
runs.push(report);
|
|
119
83
|
}
|
|
120
84
|
catch { /* skip corrupt files */ }
|
|
@@ -127,63 +91,71 @@ export function createFileStore(dir) {
|
|
|
127
91
|
if (fp != null) {
|
|
128
92
|
cachedFingerprint = fp;
|
|
129
93
|
cachedRuns = runs;
|
|
94
|
+
return cloneReports(runs);
|
|
130
95
|
}
|
|
131
96
|
return runs;
|
|
132
97
|
}
|
|
133
98
|
async function get(id) {
|
|
99
|
+
if (!id || safeArtifactFileStem(id) !== id)
|
|
100
|
+
return null;
|
|
134
101
|
migrateLegacyReportFiles(dir, 'report');
|
|
135
102
|
try {
|
|
136
103
|
const data = JSON.parse(await readFile(reportFilePath(dir, id), 'utf-8'));
|
|
137
|
-
return
|
|
104
|
+
return parseReportDocument(data, id, id);
|
|
138
105
|
}
|
|
139
106
|
catch {
|
|
140
107
|
return null;
|
|
141
108
|
}
|
|
142
109
|
}
|
|
143
|
-
async function
|
|
110
|
+
async function saveUnlocked(id, report) {
|
|
111
|
+
if (!id
|
|
112
|
+
|| safeArtifactFileStem(id) !== id
|
|
113
|
+
|| report.id !== id
|
|
114
|
+
|| !parseReportDocument(report, id, id)) {
|
|
115
|
+
throw new Error('invalid report');
|
|
116
|
+
}
|
|
144
117
|
await ensureDir();
|
|
145
118
|
migrateLegacyReportFiles(dir, 'report');
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
119
|
+
await writeJsonFileAtomicAsync(reportFilePath(dir, id), report);
|
|
120
|
+
invalidateListCache();
|
|
121
|
+
}
|
|
122
|
+
async function save(id, report) {
|
|
123
|
+
await mutations.run(id, () => saveUnlocked(id, report));
|
|
150
124
|
}
|
|
151
125
|
/**
|
|
152
126
|
* Atomic read-modify-write with in-memory mutex.
|
|
153
127
|
* Prevents concurrent updates from overwriting each other.
|
|
154
128
|
*/
|
|
155
129
|
async function update(id, mutator) {
|
|
156
|
-
return
|
|
130
|
+
return mutations.run(id, async () => {
|
|
157
131
|
const report = await get(id);
|
|
158
132
|
if (!report)
|
|
159
133
|
return null;
|
|
160
134
|
mutator(report);
|
|
161
|
-
await
|
|
135
|
+
await saveUnlocked(id, report);
|
|
162
136
|
return report;
|
|
163
137
|
});
|
|
164
138
|
}
|
|
165
139
|
async function remove(id) {
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
140
|
+
if (!id || safeArtifactFileStem(id) !== id)
|
|
141
|
+
return false;
|
|
142
|
+
return mutations.run(id, async () => {
|
|
143
|
+
migrateLegacyReportFiles(dir, 'report');
|
|
144
|
+
try {
|
|
145
|
+
await unlink(reportFilePath(dir, id));
|
|
146
|
+
invalidateListCache();
|
|
147
|
+
return true;
|
|
148
|
+
}
|
|
149
|
+
catch (err) {
|
|
150
|
+
const fsError = err;
|
|
151
|
+
if (fsError.code === 'ENOENT')
|
|
152
|
+
return false;
|
|
153
|
+
throw err;
|
|
154
|
+
}
|
|
155
|
+
});
|
|
177
156
|
}
|
|
178
157
|
async function exists(id) {
|
|
179
|
-
|
|
180
|
-
try {
|
|
181
|
-
await access(reportFilePath(dir, id));
|
|
182
|
-
return true;
|
|
183
|
-
}
|
|
184
|
-
catch {
|
|
185
|
-
return false;
|
|
186
|
-
}
|
|
158
|
+
return (await get(id)) !== null;
|
|
187
159
|
}
|
|
188
160
|
async function findByVariant(variantName) {
|
|
189
161
|
const all = await list();
|
|
@@ -231,7 +203,12 @@ export function createOverlayReportStore(projectDir, globalDir) {
|
|
|
231
203
|
return (await project.exists(id)) ? project.update(id, mutator) : global.update(id, mutator);
|
|
232
204
|
}
|
|
233
205
|
async function remove(id) {
|
|
234
|
-
|
|
206
|
+
// Migration windows can leave the same id in both roots. Always attempt both:
|
|
207
|
+
// short-circuiting after the project copy would make get() fall back to the
|
|
208
|
+
// surviving global copy and appear to resurrect a successfully deleted report.
|
|
209
|
+
const projectRemoved = await project.remove(id);
|
|
210
|
+
const globalRemoved = await global.remove(id);
|
|
211
|
+
return projectRemoved || globalRemoved;
|
|
235
212
|
}
|
|
236
213
|
async function findByVariant(variantName) {
|
|
237
214
|
return (await projectHasReports()) ? project.findByVariant(variantName) : global.findByVariant(variantName);
|
|
@@ -19,12 +19,16 @@ import { GRAPH_FILE_SUFFIX, graphFileName, isReportFileName, reportFileStem } fr
|
|
|
19
19
|
import { migrateLegacyReportFiles } from '../eval-core/report-file-migration.js';
|
|
20
20
|
import { doctorGraphDirForDoctorOutput } from '../artifact-graph/doctor.js';
|
|
21
21
|
import { evalGraphDirForReportOutput } from '../artifact-graph/eval.js';
|
|
22
|
-
import {
|
|
22
|
+
import { parseSkillHealthReport } from '../observability/skill-health-report.js';
|
|
23
|
+
import { confidenceOf, toolStabilityOf } from '../observability/skill-health-analyzer.js';
|
|
23
24
|
import { computeVerdict } from '../eval-core/verdict.js';
|
|
24
|
-
import { artifactIndexDir, listLiveDoctorCards,
|
|
25
|
+
import { artifactIndexDir, listLiveDoctorCards, listLiveObserveCards, listLiveReportCards, cardTargetSentinel } from '../eval-core/artifact-index.js';
|
|
25
26
|
import { detectInsights } from './skill-insights.js';
|
|
26
27
|
import { DEFAULT_OBSERVATIONS_DIR, loadLatestObservationInboxReports } from '../observability/inbox.js';
|
|
27
28
|
import { buildStudioDiagnosisSummary, mergeDiagnosisBundles } from '../diagnosis/studio-projection.js';
|
|
29
|
+
import { parseDoctorReport } from '../shared/doctor-report.js';
|
|
30
|
+
import { parseArtifactGraphDocument } from '../shared/artifact-graph.js';
|
|
31
|
+
import { ownRecordValue } from '../shared/record-count.js';
|
|
28
32
|
let _indexCache = null;
|
|
29
33
|
/**
|
|
30
34
|
* Sync 版 dir-content fingerprint helper,仿 `src/server/report-store.ts:80-92`
|
|
@@ -235,16 +239,39 @@ const PER_SKILL_BAND_RED_FAILURE_RATE = 0.4; // 工具失败率 ≥ 40% → red
|
|
|
235
239
|
const PER_SKILL_BAND_YELLOW_FAILURE_RATE = 0.2; // ≥ 20% → yellow
|
|
236
240
|
const PER_SKILL_BAND_YELLOW_GAP_RATE = 0.3; // 加权 gap ≥ 30% → yellow
|
|
237
241
|
function bandFromObserveHealth(h) {
|
|
238
|
-
|
|
242
|
+
const resolvedOutcomeCount = h.toolResolvedCount ?? h.toolCallCount;
|
|
243
|
+
const comparableOutcomeCount = resolvedOutcomeCount === undefined
|
|
244
|
+
? undefined
|
|
245
|
+
: Math.max(0, resolvedOutcomeCount - (h.toolCancelledCount ?? 0));
|
|
246
|
+
const failureRateMeasured = comparableOutcomeCount === undefined || comparableOutcomeCount >= 5;
|
|
247
|
+
if (h.toolFailureRate >= PER_SKILL_BAND_RED_FAILURE_RATE
|
|
248
|
+
&& failureRateMeasured)
|
|
239
249
|
return 'red';
|
|
240
250
|
const gap = h.gap?.weightedGapRate ?? 0;
|
|
241
|
-
if (gap >= PER_SKILL_BAND_YELLOW_GAP_RATE
|
|
251
|
+
if (gap >= PER_SKILL_BAND_YELLOW_GAP_RATE
|
|
252
|
+
|| (h.toolFailureRate >= PER_SKILL_BAND_YELLOW_FAILURE_RATE
|
|
253
|
+
&& failureRateMeasured))
|
|
242
254
|
return 'yellow';
|
|
243
255
|
return 'green';
|
|
244
256
|
}
|
|
245
|
-
function
|
|
246
|
-
if (
|
|
257
|
+
function normalizedObserveStability(h) {
|
|
258
|
+
if (h.toolCallCount === undefined)
|
|
259
|
+
return h.stability;
|
|
260
|
+
const resolved = h.toolResolvedCount ?? h.toolCallCount;
|
|
261
|
+
const comparable = Math.max(0, resolved - (h.toolCancelledCount ?? 0));
|
|
262
|
+
return toolStabilityOf(h.toolFailureRate, comparable, h.toolCallCount);
|
|
263
|
+
}
|
|
264
|
+
function effectiveObserveSnapshotBand(observe) {
|
|
265
|
+
if (!observe || observe.confidence === 'underpowered')
|
|
266
|
+
return 'gray';
|
|
267
|
+
if (observe.healthBand === 'green'
|
|
268
|
+
&& (observe.toolCallCount ?? 0) > 0
|
|
269
|
+
&& Math.max(0, (observe.toolResolvedCount ?? observe.toolCallCount ?? 0)
|
|
270
|
+
- (observe.toolCancelledCount ?? 0)) < 5)
|
|
247
271
|
return 'gray';
|
|
272
|
+
return observe.healthBand;
|
|
273
|
+
}
|
|
274
|
+
function combineBand(doctor, evalSnap, observe) {
|
|
248
275
|
// doctor band: status fail → red, warn → yellow, pass → green
|
|
249
276
|
const doctorBand = !doctor
|
|
250
277
|
? 'gray'
|
|
@@ -258,11 +285,13 @@ function combineBand(doctor, evalSnap, _observe) {
|
|
|
258
285
|
: evalScore < 2.5 ? 'red'
|
|
259
286
|
: evalScore < 3.5 ? 'yellow'
|
|
260
287
|
: 'green';
|
|
261
|
-
|
|
288
|
+
// underpowered observe 的色带仅供参考,不参与聚合硬结论。
|
|
289
|
+
const observeBand = effectiveObserveSnapshotBand(observe);
|
|
290
|
+
if (doctorBand === 'red' || evalBand === 'red' || observeBand === 'red')
|
|
262
291
|
return 'red';
|
|
263
|
-
if (doctorBand === 'yellow' || evalBand === 'yellow')
|
|
292
|
+
if (doctorBand === 'yellow' || evalBand === 'yellow' || observeBand === 'yellow')
|
|
264
293
|
return 'yellow';
|
|
265
|
-
if (doctorBand === 'green' || evalBand === 'green')
|
|
294
|
+
if (doctorBand === 'green' || evalBand === 'green' || observeBand === 'green')
|
|
266
295
|
return 'green';
|
|
267
296
|
return 'gray';
|
|
268
297
|
}
|
|
@@ -281,24 +310,11 @@ function latestEvalSnapshot(list) {
|
|
|
281
310
|
return null;
|
|
282
311
|
return list[list.length - 1];
|
|
283
312
|
}
|
|
284
|
-
function isArtifactGraphDocument(value) {
|
|
285
|
-
if (!value || typeof value !== 'object')
|
|
286
|
-
return false;
|
|
287
|
-
const graph = value;
|
|
288
|
-
return graph.documentKind === 'artifact-graph'
|
|
289
|
-
&& graph.schemaVersion === 1
|
|
290
|
-
&& typeof graph.graphId === 'string'
|
|
291
|
-
&& !!graph.source
|
|
292
|
-
&& (graph.source.sourceKind === 'doctor' || graph.source.sourceKind === 'eval' || graph.source.sourceKind === 'observe')
|
|
293
|
-
&& Array.isArray(graph.nodes)
|
|
294
|
-
&& Array.isArray(graph.edges);
|
|
295
|
-
}
|
|
296
313
|
function readArtifactGraph(path) {
|
|
297
314
|
try {
|
|
298
315
|
const parsed = JSON.parse(readFileSync(path, 'utf-8'));
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
return { graph: parsed, path };
|
|
316
|
+
const graph = parseArtifactGraphDocument(parsed);
|
|
317
|
+
return graph ? { graph, path } : null;
|
|
302
318
|
}
|
|
303
319
|
catch {
|
|
304
320
|
return null;
|
|
@@ -477,6 +493,7 @@ function projectEvalStage(graph, path, entry) {
|
|
|
477
493
|
const projectedEdgeIds = new Set();
|
|
478
494
|
const sampleStatusByNodeId = new Map();
|
|
479
495
|
const assertionStatusByNodeId = new Map();
|
|
496
|
+
const parentSampleStableKeyByNodeId = new Map();
|
|
480
497
|
if (skillNode)
|
|
481
498
|
projectedNodeIds.add(skillNode.id);
|
|
482
499
|
const addEdge = (edge) => {
|
|
@@ -496,15 +513,45 @@ function projectEvalStage(graph, path, entry) {
|
|
|
496
513
|
.filter((edge) => edge.toNodeId === variantNode.id && edge.edgeKind === 'derived_from')
|
|
497
514
|
.map((edge) => nodesById.get(edge.fromNodeId))
|
|
498
515
|
.filter((node) => node?.nodeKind === 'eval_result');
|
|
516
|
+
const evalResultNodeIds = new Set(evalResultNodes.map((node) => node.id));
|
|
499
517
|
for (const node of evalResultNodes)
|
|
500
518
|
projectedNodeIds.add(node.id);
|
|
501
519
|
for (const edge of graph.edges) {
|
|
502
|
-
if (!
|
|
520
|
+
if (!evalResultNodeIds.has(edge.fromNodeId) && !evalResultNodeIds.has(edge.toNodeId))
|
|
503
521
|
continue;
|
|
504
522
|
if (edge.edgeKind === 'derived_from' || edge.edgeKind === 'evaluates' || edge.edgeKind === 'passes' || edge.edgeKind === 'fails' || edge.edgeKind === 'diagnoses') {
|
|
505
523
|
addEdge(edge);
|
|
506
524
|
}
|
|
507
525
|
}
|
|
526
|
+
for (const edge of graph.edges) {
|
|
527
|
+
if (edge.edgeKind !== 'contains')
|
|
528
|
+
continue;
|
|
529
|
+
const from = nodesById.get(edge.fromNodeId);
|
|
530
|
+
const to = nodesById.get(edge.toNodeId);
|
|
531
|
+
if (from?.nodeKind !== 'sample' || to?.nodeKind !== 'assertion' || !projectedNodeIds.has(from.id))
|
|
532
|
+
continue;
|
|
533
|
+
addEdge(edge);
|
|
534
|
+
if (from.stableKey)
|
|
535
|
+
parentSampleStableKeyByNodeId.set(to.id, from.stableKey);
|
|
536
|
+
}
|
|
537
|
+
const sampleStableKeyByEvalResultId = new Map();
|
|
538
|
+
for (const edge of graph.edges) {
|
|
539
|
+
if (edge.edgeKind !== 'evaluates' || !evalResultNodeIds.has(edge.fromNodeId))
|
|
540
|
+
continue;
|
|
541
|
+
const sample = nodesById.get(edge.toNodeId);
|
|
542
|
+
if (sample?.nodeKind === 'sample' && sample.stableKey) {
|
|
543
|
+
sampleStableKeyByEvalResultId.set(edge.fromNodeId, sample.stableKey);
|
|
544
|
+
}
|
|
545
|
+
}
|
|
546
|
+
for (const edge of graph.edges) {
|
|
547
|
+
if (edge.edgeKind !== 'diagnoses')
|
|
548
|
+
continue;
|
|
549
|
+
const diagnostic = nodesById.get(edge.fromNodeId);
|
|
550
|
+
const sampleStableKey = sampleStableKeyByEvalResultId.get(edge.toNodeId);
|
|
551
|
+
if (diagnostic?.nodeKind === 'diagnostic' && sampleStableKey) {
|
|
552
|
+
parentSampleStableKeyByNodeId.set(diagnostic.id, sampleStableKey);
|
|
553
|
+
}
|
|
554
|
+
}
|
|
508
555
|
const declaredCoverageStableKeys = new Set();
|
|
509
556
|
const declaredCoverageEdges = [];
|
|
510
557
|
let coverageEdges = 0;
|
|
@@ -581,7 +628,17 @@ function projectEvalStage(graph, path, entry) {
|
|
|
581
628
|
: sourceNode.nodeKind === 'assertion'
|
|
582
629
|
? assertionStatusByNodeId.get(sourceNode.id)
|
|
583
630
|
: undefined;
|
|
584
|
-
|
|
631
|
+
const parentSampleStableKey = parentSampleStableKeyByNodeId.get(sourceNode.id)
|
|
632
|
+
?? (sourceNode.nodeKind === 'assertion'
|
|
633
|
+
? graph.nodes.find((candidate) => candidate.nodeKind === 'sample'
|
|
634
|
+
&& candidate.stableKey
|
|
635
|
+
&& sourceNode.stableKey?.startsWith(`${candidate.stableKey}:assertion:`))?.stableKey
|
|
636
|
+
: undefined);
|
|
637
|
+
return {
|
|
638
|
+
...node,
|
|
639
|
+
...(status ? { status } : {}),
|
|
640
|
+
...(parentSampleStableKey ? { parentSampleStableKey } : {}),
|
|
641
|
+
};
|
|
585
642
|
});
|
|
586
643
|
return {
|
|
587
644
|
stage: {
|
|
@@ -648,7 +705,7 @@ function graphSnapshotForEntry(entry, graphs) {
|
|
|
648
705
|
/** 扫 doctorsDir/*.report.json,按 skill 名分桶,**返回该 skill 的所有历史 snapshot**(asc 时序)。
|
|
649
706
|
* renderer 用最后一项做"当前",前面项画 sparkline。 */
|
|
650
707
|
function scanDoctorReports(dir) {
|
|
651
|
-
const out =
|
|
708
|
+
const out = Object.create(null);
|
|
652
709
|
migrateLegacyReportFiles(dir, 'doctor');
|
|
653
710
|
if (!existsSync(dir))
|
|
654
711
|
return out;
|
|
@@ -656,19 +713,13 @@ function scanDoctorReports(dir) {
|
|
|
656
713
|
if (!isReportFileName(file))
|
|
657
714
|
continue;
|
|
658
715
|
try {
|
|
659
|
-
const data = JSON.parse(readFileSync(join(dir, file), 'utf-8'));
|
|
660
|
-
|
|
661
|
-
if (!kind || !Array.isArray(data.skills))
|
|
716
|
+
const data = parseDoctorReport(JSON.parse(readFileSync(join(dir, file), 'utf-8')));
|
|
717
|
+
if (!data)
|
|
662
718
|
continue;
|
|
663
|
-
const ts = data.timestamp;
|
|
664
719
|
for (const sr of data.skills) {
|
|
665
|
-
const
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
const snap = {
|
|
669
|
-
reportId: data.id, timestamp: ts, status: sr.status,
|
|
670
|
-
passCount: passN, warnCount: warnN, failCount: failN, results: sr.results,
|
|
671
|
-
};
|
|
720
|
+
const snap = doctorSnapshot(data, sr.skillName);
|
|
721
|
+
if (!snap)
|
|
722
|
+
continue;
|
|
672
723
|
if (!out[sr.skillName])
|
|
673
724
|
out[sr.skillName] = [];
|
|
674
725
|
out[sr.skillName].push(snap);
|
|
@@ -680,6 +731,20 @@ function scanDoctorReports(dir) {
|
|
|
680
731
|
list.sort((a, b) => a.timestamp.localeCompare(b.timestamp));
|
|
681
732
|
return out;
|
|
682
733
|
}
|
|
734
|
+
function doctorSnapshot(report, skillName) {
|
|
735
|
+
const skill = report.skills.find((entry) => entry.skillName === skillName);
|
|
736
|
+
if (!skill)
|
|
737
|
+
return null;
|
|
738
|
+
return {
|
|
739
|
+
reportId: report.id,
|
|
740
|
+
timestamp: report.timestamp,
|
|
741
|
+
status: skill.status,
|
|
742
|
+
passCount: skill.results.filter((result) => result.status === 'pass').length,
|
|
743
|
+
warnCount: skill.results.filter((result) => result.status === 'warn').length,
|
|
744
|
+
failCount: skill.results.filter((result) => result.status === 'fail').length,
|
|
745
|
+
results: skill.results,
|
|
746
|
+
};
|
|
747
|
+
}
|
|
683
748
|
export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDir = DEFAULT_OBSERVATIONS_DIR, opts = {}) {
|
|
684
749
|
const includeObserveCards = opts.includeObserveCards ?? false;
|
|
685
750
|
const includeDoctorCards = opts.includeDoctorCards ?? false;
|
|
@@ -696,7 +761,7 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
|
|
|
696
761
|
return _indexCache.result;
|
|
697
762
|
}
|
|
698
763
|
// ── eval 聚合(历史 list)─────────────────────────────────
|
|
699
|
-
const evalBy =
|
|
764
|
+
const evalBy = Object.create(null);
|
|
700
765
|
for (const r of reports) {
|
|
701
766
|
if (r.kind !== 'evaluation')
|
|
702
767
|
continue;
|
|
@@ -716,7 +781,7 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
|
|
|
716
781
|
for (const list of Object.values(evalBy))
|
|
717
782
|
list.sort((a, b) => evalSnapshotSortKey(a).localeCompare(evalSnapshotSortKey(b)));
|
|
718
783
|
// ── observe 聚合(历史 list)──────────────────────────────
|
|
719
|
-
const observeBy =
|
|
784
|
+
const observeBy = Object.create(null);
|
|
720
785
|
migrateLegacyReportFiles(analysesDir, 'observe-health');
|
|
721
786
|
if (existsSync(analysesDir)) {
|
|
722
787
|
for (const file of readdirSync(analysesDir)) {
|
|
@@ -724,8 +789,8 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
|
|
|
724
789
|
if (!id)
|
|
725
790
|
continue;
|
|
726
791
|
try {
|
|
727
|
-
const data = JSON.parse(readFileSync(join(analysesDir, file), 'utf-8'));
|
|
728
|
-
if (!data
|
|
792
|
+
const data = parseSkillHealthReport(JSON.parse(readFileSync(join(analysesDir, file), 'utf-8')));
|
|
793
|
+
if (!data)
|
|
729
794
|
continue;
|
|
730
795
|
const generatedAt = data.meta.generatedAt;
|
|
731
796
|
for (const [skill, h] of Object.entries(data.bySkill)) {
|
|
@@ -733,8 +798,13 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
|
|
|
733
798
|
analysisId: id, generatedAt,
|
|
734
799
|
healthBand: bandFromObserveHealth(h),
|
|
735
800
|
failureRate: h.toolFailureRate,
|
|
801
|
+
toolCallCount: h.toolCallCount,
|
|
802
|
+
toolResolvedCount: h.toolResolvedCount,
|
|
803
|
+
toolCancelledCount: h.toolCancelledCount,
|
|
804
|
+
toolUnknownCount: h.toolUnknownCount,
|
|
736
805
|
segmentCount: h.segmentCount,
|
|
737
806
|
gapRate: h.gap?.weightedGapRate ?? 0,
|
|
807
|
+
stability: normalizedObserveStability(h),
|
|
738
808
|
// Legacy reports (pre-confidence) fall back to deriving from segmentCount.
|
|
739
809
|
confidence: h.confidence ?? confidenceOf(h.segmentCount),
|
|
740
810
|
};
|
|
@@ -750,14 +820,27 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
|
|
|
750
820
|
// 仅机器级模式合并;固定 --analyses-dir / --global 时 includeObserveCards=false,只看该目录(逃生舱语义)。
|
|
751
821
|
if (includeObserveCards) {
|
|
752
822
|
for (const card of listLiveObserveCards()) {
|
|
753
|
-
|
|
823
|
+
let report = null;
|
|
824
|
+
try {
|
|
825
|
+
report = parseSkillHealthReport(JSON.parse(readFileSync(card.path, 'utf-8')));
|
|
826
|
+
}
|
|
827
|
+
catch {
|
|
828
|
+
// Scratch cards only discover canonical reports.
|
|
829
|
+
}
|
|
830
|
+
if (!report)
|
|
831
|
+
continue;
|
|
832
|
+
for (const [skill, h] of Object.entries(report.bySkill)) {
|
|
754
833
|
const list = (observeBy[skill] ??= []);
|
|
755
834
|
if (list.some((s) => s.analysisId === card.id))
|
|
756
835
|
continue;
|
|
757
836
|
list.push({
|
|
758
|
-
analysisId: card.id, generatedAt:
|
|
759
|
-
healthBand: bandFromObserveHealth(h), failureRate: h.toolFailureRate,
|
|
760
|
-
|
|
837
|
+
analysisId: card.id, generatedAt: report.meta.generatedAt,
|
|
838
|
+
healthBand: bandFromObserveHealth(h), failureRate: h.toolFailureRate,
|
|
839
|
+
toolCallCount: h.toolCallCount, toolResolvedCount: h.toolResolvedCount,
|
|
840
|
+
toolCancelledCount: h.toolCancelledCount,
|
|
841
|
+
toolUnknownCount: h.toolUnknownCount, segmentCount: h.segmentCount,
|
|
842
|
+
gapRate: h.gap?.weightedGapRate ?? 0, stability: normalizedObserveStability(h),
|
|
843
|
+
confidence: confidenceOf(h.segmentCount),
|
|
761
844
|
});
|
|
762
845
|
}
|
|
763
846
|
}
|
|
@@ -770,8 +853,19 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
|
|
|
770
853
|
// 仅机器级模式合并;固定 --doctors-dir / --global 时 includeDoctorCards=false,只看该目录(逃生舱语义)。
|
|
771
854
|
if (includeDoctorCards) {
|
|
772
855
|
for (const card of listLiveDoctorCards()) {
|
|
773
|
-
|
|
774
|
-
|
|
856
|
+
let report = null;
|
|
857
|
+
try {
|
|
858
|
+
report = parseDoctorReport(JSON.parse(readFileSync(card.path, 'utf-8')));
|
|
859
|
+
}
|
|
860
|
+
catch {
|
|
861
|
+
// Scratch cards only discover canonical reports.
|
|
862
|
+
}
|
|
863
|
+
if (!report || report.id !== card.reportId)
|
|
864
|
+
continue;
|
|
865
|
+
const snap = doctorSnapshot(report, card.skillName);
|
|
866
|
+
if (!snap)
|
|
867
|
+
continue;
|
|
868
|
+
const list = (doctorBy[card.skillName] ??= []);
|
|
775
869
|
if (!list.some((s) => s.reportId === snap.reportId))
|
|
776
870
|
list.push(snap);
|
|
777
871
|
}
|
|
@@ -816,7 +910,7 @@ export function buildSkillIndex(reports, analysesDir, doctorsDir, observationsDi
|
|
|
816
910
|
for (const ent of entries) {
|
|
817
911
|
const evalReport = ent.eval ? reports.find((r) => r.id === ent.eval.reportId && r.kind === 'evaluation') : undefined;
|
|
818
912
|
insightsBySkill.set(ent.skillName, detectInsights(ent, evalReport ?? null, {
|
|
819
|
-
diagnostics: diagnosisBundle.bySkill
|
|
913
|
+
diagnostics: ownRecordValue(diagnosisBundle.bySkill, ent.skillName) ?? [],
|
|
820
914
|
}));
|
|
821
915
|
}
|
|
822
916
|
// 跨层口径统一:三大 snapshot(doctor / eval / observe)都空但 Diagnosis / Insight 投影出
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { isActiveDiagnosisLifecycle } from '../diagnosis/types.js';
|
|
2
|
+
import { toolCallStatus } from '../shared/tool-call-status.js';
|
|
2
3
|
/** Resolve which variant in `report` 是这个 entry / insight 关注的。
|
|
3
4
|
* 优先用调用方传的 `preferred`(entry.eval?.variantName);找不到再 fallback 到第一个非 baseline。
|
|
4
5
|
* 这避免 multi-treatment 报告里所有 detector 都默认拿第一个 treatment 的数据,导致
|
|
@@ -23,7 +24,11 @@ function summarizeToolCall(tc) {
|
|
|
23
24
|
else if (typeof inp.query === 'string')
|
|
24
25
|
arg = inp.query.slice(0, 50);
|
|
25
26
|
}
|
|
26
|
-
const
|
|
27
|
+
const resultStatus = toolCallStatus(tc);
|
|
28
|
+
const status = resultStatus === 'failure'
|
|
29
|
+
? ' [失败]'
|
|
30
|
+
: resultStatus === 'cancelled' ? ' [取消]'
|
|
31
|
+
: resultStatus === 'unknown' ? ' [状态未知]' : '';
|
|
27
32
|
return arg ? `${tool}: ${arg}${status}` : `${tool}${status}`;
|
|
28
33
|
}
|
|
29
34
|
function getFailedSamples(report, variant) {
|
|
@@ -91,6 +96,32 @@ function underpoweredCaveat(observe) {
|
|
|
91
96
|
message: `样本量仅 ${observe.segmentCount} 段(underpowered) — 以下为低置信参考信号,需累积更多真实使用 trace 才能下结论`,
|
|
92
97
|
};
|
|
93
98
|
}
|
|
99
|
+
function capByToolOutcomeConfidence(severity, observe) {
|
|
100
|
+
const resolved = observe.toolResolvedCount;
|
|
101
|
+
if (resolved === undefined)
|
|
102
|
+
return severity;
|
|
103
|
+
const comparable = Math.max(0, resolved - (observe.toolCancelledCount ?? 0));
|
|
104
|
+
if (comparable < 5)
|
|
105
|
+
return 'low';
|
|
106
|
+
if (comparable < 20 && severity === 'high')
|
|
107
|
+
return 'medium';
|
|
108
|
+
return severity;
|
|
109
|
+
}
|
|
110
|
+
function toolOutcomeCaveat(observe) {
|
|
111
|
+
const resolved = observe.toolResolvedCount;
|
|
112
|
+
if (resolved === undefined)
|
|
113
|
+
return null;
|
|
114
|
+
const cancelled = observe.toolCancelledCount ?? 0;
|
|
115
|
+
const comparable = Math.max(0, resolved - cancelled);
|
|
116
|
+
if (comparable >= 20)
|
|
117
|
+
return null;
|
|
118
|
+
const unknown = observe.toolUnknownCount ?? Math.max(0, (observe.toolCallCount ?? resolved) - resolved);
|
|
119
|
+
return {
|
|
120
|
+
perspective: 'observe',
|
|
121
|
+
status: 'silent',
|
|
122
|
+
message: `仅 ${comparable} 次工具结果可比较${cancelled > 0 ? `;另有 ${cancelled} 次取消` : ''}${unknown > 0 ? `;另有 ${unknown} 次状态未知` : ''}。失败率的结果覆盖度不足,需积累更多可比较调用`,
|
|
123
|
+
};
|
|
124
|
+
}
|
|
94
125
|
function pickIllustrations(samples, n = 2) {
|
|
95
126
|
return samples.slice(0, n).map((s) => ({
|
|
96
127
|
sampleId: s.sampleId,
|
|
@@ -396,8 +427,16 @@ function detectProductionInstability(doctor, observe) {
|
|
|
396
427
|
return null;
|
|
397
428
|
const depRule = doctor?.results.find((r) => r.ruleId === 'dependencies_present'
|
|
398
429
|
&& (r.status === 'warn' || r.status === 'fail'));
|
|
399
|
-
const severity = capByObserveConfidence('high', observe);
|
|
400
|
-
const
|
|
430
|
+
const severity = capByToolOutcomeConfidence(capByObserveConfidence('high', observe), observe);
|
|
431
|
+
const caveats = [
|
|
432
|
+
underpoweredCaveat(observe),
|
|
433
|
+
toolOutcomeCaveat(observe),
|
|
434
|
+
].filter((value) => value !== null);
|
|
435
|
+
const resolvedOutcomeCount = observe.toolResolvedCount
|
|
436
|
+
?? observe.toolCallCount
|
|
437
|
+
?? observe.segmentCount;
|
|
438
|
+
const comparableOutcomeCount = Math.max(0, resolvedOutcomeCount - (observe.toolCancelledCount ?? 0));
|
|
439
|
+
const failureCount = Math.round(comparableOutcomeCount * observe.failureRate);
|
|
401
440
|
return {
|
|
402
441
|
id: 'production-instability',
|
|
403
442
|
category: 'production-instability',
|
|
@@ -405,7 +444,7 @@ function detectProductionInstability(doctor, observe) {
|
|
|
405
444
|
title: `生产环境跑这个 skill 时,工具失败率 ${(observe.failureRate * 100).toFixed(0)}%`,
|
|
406
445
|
description: '真实用户用这个 skill,LLM 调出去的工具经常报错。可能是凭证/网络/上游 SLA 问题,也可能是 skill 教错了用什么工具/参数。',
|
|
407
446
|
severity,
|
|
408
|
-
affectedCount:
|
|
447
|
+
affectedCount: failureCount,
|
|
409
448
|
stageRefs: {
|
|
410
449
|
observeRefs: ['high-failure-rate'],
|
|
411
450
|
...(depRule ? { doctorRuleIds: ['dependencies_present'] } : {}),
|
|
@@ -427,7 +466,7 @@ function detectProductionInstability(doctor, observe) {
|
|
|
427
466
|
? 'doctor 没警告依赖,失败可能在 CLI/凭证/上游 API 层(skill 自身可能没错)'
|
|
428
467
|
: '没跑 doctor,无法对照',
|
|
429
468
|
},
|
|
430
|
-
...
|
|
469
|
+
...caveats,
|
|
431
470
|
],
|
|
432
471
|
recommendations: [
|
|
433
472
|
{
|