oh-my-knowledge 0.48.0 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -18
- package/README.zh.md +55 -23
- package/dist/analysis/coverage-analyzer.d.ts +1 -0
- package/dist/analysis/coverage-analyzer.js +125 -62
- package/dist/analysis/failure-clusterer.js +2 -1
- package/dist/analysis/gap-analyzer.d.ts +2 -2
- package/dist/analysis/gap-analyzer.js +13 -3
- package/dist/analysis/hedging-classifier.d.ts +2 -2
- package/dist/analysis/hedging-classifier.js +3 -4
- package/dist/analysis/report-diagnostics.js +9 -7
- package/dist/analysis/sample-diagnostics.js +6 -6
- package/dist/artifact-graph/doctor.js +15 -7
- package/dist/assets/agent-skills/omk/SKILL.md +27 -7
- package/dist/assets/agent-skills/omk/references/commands.md +18 -17
- package/dist/authoring/evolver.d.ts +10 -6
- package/dist/authoring/evolver.js +496 -83
- package/dist/authoring/generator.d.ts +3 -3
- package/dist/authoring/generator.js +5 -10
- package/dist/authoring/sample-fixer.d.ts +8 -6
- package/dist/authoring/sample-fixer.js +76 -5
- package/dist/cli/commands/doctor.js +31 -14
- package/dist/cli/commands/eval/index.d.ts +3 -0
- package/dist/cli/commands/eval/index.js +163 -17
- package/dist/cli/commands/evolve.d.ts +4 -4
- package/dist/cli/commands/evolve.js +27 -13
- package/dist/cli/commands/init.js +16 -3
- package/dist/cli/commands/observe/inbox.js +28 -21
- package/dist/cli/commands/observe/index.js +20 -11
- package/dist/cli/commands/observe/ingest.d.ts +3 -0
- package/dist/cli/commands/observe/ingest.js +30 -2
- package/dist/cli/commands/sample.d.ts +6 -3
- package/dist/cli/commands/sample.js +72 -68
- package/dist/cli/lib/codex-model-hint.d.ts +9 -0
- package/dist/cli/lib/codex-model-hint.js +45 -0
- package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
- package/dist/cli/lib/generation-failure-hint.js +61 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +4 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +38 -6
- package/dist/cli/lib/i18n-dict/help.js +6 -6
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +13 -9
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +34 -2
- package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
- package/dist/cli/lib/llm-failure-classifier.js +8 -0
- package/dist/cli/lib/parse-run-config.d.ts +6 -5
- package/dist/cli/lib/parse-run-config.js +16 -9
- package/dist/cli/lib/runtime-defaults.d.ts +21 -0
- package/dist/cli/lib/runtime-defaults.js +79 -0
- package/dist/diagnosis/observe-mapper.js +14 -15
- package/dist/diagnosis/observe-producer.js +3 -1
- package/dist/diagnosis/studio-projection.js +14 -7
- package/dist/diagnosis/types.d.ts +2 -0
- package/dist/diagnosis/types.js +12 -0
- package/dist/doctor/endpoint-rule.js +2 -1
- package/dist/eval-core/artifact-file-names.js +18 -1
- package/dist/eval-core/artifact-index.d.ts +7 -11
- package/dist/eval-core/artifact-index.js +139 -80
- package/dist/eval-core/cache.d.ts +12 -3
- package/dist/eval-core/cache.js +89 -29
- package/dist/eval-core/comparability.js +10 -6
- package/dist/eval-core/evaluation-execution.d.ts +2 -1
- package/dist/eval-core/evaluation-execution.js +122 -37
- package/dist/eval-core/evaluation-job.d.ts +4 -1
- package/dist/eval-core/evaluation-job.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +15 -13
- package/dist/eval-core/evaluation-reporting.js +54 -52
- package/dist/eval-core/execution-strategy.d.ts +2 -0
- package/dist/eval-core/execution-strategy.js +11 -9
- package/dist/eval-core/fact-checker.js +15 -7
- package/dist/eval-core/holdout.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +2 -2
- package/dist/eval-core/mock-hook.cjs +23 -6
- package/dist/eval-core/mocks-runtime.js +30 -8
- package/dist/eval-core/report-document.d.ts +12 -0
- package/dist/eval-core/report-document.js +1151 -0
- package/dist/eval-core/report-extensions.d.ts +4 -0
- package/dist/eval-core/report-extensions.js +500 -0
- package/dist/eval-core/report-file-migration.js +7 -2
- package/dist/eval-core/resume-compatibility.d.ts +31 -0
- package/dist/eval-core/resume-compatibility.js +141 -0
- package/dist/eval-core/sample-fingerprint.d.ts +12 -0
- package/dist/eval-core/sample-fingerprint.js +193 -0
- package/dist/eval-core/schema.js +86 -31
- package/dist/eval-core/verdict.d.ts +8 -4
- package/dist/eval-core/verdict.js +24 -10
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
- package/dist/eval-workflows/evaluation-pipeline.js +12 -7
- package/dist/eval-workflows/run-evaluation.d.ts +9 -7
- package/dist/eval-workflows/run-evaluation.js +79 -51
- package/dist/executors/anthropic-api.js +65 -9
- package/dist/executors/claude-cli.js +16 -79
- package/dist/executors/claude-protocol.d.ts +28 -0
- package/dist/executors/claude-protocol.js +180 -0
- package/dist/executors/claude-sdk-trace.js +56 -28
- package/dist/executors/claude-sdk.d.ts +1 -0
- package/dist/executors/claude-sdk.js +39 -93
- package/dist/executors/codex-cli-trace.js +166 -31
- package/dist/executors/codex-cli.d.ts +6 -8
- package/dist/executors/codex-cli.js +49 -151
- package/dist/executors/codex-protocol.d.ts +24 -0
- package/dist/executors/codex-protocol.js +234 -0
- package/dist/executors/codex-sdk.js +68 -120
- package/dist/executors/gemini.js +88 -13
- package/dist/executors/index.d.ts +2 -3
- package/dist/executors/index.js +5 -3
- package/dist/executors/openai-api.js +70 -9
- package/dist/executors/runtime-fingerprint.js +88 -11
- package/dist/executors/script-command.d.ts +8 -0
- package/dist/executors/script-command.js +87 -0
- package/dist/executors/script.js +202 -29
- package/dist/executors/shared.d.ts +35 -3
- package/dist/executors/shared.js +113 -15
- package/dist/grading/assertions.d.ts +1 -1
- package/dist/grading/assertions.js +19 -9
- package/dist/grading/diagnostic.d.ts +9 -2
- package/dist/grading/diagnostic.js +25 -2
- package/dist/grading/index.js +10 -4
- package/dist/grading/judge.js +19 -6
- package/dist/grading/layered-scores.d.ts +2 -3
- package/dist/grading/layered-scores.js +2 -3
- package/dist/inputs/load-samples.d.ts +1 -2
- package/dist/inputs/load-samples.js +23 -1
- package/dist/inputs/mcp-resolver.js +6 -3
- package/dist/inputs/sample-document.d.ts +11 -0
- package/dist/inputs/sample-document.js +96 -0
- package/dist/managed/evidence.d.ts +1 -0
- package/dist/managed/evidence.js +1 -1
- package/dist/managed/store.js +200 -91
- package/dist/observability/codex-trace-adapter.d.ts +5 -0
- package/dist/observability/codex-trace-adapter.js +850 -0
- package/dist/observability/experience.d.ts +32 -6
- package/dist/observability/experience.js +2695 -459
- package/dist/observability/feedback-matchers.js +16 -1
- package/dist/observability/inbox-view-model.d.ts +1 -1
- package/dist/observability/inbox-view-model.js +19 -14
- package/dist/observability/inbox.d.ts +7 -1
- package/dist/observability/inbox.js +632 -124
- package/dist/observability/problem-patterns.js +2 -0
- package/dist/observability/review-state.d.ts +6 -0
- package/dist/observability/review-state.js +235 -63
- package/dist/observability/skill-chain-advisories.js +1 -1
- package/dist/observability/skill-chain.js +17 -4
- package/dist/observability/skill-health-analyzer.d.ts +32 -7
- package/dist/observability/skill-health-analyzer.js +194 -121
- package/dist/observability/skill-health-report.d.ts +10 -0
- package/dist/observability/skill-health-report.js +620 -0
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +1 -1
- package/dist/observability/soft-standards/index.js +1 -1
- package/dist/observability/soft-standards/llm-extractor.js +8 -10
- package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
- package/dist/observability/soft-standards/skill-standards-store.js +59 -18
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/observability/trace-adapter.d.ts +12 -7
- package/dist/observability/trace-adapter.js +11 -9
- package/dist/observability/trace-attribution.d.ts +13 -5
- package/dist/observability/trace-attribution.js +315 -21
- package/dist/observability/trace-ingestion.d.ts +9 -0
- package/dist/observability/trace-ingestion.js +80 -0
- package/dist/observability/trace-ir.d.ts +113 -0
- package/dist/observability/trace-ir.js +87 -0
- package/dist/observability/trace-segmenter.d.ts +19 -6
- package/dist/observability/trace-segmenter.js +377 -196
- package/dist/observability/trace-session-index.d.ts +19 -0
- package/dist/observability/trace-session-index.js +68 -0
- package/dist/observability/trace-source.d.ts +12 -4
- package/dist/observability/trace-source.js +939 -215
- package/dist/renderer/html-renderer.js +37 -6
- package/dist/renderer/icons.js +3 -0
- package/dist/renderer/observation-inbox-renderer.js +208 -90
- package/dist/renderer/skill-detail-renderer.js +452 -109
- package/dist/renderer/skill-health-renderer.js +69 -12
- package/dist/renderer/summary.js +28 -7
- package/dist/renderer/table.js +21 -4
- package/dist/renderer/test-view.d.ts +1 -0
- package/dist/renderer/test-view.js +44 -9
- package/dist/server/indexed-report-store.js +14 -18
- package/dist/server/job-store.js +64 -26
- package/dist/server/report-server.js +190 -78
- package/dist/server/report-store.js +57 -80
- package/dist/server/skill-index.js +143 -49
- package/dist/server/skill-insights.js +44 -5
- package/dist/shared/artifact-graph.d.ts +3 -0
- package/dist/shared/artifact-graph.js +224 -0
- package/dist/shared/assertion-types.d.ts +8 -0
- package/dist/shared/assertion-types.js +46 -0
- package/dist/shared/atomic-json.d.ts +8 -0
- package/dist/shared/atomic-json.js +33 -0
- package/dist/shared/diagnosis-schema.d.ts +9 -0
- package/dist/shared/diagnosis-schema.js +181 -0
- package/dist/shared/doctor-report.d.ts +3 -0
- package/dist/shared/doctor-report.js +103 -0
- package/dist/shared/evaluation-job.d.ts +6 -0
- package/dist/shared/evaluation-job.js +217 -0
- package/dist/shared/executor-result.d.ts +17 -0
- package/dist/shared/executor-result.js +221 -0
- package/dist/shared/file-lock.d.ts +12 -0
- package/dist/shared/file-lock.js +129 -0
- package/dist/shared/json-value.d.ts +5 -0
- package/dist/shared/json-value.js +36 -0
- package/dist/shared/keyed-mutex.d.ts +7 -0
- package/dist/shared/keyed-mutex.js +24 -0
- package/dist/shared/record-count.d.ts +8 -0
- package/dist/shared/record-count.js +43 -0
- package/dist/shared/sample-contract.d.ts +3 -0
- package/dist/shared/sample-contract.js +332 -0
- package/dist/shared/timestamp.d.ts +6 -0
- package/dist/shared/timestamp.js +64 -0
- package/dist/shared/token-usage.d.ts +19 -0
- package/dist/shared/token-usage.js +50 -0
- package/dist/shared/tool-call-status.d.ts +8 -0
- package/dist/shared/tool-call-status.js +28 -0
- package/dist/shared/tool-identity.d.ts +21 -0
- package/dist/shared/tool-identity.js +84 -0
- package/dist/shared/tool-search.js +73 -16
- package/dist/shared/trace-projection.d.ts +5 -0
- package/dist/shared/trace-projection.js +20 -0
- package/dist/shared/trace-source-kind.d.ts +3 -0
- package/dist/shared/trace-source-kind.js +12 -0
- package/dist/types/diagnosis.d.ts +2 -0
- package/dist/types/eval.d.ts +4 -0
- package/dist/types/executor.d.ts +32 -5
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/judge.d.ts +2 -0
- package/dist/types/observability.d.ts +116 -9
- package/dist/types/report.d.ts +58 -6
- package/dist/types/skill-index.d.ts +7 -0
- package/dist/types/trace.d.ts +2 -0
- package/dist/types/trace.js +1 -0
- package/package.json +9 -5
|
@@ -9,10 +9,15 @@
|
|
|
9
9
|
* 三域共用 `artifactIndexDir(domain)` + tmp-rename 原子写 + 指纹缓存读 的同一套机制,各域只差「投影成卡片」
|
|
10
10
|
* 与「卡片还原成 snapshot」两段域特定逻辑。
|
|
11
11
|
*/
|
|
12
|
-
import { existsSync,
|
|
13
|
-
import { join, resolve } from 'node:path';
|
|
12
|
+
import { existsSync, readdirSync, readFileSync, unlinkSync, statSync } from 'node:fs';
|
|
13
|
+
import { dirname, join, resolve } from 'node:path';
|
|
14
14
|
import { DEFAULT_ARTIFACT_INDEX_DIR } from './default-dirs.js';
|
|
15
15
|
import { globalReportsDir, globalDoctorsDir, globalObserveHealthDir } from './measurement-dirs.js';
|
|
16
|
+
import { setOwnRecordValue, sumRecordCounts } from '../shared/record-count.js';
|
|
17
|
+
import { writeJsonFileAtomic } from '../shared/atomic-json.js';
|
|
18
|
+
import { isRfc3339Timestamp } from '../shared/timestamp.js';
|
|
19
|
+
import { reportFilePath, safeArtifactFileStem } from './artifact-file-names.js';
|
|
20
|
+
import { parseReportDocument, parseReportIndexCard } from './report-document.js';
|
|
16
21
|
// ── 通用机制(域无关)──────────────────────────────────────────────────────────
|
|
17
22
|
/** 索引根:`OMK_ARTIFACT_INDEX_DIR` 覆盖(测试隔离,仿 OMK_TREES_DIR),默认 state/artifact-index。 */
|
|
18
23
|
function artifactIndexRoot() {
|
|
@@ -32,22 +37,29 @@ export function shouldIndexReport(outputDir) {
|
|
|
32
37
|
return shouldIndexDir(outputDir, globalReportsDir());
|
|
33
38
|
}
|
|
34
39
|
function safeFileName(id) {
|
|
35
|
-
return id
|
|
40
|
+
return safeArtifactFileStem(id);
|
|
41
|
+
}
|
|
42
|
+
function isCanonicalCardId(value) {
|
|
43
|
+
return typeof value === 'string' && value.length > 0 && safeFileName(value) === value;
|
|
44
|
+
}
|
|
45
|
+
function isCanonicalCardPath(path, id) {
|
|
46
|
+
return typeof path === 'string'
|
|
47
|
+
&& resolve(path) === path
|
|
48
|
+
&& reportFilePath(dirname(path), id) === path;
|
|
36
49
|
}
|
|
37
50
|
// 卡片读侧小 guard:索引是可重生 scratch,坏卡片(脏文件 / 别域误落 / 字段缺失)读侧从严跳过,
|
|
38
51
|
// 不让 undefined / 非法枚举 / NaN 当可信输入污染 studio。
|
|
39
52
|
const isFiniteNumber = (v) => typeof v === 'number' && Number.isFinite(v);
|
|
53
|
+
const isNonNegativeInteger = (v) => Number.isSafeInteger(v) && v >= 0;
|
|
54
|
+
const isRate = (v) => isFiniteNumber(v) && v >= 0 && v <= 1;
|
|
40
55
|
const isDoctorStatus = (v) => v === 'pass' || v === 'warn' || v === 'fail';
|
|
41
56
|
const isHealthBand = (v) => v === 'green' || v === 'yellow' || v === 'red';
|
|
42
57
|
const isConfidence = (v) => v === undefined || v === 'high' || v === 'low' || v === 'underpowered';
|
|
43
58
|
/** 原子写一张卡片(tmp+rename,防半截 JSON 被 reader 读到)。 */
|
|
44
59
|
function writeCard(domain, id, card) {
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
const tmp = `${target}.tmp.${process.pid}.${Math.random().toString(36).slice(2)}`;
|
|
49
|
-
writeFileSync(tmp, JSON.stringify(card, null, 2));
|
|
50
|
-
renameSync(tmp, target);
|
|
60
|
+
if (!isCanonicalCardId(id))
|
|
61
|
+
throw new Error('invalid artifact index id');
|
|
62
|
+
writeJsonFileAtomic(join(artifactIndexDir(domain), `${id}.json`), card);
|
|
51
63
|
}
|
|
52
64
|
/** 读某域全部卡片,逐条用 `valid` 谓词过滤坏文件 / 缺字段(索引是 scratch,读侧从严)。 */
|
|
53
65
|
function readArtifactCards(domain, valid) {
|
|
@@ -62,12 +74,14 @@ function readArtifactCards(domain, valid) {
|
|
|
62
74
|
return [];
|
|
63
75
|
}
|
|
64
76
|
const out = [];
|
|
65
|
-
for (const f of files) {
|
|
77
|
+
for (const f of files.sort()) {
|
|
66
78
|
if (!f.endsWith('.json') || f.includes('.json.tmp.'))
|
|
67
79
|
continue;
|
|
68
80
|
try {
|
|
69
81
|
const c = JSON.parse(readFileSync(join(dir, f), 'utf-8'));
|
|
70
|
-
if (valid(c)
|
|
82
|
+
if (valid(c)
|
|
83
|
+
&& isCanonicalCardId(c.id)
|
|
84
|
+
&& f === `${c.id}.json`)
|
|
71
85
|
out.push(c);
|
|
72
86
|
}
|
|
73
87
|
catch { /* skip corrupt card */ }
|
|
@@ -76,8 +90,10 @@ function readArtifactCards(domain, valid) {
|
|
|
76
90
|
}
|
|
77
91
|
/** 删某域某 id 的卡片。幂等、best-effort。返回卡片是否曾存在。 */
|
|
78
92
|
function removeArtifactCard(domain, id) {
|
|
93
|
+
if (!isCanonicalCardId(id))
|
|
94
|
+
return false;
|
|
79
95
|
try {
|
|
80
|
-
const p = join(artifactIndexDir(domain), `${
|
|
96
|
+
const p = join(artifactIndexDir(domain), `${id}.json`);
|
|
81
97
|
if (!existsSync(p))
|
|
82
98
|
return false;
|
|
83
99
|
unlinkSync(p);
|
|
@@ -103,33 +119,20 @@ function liveCards(cards) {
|
|
|
103
119
|
* 缓存失效 → live 过滤生效。读卡片只为拿 path,字段宽松。
|
|
104
120
|
*/
|
|
105
121
|
export function cardTargetSentinel(domain) {
|
|
106
|
-
const
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
files = readdirSync(dir);
|
|
112
|
-
}
|
|
113
|
-
catch {
|
|
114
|
-
return '';
|
|
115
|
-
}
|
|
122
|
+
const cards = domain === 'report'
|
|
123
|
+
? listReportCards()
|
|
124
|
+
: domain === 'doctor'
|
|
125
|
+
? listDoctorCards()
|
|
126
|
+
: listObserveCards();
|
|
116
127
|
const parts = [];
|
|
117
|
-
for (const
|
|
118
|
-
if (!f.endsWith('.json') || f.includes('.json.tmp.'))
|
|
119
|
-
continue;
|
|
128
|
+
for (const card of cards) {
|
|
120
129
|
try {
|
|
121
|
-
const
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
parts.push(`${f}=${s.mtimeMs}:${s.size}`);
|
|
127
|
-
}
|
|
128
|
-
catch {
|
|
129
|
-
parts.push(`${f}=gone`);
|
|
130
|
-
}
|
|
130
|
+
const stat = statSync(card.path);
|
|
131
|
+
parts.push(`${card.id}.json=${stat.mtimeMs}:${stat.size}`);
|
|
132
|
+
}
|
|
133
|
+
catch {
|
|
134
|
+
parts.push(`${card.id}.json=gone`);
|
|
131
135
|
}
|
|
132
|
-
catch { /* skip corrupt */ }
|
|
133
136
|
}
|
|
134
137
|
return parts.sort().join(',');
|
|
135
138
|
}
|
|
@@ -137,8 +140,8 @@ export function cardTargetSentinel(domain) {
|
|
|
137
140
|
/** 报告投影成卡片(剥掉 results 重体)。 */
|
|
138
141
|
function reportCard(report, sourcePath) {
|
|
139
142
|
return report.kind === 'evaluation'
|
|
140
|
-
? { domain: 'report', id: report.id, path: sourcePath, kind: 'evaluation', meta: report.meta, summary: report.summary }
|
|
141
|
-
: { domain: 'report', id: report.id, path: sourcePath, kind: 'batch-evaluation', meta: report.meta, items: report.items };
|
|
143
|
+
? { domain: 'report', id: report.id, path: resolve(sourcePath), kind: 'evaluation', meta: report.meta, summary: report.summary }
|
|
144
|
+
: { domain: 'report', id: report.id, path: resolve(sourcePath), kind: 'batch-evaluation', meta: report.meta, items: report.items };
|
|
142
145
|
}
|
|
143
146
|
/**
|
|
144
147
|
* persistReport 的索引钩子:报告落盘后 best-effort 追加卡片。永不抛、永不阻断报告落盘。
|
|
@@ -148,10 +151,10 @@ export function indexReportWrite(report, sourcePath, outputDir) {
|
|
|
148
151
|
try {
|
|
149
152
|
if (!shouldIndexReport(outputDir))
|
|
150
153
|
return;
|
|
151
|
-
const
|
|
152
|
-
if (
|
|
154
|
+
const parsed = parseReportDocument(report, report.id, report.id);
|
|
155
|
+
if (!parsed)
|
|
153
156
|
return;
|
|
154
|
-
writeCard('report', report.id, reportCard(
|
|
157
|
+
writeCard('report', report.id, reportCard(parsed, sourcePath));
|
|
155
158
|
}
|
|
156
159
|
catch {
|
|
157
160
|
// 索引可重建,失败静默(可选 stderr warn);正文已落盘不受影响。
|
|
@@ -159,24 +162,12 @@ export function indexReportWrite(report, sourcePath, outputDir) {
|
|
|
159
162
|
}
|
|
160
163
|
/** 读 report 域全部卡片(跳过坏文件 / 缺字段 / 坏 kind)。 */
|
|
161
164
|
export function listReportCards() {
|
|
162
|
-
return readArtifactCards('report', (
|
|
163
|
-
const card = c;
|
|
164
|
-
// kind 白名单:cardToReportDocument 对任何非 evaluation 一律按 batch 投影,坏 kind(拼错 / 'doctor')
|
|
165
|
-
// 会污染机器级 list 成空 batch 报告。索引是可重生 scratch,读侧从严跳过坏卡片。
|
|
166
|
-
const kindOk = card?.kind === 'evaluation' || card?.kind === 'batch-evaluation';
|
|
167
|
-
return !!card && card.domain === 'report' && kindOk && typeof card.id === 'string' && typeof card.path === 'string' && !!card.meta;
|
|
168
|
-
});
|
|
165
|
+
return readArtifactCards('report', (value) => parseReportIndexCard(value) !== null);
|
|
169
166
|
}
|
|
170
167
|
/** report 卡片(过滤悬空真身):供 studio 机器级 list / findBy 展示。 */
|
|
171
168
|
export function listLiveReportCards() {
|
|
172
169
|
return liveCards(listReportCards());
|
|
173
170
|
}
|
|
174
|
-
/** 卡片 → ReportDocument(results:[]):供 studio list / buildSkillIndex / trend 消费(它们不读 results)。 */
|
|
175
|
-
export function cardToReportDocument(card) {
|
|
176
|
-
return card.kind === 'evaluation'
|
|
177
|
-
? { kind: 'evaluation', id: card.id, meta: card.meta, summary: card.summary ?? {}, results: [] }
|
|
178
|
-
: { kind: 'batch-evaluation', id: card.id, mode: 'skill', meta: card.meta, items: card.items ?? [] };
|
|
179
|
-
}
|
|
180
171
|
/** 删 report 域某 id 的卡片(DELETE 报告时连卡片一起删,使其从机器级 list 消失)。幂等、best-effort。 */
|
|
181
172
|
export function removeReportCard(id) {
|
|
182
173
|
return removeArtifactCard('report', id);
|
|
@@ -186,7 +177,11 @@ export function indexDoctorWrite(card, outputDir) {
|
|
|
186
177
|
try {
|
|
187
178
|
if (!shouldIndexDir(outputDir, globalDoctorsDir()))
|
|
188
179
|
return;
|
|
189
|
-
writeCard('doctor', card.id, {
|
|
180
|
+
writeCard('doctor', card.id, {
|
|
181
|
+
domain: 'doctor',
|
|
182
|
+
...card,
|
|
183
|
+
path: resolve(card.path),
|
|
184
|
+
});
|
|
190
185
|
}
|
|
191
186
|
catch { /* 索引可重建,失败静默 */ }
|
|
192
187
|
}
|
|
@@ -194,26 +189,30 @@ export function indexDoctorWrite(card, outputDir) {
|
|
|
194
189
|
export function listDoctorCards() {
|
|
195
190
|
return readArtifactCards('doctor', (c) => {
|
|
196
191
|
const card = c;
|
|
197
|
-
return !!card && card.domain === 'doctor' &&
|
|
198
|
-
&&
|
|
192
|
+
return !!card && card.domain === 'doctor' && isCanonicalCardId(card.id)
|
|
193
|
+
&& isCanonicalCardPath(card.path, card.id)
|
|
194
|
+
&& typeof card.skillName === 'string' && card.skillName.length > 0
|
|
195
|
+
&& typeof card.reportId === 'string' && card.reportId.length > 0
|
|
196
|
+
&& isRfc3339Timestamp(card.timestamp)
|
|
199
197
|
&& isDoctorStatus(card.status)
|
|
200
|
-
&&
|
|
198
|
+
&& isNonNegativeInteger(card.passCount)
|
|
199
|
+
&& isNonNegativeInteger(card.warnCount)
|
|
200
|
+
&& isNonNegativeInteger(card.failCount)
|
|
201
|
+
&& (() => {
|
|
202
|
+
try {
|
|
203
|
+
sumRecordCounts(card.passCount, card.warnCount, card.failCount);
|
|
204
|
+
return true;
|
|
205
|
+
}
|
|
206
|
+
catch {
|
|
207
|
+
return false;
|
|
208
|
+
}
|
|
209
|
+
})();
|
|
201
210
|
});
|
|
202
211
|
}
|
|
203
212
|
/** doctor 卡片(过滤悬空真身):供 buildSkillIndex 机器级合并。 */
|
|
204
213
|
export function listLiveDoctorCards() {
|
|
205
214
|
return liveCards(listDoctorCards());
|
|
206
215
|
}
|
|
207
|
-
/** doctor 卡片 → (skillName, SkillDoctorSnapshot)。results:[](逐规则详情需回源项目看)。 */
|
|
208
|
-
export function cardToDoctorSnapshot(card) {
|
|
209
|
-
return {
|
|
210
|
-
skillName: card.skillName,
|
|
211
|
-
snap: {
|
|
212
|
-
reportId: card.reportId, timestamp: card.timestamp, status: card.status,
|
|
213
|
-
passCount: card.passCount, warnCount: card.warnCount, failCount: card.failCount, results: [],
|
|
214
|
-
},
|
|
215
|
-
};
|
|
216
|
-
}
|
|
217
216
|
/** 删 doctor 域某 id(文件 stem)的卡片。doctor 历史按 50/skill prune,删正文时必须连卡片一起删,
|
|
218
217
|
* 否则被 prune 掉的报告会经卡片合并在本项目 studio「复活」。 */
|
|
219
218
|
export function removeDoctorCard(id) {
|
|
@@ -226,13 +225,21 @@ export function indexObserveWrite(report, sourcePath, outputDir, id) {
|
|
|
226
225
|
return;
|
|
227
226
|
const bySkill = {};
|
|
228
227
|
for (const [name, h] of Object.entries(report.bySkill || {})) {
|
|
229
|
-
bySkill
|
|
230
|
-
toolFailureRate: h.toolFailureRate,
|
|
231
|
-
|
|
232
|
-
|
|
228
|
+
setOwnRecordValue(bySkill, name, {
|
|
229
|
+
toolFailureRate: h.toolFailureRate,
|
|
230
|
+
toolFailureCount: h.toolFailureCount,
|
|
231
|
+
toolCallCount: h.toolCallCount,
|
|
232
|
+
toolResolvedCount: h.toolResolvedCount,
|
|
233
|
+
toolCancelledCount: h.toolCancelledCount,
|
|
234
|
+
toolUnknownCount: h.toolUnknownCount,
|
|
235
|
+
segmentCount: h.segmentCount,
|
|
236
|
+
gap: { weightedGapRate: h.gap?.weightedGapRate ?? 0 },
|
|
237
|
+
confidence: h.confidence,
|
|
238
|
+
stability: h.stability,
|
|
239
|
+
});
|
|
233
240
|
}
|
|
234
241
|
const card = {
|
|
235
|
-
domain: 'observe-health', id, path: sourcePath,
|
|
242
|
+
domain: 'observe-health', id, path: resolve(sourcePath),
|
|
236
243
|
meta: { generatedAt: report.meta.generatedAt, sessionCount: report.meta.sessionCount, segmentCount: report.meta.segmentCount },
|
|
237
244
|
overall: { healthBand: report.overall.healthBand, confidence: report.overall.confidence },
|
|
238
245
|
bySkill,
|
|
@@ -245,26 +252,78 @@ export function indexObserveWrite(report, sourcePath, outputDir, id) {
|
|
|
245
252
|
export function listObserveCards() {
|
|
246
253
|
return readArtifactCards('observe-health', (c) => {
|
|
247
254
|
const card = c;
|
|
248
|
-
if (!card
|
|
255
|
+
if (!card
|
|
256
|
+
|| card.domain !== 'observe-health'
|
|
257
|
+
|| !isCanonicalCardId(card.id)
|
|
258
|
+
|| !isCanonicalCardPath(card.path, card.id))
|
|
249
259
|
return false;
|
|
250
260
|
const meta = card.meta;
|
|
251
261
|
const overall = card.overall;
|
|
252
262
|
const bySkill = card.bySkill;
|
|
253
|
-
if (!meta
|
|
263
|
+
if (!meta
|
|
264
|
+
|| !isNonNegativeInteger(meta.sessionCount)
|
|
265
|
+
|| !isNonNegativeInteger(meta.segmentCount)
|
|
266
|
+
|| !isRfc3339Timestamp(meta.generatedAt))
|
|
254
267
|
return false;
|
|
255
268
|
if (!overall || !isHealthBand(overall.healthBand) || !isConfidence(overall.confidence))
|
|
256
269
|
return false;
|
|
257
|
-
if (!bySkill || typeof bySkill !== 'object')
|
|
270
|
+
if (!bySkill || typeof bySkill !== 'object' || Array.isArray(bySkill))
|
|
258
271
|
return false;
|
|
259
272
|
// 每个 skill 的标量也校验:坏 toolFailureRate / segmentCount / gap.weightedGapRate 会让 bandFromObserveHealth /
|
|
260
273
|
// 趋势 / 健康快照出 NaN(buildSkillIndex 直接取 h.gap?.weightedGapRate ?? 0 进 gapRate、参与阈值判断)。
|
|
261
|
-
|
|
262
|
-
|
|
274
|
+
const segmentCounts = [];
|
|
275
|
+
for (const [skillName, h] of Object.entries(bySkill)) {
|
|
276
|
+
if (skillName.length === 0)
|
|
277
|
+
return false;
|
|
278
|
+
if (!h || !isRate(h.toolFailureRate) || !isNonNegativeInteger(h.segmentCount) || !isConfidence(h.confidence))
|
|
279
|
+
return false;
|
|
280
|
+
segmentCounts.push(h.segmentCount);
|
|
281
|
+
if (h.toolFailureCount !== undefined && !isNonNegativeInteger(h.toolFailureCount))
|
|
282
|
+
return false;
|
|
283
|
+
if (h.toolCallCount !== undefined && !isNonNegativeInteger(h.toolCallCount))
|
|
284
|
+
return false;
|
|
285
|
+
if (h.toolResolvedCount !== undefined && !isNonNegativeInteger(h.toolResolvedCount))
|
|
286
|
+
return false;
|
|
287
|
+
if (h.toolCancelledCount !== undefined && !isNonNegativeInteger(h.toolCancelledCount))
|
|
288
|
+
return false;
|
|
289
|
+
if (h.toolUnknownCount !== undefined && !isNonNegativeInteger(h.toolUnknownCount))
|
|
263
290
|
return false;
|
|
264
|
-
if (h.
|
|
291
|
+
if (h.toolCallCount === undefined
|
|
292
|
+
&& (h.toolResolvedCount !== undefined
|
|
293
|
+
|| h.toolCancelledCount !== undefined
|
|
294
|
+
|| h.toolUnknownCount !== undefined))
|
|
295
|
+
return false;
|
|
296
|
+
if (h.toolCallCount !== undefined
|
|
297
|
+
&& ((h.toolResolvedCount ?? 0) > h.toolCallCount
|
|
298
|
+
|| (h.toolCancelledCount ?? 0) > (h.toolResolvedCount ?? h.toolCallCount)
|
|
299
|
+
|| (h.toolUnknownCount ?? 0) > h.toolCallCount
|
|
300
|
+
|| (h.toolResolvedCount !== undefined
|
|
301
|
+
&& h.toolUnknownCount !== undefined
|
|
302
|
+
&& h.toolResolvedCount + h.toolUnknownCount !== h.toolCallCount)))
|
|
303
|
+
return false;
|
|
304
|
+
if (h.toolFailureCount !== undefined
|
|
305
|
+
&& h.toolCallCount !== undefined) {
|
|
306
|
+
const resolved = h.toolResolvedCount ?? h.toolCallCount;
|
|
307
|
+
const comparable = resolved - (h.toolCancelledCount ?? 0);
|
|
308
|
+
const expectedFailureRate = comparable > 0
|
|
309
|
+
? Number((h.toolFailureCount / comparable).toFixed(4))
|
|
310
|
+
: 0;
|
|
311
|
+
if (h.toolFailureCount > comparable
|
|
312
|
+
|| Math.abs(h.toolFailureRate - expectedFailureRate) > 0.0001)
|
|
313
|
+
return false;
|
|
314
|
+
}
|
|
315
|
+
if (h.stability !== undefined
|
|
316
|
+
&& !['stable', 'unstable', 'very-unstable', 'unknown'].includes(h.stability))
|
|
317
|
+
return false;
|
|
318
|
+
if (h.gap !== undefined && h.gap.weightedGapRate !== undefined && !isRate(h.gap.weightedGapRate))
|
|
265
319
|
return false;
|
|
266
320
|
}
|
|
267
|
-
|
|
321
|
+
try {
|
|
322
|
+
return sumRecordCounts(...segmentCounts) === meta.segmentCount;
|
|
323
|
+
}
|
|
324
|
+
catch {
|
|
325
|
+
return false;
|
|
326
|
+
}
|
|
268
327
|
});
|
|
269
328
|
}
|
|
270
329
|
/** observe 卡片(过滤悬空真身):供 listAnalyses / buildSkillIndex 机器级合并。 */
|
|
@@ -2,8 +2,9 @@
|
|
|
2
2
|
* Executor result cache.
|
|
3
3
|
*
|
|
4
4
|
* Caches successful executor results to disk to avoid redundant API calls.
|
|
5
|
-
* Cache key
|
|
6
|
-
* runtime + mocks + mocksStrict + effort + artifactContentHash
|
|
5
|
+
* Cache key v9 = sha256(model + system + prompt + cwd + allowedSkills + executor +
|
|
6
|
+
* runtime + mocks + mocksStrict + effort + artifactContentHash +
|
|
7
|
+
* sampleExecutionDependencyHash).
|
|
7
8
|
* Loaded into memory on init, flushed to disk on save().
|
|
8
9
|
*
|
|
9
10
|
* Prefix bumps intentionally invalidate old entries when construct-validity
|
|
@@ -20,6 +21,12 @@
|
|
|
20
21
|
* 不动 system → 旧 key 会命中旧输出、贴到新 artifactHashes 上,形成静默测量污染。把
|
|
21
22
|
* contentHash 纳入 key,资产变即重跑。git skill 的 contentHash 只随 SKILL.md 变(其资产不
|
|
22
23
|
* 暴露给 executor、本就不该触发重跑),口径自洽
|
|
24
|
+
* - v7: executor 返回值进入统一契约校验,且所有未知成本路径显式记录 provenance。
|
|
25
|
+
* 旧缓存缺少这些语义,不能在新报告中继续复用。
|
|
26
|
+
* - v8: 所有 executor 的工具身份在统一边界归一化。旧缓存里的 provider-native
|
|
27
|
+
* 工具名不能继续参与 source-neutral 工具断言与分布统计。
|
|
28
|
+
* - v9: sample mock 的 `return_file` 内容指纹进入 key。只哈声明路径会在 fixture
|
|
29
|
+
* 内容变化后错误复用旧执行结果。
|
|
23
30
|
*/
|
|
24
31
|
import type { ExecutorCache } from '../types/index.js';
|
|
25
32
|
export declare function createCache(cacheDir: string): ExecutorCache;
|
|
@@ -35,4 +42,6 @@ mocksStrict?: boolean,
|
|
|
35
42
|
effort?: string,
|
|
36
43
|
/** artifact 内容指纹(整树 / 单文件哈)。本地 dir-skill 改 references/ 资产只动此值、不动 system,
|
|
37
44
|
* 不进 key 会让改资产后命中旧输出 → 静默污染。空(baseline / 无 skill)等价无指纹。 */
|
|
38
|
-
artifactContentHash?: string
|
|
45
|
+
artifactContentHash?: string,
|
|
46
|
+
/** External sample files that can alter executor output, currently mock return_file fixtures. */
|
|
47
|
+
sampleExecutionDependencyHash?: string): string;
|
package/dist/eval-core/cache.js
CHANGED
|
@@ -2,8 +2,9 @@
|
|
|
2
2
|
* Executor result cache.
|
|
3
3
|
*
|
|
4
4
|
* Caches successful executor results to disk to avoid redundant API calls.
|
|
5
|
-
* Cache key
|
|
6
|
-
* runtime + mocks + mocksStrict + effort + artifactContentHash
|
|
5
|
+
* Cache key v9 = sha256(model + system + prompt + cwd + allowedSkills + executor +
|
|
6
|
+
* runtime + mocks + mocksStrict + effort + artifactContentHash +
|
|
7
|
+
* sampleExecutionDependencyHash).
|
|
7
8
|
* Loaded into memory on init, flushed to disk on save().
|
|
8
9
|
*
|
|
9
10
|
* Prefix bumps intentionally invalidate old entries when construct-validity
|
|
@@ -20,10 +21,19 @@
|
|
|
20
21
|
* 不动 system → 旧 key 会命中旧输出、贴到新 artifactHashes 上,形成静默测量污染。把
|
|
21
22
|
* contentHash 纳入 key,资产变即重跑。git skill 的 contentHash 只随 SKILL.md 变(其资产不
|
|
22
23
|
* 暴露给 executor、本就不该触发重跑),口径自洽
|
|
24
|
+
* - v7: executor 返回值进入统一契约校验,且所有未知成本路径显式记录 provenance。
|
|
25
|
+
* 旧缓存缺少这些语义,不能在新报告中继续复用。
|
|
26
|
+
* - v8: 所有 executor 的工具身份在统一边界归一化。旧缓存里的 provider-native
|
|
27
|
+
* 工具名不能继续参与 source-neutral 工具断言与分布统计。
|
|
28
|
+
* - v9: sample mock 的 `return_file` 内容指纹进入 key。只哈声明路径会在 fixture
|
|
29
|
+
* 内容变化后错误复用旧执行结果。
|
|
23
30
|
*/
|
|
24
|
-
import { readFileSync,
|
|
31
|
+
import { readFileSync, mkdirSync, existsSync } from 'node:fs';
|
|
25
32
|
import { join } from 'node:path';
|
|
26
33
|
import { createHash } from 'node:crypto';
|
|
34
|
+
import { executorResultValidationError, normalizeExecResultToolIdentities, parseExecResult, } from '../shared/executor-result.js';
|
|
35
|
+
import { writeJsonFileAtomic } from '../shared/atomic-json.js';
|
|
36
|
+
import { withFileLock } from '../shared/file-lock.js';
|
|
27
37
|
const CACHE_FILE = 'executor-cache.json';
|
|
28
38
|
/** v5 保留 turns / toolCalls 后单 entry 可达 5–50 KB,长期使用会无界膨胀。
|
|
29
39
|
* Map iteration 是插入序,set() 时若 key 已存在先 delete 再 set 把它移到末尾 → 实现 LRU。
|
|
@@ -46,22 +56,9 @@ export function createCache(cacheDir) {
|
|
|
46
56
|
mkdirSync(cacheDir, { recursive: true });
|
|
47
57
|
const filePath = join(cacheDir, CACHE_FILE);
|
|
48
58
|
const cap = resolveCacheCap();
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
const store = new Map();
|
|
53
|
-
if (existsSync(filePath)) {
|
|
54
|
-
try {
|
|
55
|
-
const raw = JSON.parse(readFileSync(filePath, 'utf-8'));
|
|
56
|
-
for (const [k, v] of Object.entries(raw))
|
|
57
|
-
store.set(k, v);
|
|
58
|
-
// 加载完也 enforce 一次 cap,处理上次 process exit 前没保存到的极端膨胀。
|
|
59
|
-
evictUntilWithinCap(store, cap);
|
|
60
|
-
}
|
|
61
|
-
catch {
|
|
62
|
-
// 老 cache 坏了忽略 — 重跑会重建。
|
|
63
|
-
}
|
|
64
|
-
}
|
|
59
|
+
const store = readCacheStore(filePath, cap);
|
|
60
|
+
const pendingWrites = new Map();
|
|
61
|
+
const touchedKeys = new Set();
|
|
65
62
|
let dirty = false;
|
|
66
63
|
return {
|
|
67
64
|
get(key) {
|
|
@@ -71,6 +68,7 @@ export function createCache(cacheDir) {
|
|
|
71
68
|
// LRU touch:命中时移到末尾(最新)。这样 evict 时永远从最旧端拿。
|
|
72
69
|
store.delete(key);
|
|
73
70
|
store.set(key, v);
|
|
71
|
+
touchedKeys.add(key);
|
|
74
72
|
dirty = true;
|
|
75
73
|
return v;
|
|
76
74
|
},
|
|
@@ -78,21 +76,51 @@ export function createCache(cacheDir) {
|
|
|
78
76
|
// 保留完整 ExecResult(含 turns / toolCalls):工具类 assertion (tool_called /
|
|
79
77
|
// tool_input_contains / tools_called)和 diagnostic 要看 trace,砍掉的话 cached
|
|
80
78
|
// rerun 进 grade() 时工具断言为空、diagnostic 没真实证据,跟 cold run 不一致。
|
|
79
|
+
const validationError = executorResultValidationError(value);
|
|
80
|
+
if (validationError)
|
|
81
|
+
throw new Error(`invalid executor cache entry: ${validationError}`);
|
|
82
|
+
const normalized = normalizeExecResultToolIdentities(value);
|
|
81
83
|
if (store.has(key))
|
|
82
84
|
store.delete(key);
|
|
83
|
-
store.set(key,
|
|
85
|
+
store.set(key, normalized);
|
|
86
|
+
pendingWrites.delete(key);
|
|
87
|
+
pendingWrites.set(key, normalized);
|
|
88
|
+
touchedKeys.delete(key);
|
|
84
89
|
evictUntilWithinCap(store, cap);
|
|
90
|
+
for (const pendingKey of pendingWrites.keys()) {
|
|
91
|
+
if (!store.has(pendingKey))
|
|
92
|
+
pendingWrites.delete(pendingKey);
|
|
93
|
+
}
|
|
85
94
|
dirty = true;
|
|
86
95
|
},
|
|
87
96
|
save() {
|
|
88
97
|
if (!dirty)
|
|
89
98
|
return;
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
99
|
+
withFileLock(`${filePath}.lock`, () => {
|
|
100
|
+
// Another eval process may have saved after this cache instance loaded.
|
|
101
|
+
// Merge only local writes and LRU touches into the latest disk state;
|
|
102
|
+
// replacing it with the whole stale in-memory snapshot loses entries.
|
|
103
|
+
const merged = readCacheStore(filePath, cap);
|
|
104
|
+
for (const key of touchedKeys) {
|
|
105
|
+
const value = merged.get(key);
|
|
106
|
+
if (!value)
|
|
107
|
+
continue;
|
|
108
|
+
merged.delete(key);
|
|
109
|
+
merged.set(key, value);
|
|
110
|
+
}
|
|
111
|
+
for (const [key, value] of pendingWrites) {
|
|
112
|
+
if (merged.has(key))
|
|
113
|
+
merged.delete(key);
|
|
114
|
+
merged.set(key, value);
|
|
115
|
+
}
|
|
116
|
+
evictUntilWithinCap(merged, cap);
|
|
117
|
+
writeCacheStore(filePath, merged);
|
|
118
|
+
store.clear();
|
|
119
|
+
for (const [key, value] of merged)
|
|
120
|
+
store.set(key, value);
|
|
121
|
+
}, { label: 'executor cache' });
|
|
122
|
+
pendingWrites.clear();
|
|
123
|
+
touchedKeys.clear();
|
|
96
124
|
dirty = false;
|
|
97
125
|
},
|
|
98
126
|
size() {
|
|
@@ -100,6 +128,36 @@ export function createCache(cacheDir) {
|
|
|
100
128
|
},
|
|
101
129
|
};
|
|
102
130
|
}
|
|
131
|
+
function readCacheStore(filePath, cap) {
|
|
132
|
+
// Map guarantees iteration order, which is the persisted LRU order.
|
|
133
|
+
const store = new Map();
|
|
134
|
+
if (!existsSync(filePath))
|
|
135
|
+
return store;
|
|
136
|
+
try {
|
|
137
|
+
const raw = JSON.parse(readFileSync(filePath, 'utf-8'));
|
|
138
|
+
if (!raw || typeof raw !== 'object' || Array.isArray(raw)) {
|
|
139
|
+
throw new Error('invalid executor cache root');
|
|
140
|
+
}
|
|
141
|
+
for (const [key, value] of Object.entries(raw)) {
|
|
142
|
+
const parsed = parseExecResult(value);
|
|
143
|
+
if (parsed)
|
|
144
|
+
store.set(key, parsed);
|
|
145
|
+
}
|
|
146
|
+
evictUntilWithinCap(store, cap);
|
|
147
|
+
}
|
|
148
|
+
catch {
|
|
149
|
+
// A corrupt cache is disposable evidence acceleration, never report data.
|
|
150
|
+
store.clear();
|
|
151
|
+
}
|
|
152
|
+
return store;
|
|
153
|
+
}
|
|
154
|
+
function writeCacheStore(filePath, store) {
|
|
155
|
+
// Keep the historical JSON object format. Property order carries LRU state.
|
|
156
|
+
const serialized = {};
|
|
157
|
+
for (const [key, value] of store)
|
|
158
|
+
serialized[key] = value;
|
|
159
|
+
writeJsonFileAtomic(filePath, serialized);
|
|
160
|
+
}
|
|
103
161
|
function evictUntilWithinCap(store, cap) {
|
|
104
162
|
if (!Number.isFinite(cap))
|
|
105
163
|
return;
|
|
@@ -122,7 +180,9 @@ mocksStrict,
|
|
|
122
180
|
effort,
|
|
123
181
|
/** artifact 内容指纹(整树 / 单文件哈)。本地 dir-skill 改 references/ 资产只动此值、不动 system,
|
|
124
182
|
* 不进 key 会让改资产后命中旧输出 → 静默污染。空(baseline / 无 skill)等价无指纹。 */
|
|
125
|
-
artifactContentHash
|
|
183
|
+
artifactContentHash,
|
|
184
|
+
/** External sample files that can alter executor output, currently mock return_file fixtures. */
|
|
185
|
+
sampleExecutionDependencyHash) {
|
|
126
186
|
// allowedSkills 序列化:undefined → "" / [] → "[]" / [...] → 排序后 JSON。
|
|
127
187
|
// 排序保证 ["a","b"] 和 ["b","a"] 命中同一缓存(语义等价)。
|
|
128
188
|
const isoStr = allowedSkills === undefined
|
|
@@ -138,8 +198,8 @@ artifactContentHash) {
|
|
|
138
198
|
// executor + runtime + effort 进 cache key:同 model 名走不同 executor 或同 executor
|
|
139
199
|
// 换 binary/SDK 版本时输出可能不同,旧 cache 不可复用。
|
|
140
200
|
const hash = createHash('sha256')
|
|
141
|
-
.update(`${model || ''}\n${system || ''}\n${prompt || ''}\n${cwd || ''}\n${isoStr}\n${executor || ''}\n${runtimeFingerprint || ''}\n${mockStr}\n${strictStr}\n${effortStr}\n${artifactContentHash || ''}`)
|
|
201
|
+
.update(`${model || ''}\n${system || ''}\n${prompt || ''}\n${cwd || ''}\n${isoStr}\n${executor || ''}\n${runtimeFingerprint || ''}\n${mockStr}\n${strictStr}\n${effortStr}\n${artifactContentHash || ''}\n${sampleExecutionDependencyHash || ''}`)
|
|
142
202
|
.digest('hex')
|
|
143
203
|
.slice(0, 16);
|
|
144
|
-
return `
|
|
204
|
+
return `v9:${hash}`;
|
|
145
205
|
}
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { ownRecordValue } from '../shared/record-count.js';
|
|
1
2
|
function stableStringify(value) {
|
|
2
3
|
if (value === null || typeof value !== 'object')
|
|
3
4
|
return JSON.stringify(value);
|
|
@@ -88,7 +89,7 @@ function runtimeMapKeys(meta) {
|
|
|
88
89
|
function reportExecutorRuntimeWarnings(report, warnings) {
|
|
89
90
|
const runtimes = report.meta.executorRuntimes;
|
|
90
91
|
if (runtimes && Object.keys(runtimes).length > 0) {
|
|
91
|
-
const missing = report.meta.variants.filter((variant) => !runtimes
|
|
92
|
+
const missing = report.meta.variants.filter((variant) => !ownRecordValue(runtimes, variant));
|
|
92
93
|
if (missing.length > 0) {
|
|
93
94
|
push(warnings, 'executor_runtime_missing', `报告缺少部分 variant 的 executor runtime 指纹: ${missing.join(', ')}。`, `Report is missing executor runtime fingerprints for variants: ${missing.join(', ')}.`);
|
|
94
95
|
}
|
|
@@ -116,12 +117,14 @@ function countSampleHashMismatches(a, b) {
|
|
|
116
117
|
let common = 0;
|
|
117
118
|
const ids = new Set([...Object.keys(ah), ...Object.keys(bh)]);
|
|
118
119
|
for (const id of ids) {
|
|
119
|
-
|
|
120
|
+
const aHash = ownRecordValue(ah, id);
|
|
121
|
+
const bHash = ownRecordValue(bh, id);
|
|
122
|
+
if (aHash == null || bHash == null) {
|
|
120
123
|
missing++;
|
|
121
124
|
continue;
|
|
122
125
|
}
|
|
123
126
|
common++;
|
|
124
|
-
if (
|
|
127
|
+
if (aHash !== bHash)
|
|
125
128
|
mismatched++;
|
|
126
129
|
}
|
|
127
130
|
return { mismatched, missing, common };
|
|
@@ -175,13 +178,14 @@ export function crossReportComparabilityWarnings(before, after) {
|
|
|
175
178
|
const hasExecutorRuntimeMap = bExecutorRuntimeKeys.length > 0 || aExecutorRuntimeKeys.length > 0;
|
|
176
179
|
if (hasExecutorRuntimeMap) {
|
|
177
180
|
const keys = sorted([...new Set([...b.variants, ...a.variants, ...bExecutorRuntimeKeys, ...aExecutorRuntimeKeys])]);
|
|
178
|
-
const missing = keys.filter((key) => !b.executorRuntimes
|
|
181
|
+
const missing = keys.filter((key) => !ownRecordValue(b.executorRuntimes ?? {}, key)
|
|
182
|
+
|| !ownRecordValue(a.executorRuntimes ?? {}, key));
|
|
179
183
|
if (missing.length > 0) {
|
|
180
184
|
push(warnings, 'executor_runtime_missing', `至少一份报告缺少 per-variant executor runtime 指纹: ${missing.join(', ')}。`, `At least one report is missing per-variant executor runtime fingerprints: ${missing.join(', ')}.`);
|
|
181
185
|
}
|
|
182
186
|
for (const key of keys) {
|
|
183
|
-
const beforeRuntime = b.executorRuntimes
|
|
184
|
-
const afterRuntime = a.executorRuntimes
|
|
187
|
+
const beforeRuntime = ownRecordValue(b.executorRuntimes ?? {}, key);
|
|
188
|
+
const afterRuntime = ownRecordValue(a.executorRuntimes ?? {}, key);
|
|
185
189
|
if (beforeRuntime?.fingerprint && afterRuntime?.fingerprint && beforeRuntime.fingerprint !== afterRuntime.fingerprint) {
|
|
186
190
|
push(warnings, 'executor_runtime_mismatch', `variant ${key} executor runtime 指纹不同: ${runtimeLabel(beforeRuntime)} → ${runtimeLabel(afterRuntime)}。`, `Variant ${key} executor runtime fingerprint changed: ${runtimeLabel(beforeRuntime)} → ${runtimeLabel(afterRuntime)}.`);
|
|
187
191
|
}
|
|
@@ -49,7 +49,8 @@ export declare function executeTasks({ tasks, executor, executorName, model, noJ
|
|
|
49
49
|
skipped: number;
|
|
50
50
|
budgetExhausted: boolean;
|
|
51
51
|
}>;
|
|
52
|
-
export declare function
|
|
52
|
+
export declare function preflightRuntimeLabel(executorName: string, model: string): string;
|
|
53
|
+
export declare function preflight(executor: ExecutorFn, model: string, timeoutMs?: number, label?: string): Promise<void>;
|
|
53
54
|
/**
|
|
54
55
|
* Preflight every unique `(executor, model)` judge in `judgeModels`. Used by the
|
|
55
56
|
* eval pipeline to fail fast when any ensemble member is misconfigured (404 model,
|