oh-my-knowledge 0.48.0 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -18
- package/README.zh.md +55 -23
- package/dist/analysis/coverage-analyzer.d.ts +1 -0
- package/dist/analysis/coverage-analyzer.js +125 -62
- package/dist/analysis/failure-clusterer.js +2 -1
- package/dist/analysis/gap-analyzer.d.ts +2 -2
- package/dist/analysis/gap-analyzer.js +13 -3
- package/dist/analysis/hedging-classifier.d.ts +2 -2
- package/dist/analysis/hedging-classifier.js +3 -4
- package/dist/analysis/report-diagnostics.js +9 -7
- package/dist/analysis/sample-diagnostics.js +6 -6
- package/dist/artifact-graph/doctor.js +15 -7
- package/dist/assets/agent-skills/omk/SKILL.md +27 -7
- package/dist/assets/agent-skills/omk/references/commands.md +18 -17
- package/dist/authoring/evolver.d.ts +10 -6
- package/dist/authoring/evolver.js +496 -83
- package/dist/authoring/generator.d.ts +3 -3
- package/dist/authoring/generator.js +5 -10
- package/dist/authoring/sample-fixer.d.ts +8 -6
- package/dist/authoring/sample-fixer.js +76 -5
- package/dist/cli/commands/doctor.js +31 -14
- package/dist/cli/commands/eval/index.d.ts +3 -0
- package/dist/cli/commands/eval/index.js +163 -17
- package/dist/cli/commands/evolve.d.ts +4 -4
- package/dist/cli/commands/evolve.js +27 -13
- package/dist/cli/commands/init.js +16 -3
- package/dist/cli/commands/observe/inbox.js +28 -21
- package/dist/cli/commands/observe/index.js +20 -11
- package/dist/cli/commands/observe/ingest.d.ts +3 -0
- package/dist/cli/commands/observe/ingest.js +30 -2
- package/dist/cli/commands/sample.d.ts +6 -3
- package/dist/cli/commands/sample.js +72 -68
- package/dist/cli/lib/codex-model-hint.d.ts +9 -0
- package/dist/cli/lib/codex-model-hint.js +45 -0
- package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
- package/dist/cli/lib/generation-failure-hint.js +61 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +4 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +38 -6
- package/dist/cli/lib/i18n-dict/help.js +6 -6
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +13 -9
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +34 -2
- package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
- package/dist/cli/lib/llm-failure-classifier.js +8 -0
- package/dist/cli/lib/parse-run-config.d.ts +6 -5
- package/dist/cli/lib/parse-run-config.js +16 -9
- package/dist/cli/lib/runtime-defaults.d.ts +21 -0
- package/dist/cli/lib/runtime-defaults.js +79 -0
- package/dist/diagnosis/observe-mapper.js +14 -15
- package/dist/diagnosis/observe-producer.js +3 -1
- package/dist/diagnosis/studio-projection.js +14 -7
- package/dist/diagnosis/types.d.ts +2 -0
- package/dist/diagnosis/types.js +12 -0
- package/dist/doctor/endpoint-rule.js +2 -1
- package/dist/eval-core/artifact-file-names.js +18 -1
- package/dist/eval-core/artifact-index.d.ts +7 -11
- package/dist/eval-core/artifact-index.js +139 -80
- package/dist/eval-core/cache.d.ts +12 -3
- package/dist/eval-core/cache.js +89 -29
- package/dist/eval-core/comparability.js +10 -6
- package/dist/eval-core/evaluation-execution.d.ts +2 -1
- package/dist/eval-core/evaluation-execution.js +122 -37
- package/dist/eval-core/evaluation-job.d.ts +4 -1
- package/dist/eval-core/evaluation-job.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +15 -13
- package/dist/eval-core/evaluation-reporting.js +54 -52
- package/dist/eval-core/execution-strategy.d.ts +2 -0
- package/dist/eval-core/execution-strategy.js +11 -9
- package/dist/eval-core/fact-checker.js +15 -7
- package/dist/eval-core/holdout.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +2 -2
- package/dist/eval-core/mock-hook.cjs +23 -6
- package/dist/eval-core/mocks-runtime.js +30 -8
- package/dist/eval-core/report-document.d.ts +12 -0
- package/dist/eval-core/report-document.js +1151 -0
- package/dist/eval-core/report-extensions.d.ts +4 -0
- package/dist/eval-core/report-extensions.js +500 -0
- package/dist/eval-core/report-file-migration.js +7 -2
- package/dist/eval-core/resume-compatibility.d.ts +31 -0
- package/dist/eval-core/resume-compatibility.js +141 -0
- package/dist/eval-core/sample-fingerprint.d.ts +12 -0
- package/dist/eval-core/sample-fingerprint.js +193 -0
- package/dist/eval-core/schema.js +86 -31
- package/dist/eval-core/verdict.d.ts +8 -4
- package/dist/eval-core/verdict.js +24 -10
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
- package/dist/eval-workflows/evaluation-pipeline.js +12 -7
- package/dist/eval-workflows/run-evaluation.d.ts +9 -7
- package/dist/eval-workflows/run-evaluation.js +79 -51
- package/dist/executors/anthropic-api.js +65 -9
- package/dist/executors/claude-cli.js +16 -79
- package/dist/executors/claude-protocol.d.ts +28 -0
- package/dist/executors/claude-protocol.js +180 -0
- package/dist/executors/claude-sdk-trace.js +56 -28
- package/dist/executors/claude-sdk.d.ts +1 -0
- package/dist/executors/claude-sdk.js +39 -93
- package/dist/executors/codex-cli-trace.js +166 -31
- package/dist/executors/codex-cli.d.ts +6 -8
- package/dist/executors/codex-cli.js +49 -151
- package/dist/executors/codex-protocol.d.ts +24 -0
- package/dist/executors/codex-protocol.js +234 -0
- package/dist/executors/codex-sdk.js +68 -120
- package/dist/executors/gemini.js +88 -13
- package/dist/executors/index.d.ts +2 -3
- package/dist/executors/index.js +5 -3
- package/dist/executors/openai-api.js +70 -9
- package/dist/executors/runtime-fingerprint.js +88 -11
- package/dist/executors/script-command.d.ts +8 -0
- package/dist/executors/script-command.js +87 -0
- package/dist/executors/script.js +202 -29
- package/dist/executors/shared.d.ts +35 -3
- package/dist/executors/shared.js +113 -15
- package/dist/grading/assertions.d.ts +1 -1
- package/dist/grading/assertions.js +19 -9
- package/dist/grading/diagnostic.d.ts +9 -2
- package/dist/grading/diagnostic.js +25 -2
- package/dist/grading/index.js +10 -4
- package/dist/grading/judge.js +19 -6
- package/dist/grading/layered-scores.d.ts +2 -3
- package/dist/grading/layered-scores.js +2 -3
- package/dist/inputs/load-samples.d.ts +1 -2
- package/dist/inputs/load-samples.js +23 -1
- package/dist/inputs/mcp-resolver.js +6 -3
- package/dist/inputs/sample-document.d.ts +11 -0
- package/dist/inputs/sample-document.js +96 -0
- package/dist/managed/evidence.d.ts +1 -0
- package/dist/managed/evidence.js +1 -1
- package/dist/managed/store.js +200 -91
- package/dist/observability/codex-trace-adapter.d.ts +5 -0
- package/dist/observability/codex-trace-adapter.js +850 -0
- package/dist/observability/experience.d.ts +32 -6
- package/dist/observability/experience.js +2695 -459
- package/dist/observability/feedback-matchers.js +16 -1
- package/dist/observability/inbox-view-model.d.ts +1 -1
- package/dist/observability/inbox-view-model.js +19 -14
- package/dist/observability/inbox.d.ts +7 -1
- package/dist/observability/inbox.js +632 -124
- package/dist/observability/problem-patterns.js +2 -0
- package/dist/observability/review-state.d.ts +6 -0
- package/dist/observability/review-state.js +235 -63
- package/dist/observability/skill-chain-advisories.js +1 -1
- package/dist/observability/skill-chain.js +17 -4
- package/dist/observability/skill-health-analyzer.d.ts +32 -7
- package/dist/observability/skill-health-analyzer.js +194 -121
- package/dist/observability/skill-health-report.d.ts +10 -0
- package/dist/observability/skill-health-report.js +620 -0
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +1 -1
- package/dist/observability/soft-standards/index.js +1 -1
- package/dist/observability/soft-standards/llm-extractor.js +8 -10
- package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
- package/dist/observability/soft-standards/skill-standards-store.js +59 -18
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/observability/trace-adapter.d.ts +12 -7
- package/dist/observability/trace-adapter.js +11 -9
- package/dist/observability/trace-attribution.d.ts +13 -5
- package/dist/observability/trace-attribution.js +315 -21
- package/dist/observability/trace-ingestion.d.ts +9 -0
- package/dist/observability/trace-ingestion.js +80 -0
- package/dist/observability/trace-ir.d.ts +113 -0
- package/dist/observability/trace-ir.js +87 -0
- package/dist/observability/trace-segmenter.d.ts +19 -6
- package/dist/observability/trace-segmenter.js +377 -196
- package/dist/observability/trace-session-index.d.ts +19 -0
- package/dist/observability/trace-session-index.js +68 -0
- package/dist/observability/trace-source.d.ts +12 -4
- package/dist/observability/trace-source.js +939 -215
- package/dist/renderer/html-renderer.js +37 -6
- package/dist/renderer/icons.js +3 -0
- package/dist/renderer/observation-inbox-renderer.js +208 -90
- package/dist/renderer/skill-detail-renderer.js +452 -109
- package/dist/renderer/skill-health-renderer.js +69 -12
- package/dist/renderer/summary.js +28 -7
- package/dist/renderer/table.js +21 -4
- package/dist/renderer/test-view.d.ts +1 -0
- package/dist/renderer/test-view.js +44 -9
- package/dist/server/indexed-report-store.js +14 -18
- package/dist/server/job-store.js +64 -26
- package/dist/server/report-server.js +190 -78
- package/dist/server/report-store.js +57 -80
- package/dist/server/skill-index.js +143 -49
- package/dist/server/skill-insights.js +44 -5
- package/dist/shared/artifact-graph.d.ts +3 -0
- package/dist/shared/artifact-graph.js +224 -0
- package/dist/shared/assertion-types.d.ts +8 -0
- package/dist/shared/assertion-types.js +46 -0
- package/dist/shared/atomic-json.d.ts +8 -0
- package/dist/shared/atomic-json.js +33 -0
- package/dist/shared/diagnosis-schema.d.ts +9 -0
- package/dist/shared/diagnosis-schema.js +181 -0
- package/dist/shared/doctor-report.d.ts +3 -0
- package/dist/shared/doctor-report.js +103 -0
- package/dist/shared/evaluation-job.d.ts +6 -0
- package/dist/shared/evaluation-job.js +217 -0
- package/dist/shared/executor-result.d.ts +17 -0
- package/dist/shared/executor-result.js +221 -0
- package/dist/shared/file-lock.d.ts +12 -0
- package/dist/shared/file-lock.js +129 -0
- package/dist/shared/json-value.d.ts +5 -0
- package/dist/shared/json-value.js +36 -0
- package/dist/shared/keyed-mutex.d.ts +7 -0
- package/dist/shared/keyed-mutex.js +24 -0
- package/dist/shared/record-count.d.ts +8 -0
- package/dist/shared/record-count.js +43 -0
- package/dist/shared/sample-contract.d.ts +3 -0
- package/dist/shared/sample-contract.js +332 -0
- package/dist/shared/timestamp.d.ts +6 -0
- package/dist/shared/timestamp.js +64 -0
- package/dist/shared/token-usage.d.ts +19 -0
- package/dist/shared/token-usage.js +50 -0
- package/dist/shared/tool-call-status.d.ts +8 -0
- package/dist/shared/tool-call-status.js +28 -0
- package/dist/shared/tool-identity.d.ts +21 -0
- package/dist/shared/tool-identity.js +84 -0
- package/dist/shared/tool-search.js +73 -16
- package/dist/shared/trace-projection.d.ts +5 -0
- package/dist/shared/trace-projection.js +20 -0
- package/dist/shared/trace-source-kind.d.ts +3 -0
- package/dist/shared/trace-source-kind.js +12 -0
- package/dist/types/diagnosis.d.ts +2 -0
- package/dist/types/eval.d.ts +4 -0
- package/dist/types/executor.d.ts +32 -5
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/judge.d.ts +2 -0
- package/dist/types/observability.d.ts +116 -9
- package/dist/types/report.d.ts +58 -6
- package/dist/types/skill-index.d.ts +7 -0
- package/dist/types/trace.d.ts +2 -0
- package/dist/types/trace.js +1 -0
- package/package.json +9 -5
|
@@ -4,9 +4,23 @@ import { safeSliceForJson } from '../util/safe-slice.js';
|
|
|
4
4
|
import { grade } from '../grading/index.js';
|
|
5
5
|
import { checkFacts } from './fact-checker.js';
|
|
6
6
|
import { resolveExecutionStrategy } from './execution-strategy.js';
|
|
7
|
-
import { DEFAULT_CACHE_DIR } from './default-dirs.js';
|
|
7
|
+
import { DEFAULT_CACHE_DIR, DEFAULT_ISOLATED_CWD_DIR } from './default-dirs.js';
|
|
8
8
|
import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
|
|
9
|
-
import {
|
|
9
|
+
import { ownRecordValue, setOwnRecordValue, } from '../shared/record-count.js';
|
|
10
|
+
import { executorResultValidationError, normalizeExecResultToolIdentities, } from '../shared/executor-result.js';
|
|
11
|
+
import { hashSampleExecutionDependencies } from './sample-fingerprint.js';
|
|
12
|
+
import { resolveDiagnosticTarget } from '../grading/diagnostic.js';
|
|
13
|
+
import { dirname, join, resolve } from 'node:path';
|
|
14
|
+
import { mkdir, mkdtemp, rm } from 'node:fs/promises';
|
|
15
|
+
const PREFLIGHT_RUNTIME_LABEL_EXECUTORS = new Set([
|
|
16
|
+
'claude',
|
|
17
|
+
'claude-sdk',
|
|
18
|
+
'codex',
|
|
19
|
+
'codex-sdk',
|
|
20
|
+
'gemini',
|
|
21
|
+
'anthropic-api',
|
|
22
|
+
'openai-api',
|
|
23
|
+
]);
|
|
10
24
|
async function runWithConcurrency(tasks, concurrency, fn) {
|
|
11
25
|
let index = 0;
|
|
12
26
|
async function worker() {
|
|
@@ -28,7 +42,9 @@ function makeErrorResult(error) {
|
|
|
28
42
|
outputTokens: 0,
|
|
29
43
|
cacheReadTokens: 0,
|
|
30
44
|
cacheCreationTokens: 0,
|
|
45
|
+
tokenUsageReportedByExecutor: false,
|
|
31
46
|
costUSD: 0,
|
|
47
|
+
costReportedByExecutor: false,
|
|
32
48
|
stopReason: 'error',
|
|
33
49
|
numTurns: 0,
|
|
34
50
|
error: message,
|
|
@@ -37,6 +53,24 @@ function makeErrorResult(error) {
|
|
|
37
53
|
function sleep(ms) {
|
|
38
54
|
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
39
55
|
}
|
|
56
|
+
async function executeWithAttemptIsolation(executor, input, isolatedCwd) {
|
|
57
|
+
if (!isolatedCwd)
|
|
58
|
+
return executor(input);
|
|
59
|
+
await mkdir(DEFAULT_ISOLATED_CWD_DIR, { recursive: true });
|
|
60
|
+
const runtimeCwd = await mkdtemp(join(DEFAULT_ISOLATED_CWD_DIR, 'attempt-'));
|
|
61
|
+
try {
|
|
62
|
+
return await executor({ ...input, cwd: runtimeCwd });
|
|
63
|
+
}
|
|
64
|
+
finally {
|
|
65
|
+
try {
|
|
66
|
+
await rm(runtimeCwd, { recursive: true, force: true, maxRetries: 3, retryDelay: 50 });
|
|
67
|
+
}
|
|
68
|
+
catch (error) {
|
|
69
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
70
|
+
process.stderr.write(`[omk] 无法清理 baseline 隔离目录 ${runtimeCwd}:${message}\n`);
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
}
|
|
40
74
|
export async function executeTasks({ tasks, executor, executorName, model, noJudge, samplesPath, samplesBaseDir, concurrency, timeoutMs, noCache, verbose, onProgress, retry = 0, existingResults, judgeRepeat = 1, judgeModels, judgeExecutors, lengthDebias = true, budget, effort, noDiagnostic = false, }) {
|
|
41
75
|
const results = {};
|
|
42
76
|
let started = 0;
|
|
@@ -44,17 +78,26 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
|
|
|
44
78
|
let skipped = 0;
|
|
45
79
|
let totalCostUSD = 0;
|
|
46
80
|
let budgetExhausted = false;
|
|
81
|
+
if (!executorName && !noCache) {
|
|
82
|
+
throw new Error('executeTasks requires executorName when cache is enabled; '
|
|
83
|
+
+ 'an anonymous executor cannot have a safe cross-run cache identity');
|
|
84
|
+
}
|
|
85
|
+
const effectiveExecutorName = executorName ?? 'custom-executor';
|
|
47
86
|
// Seed results from previous run (--resume)
|
|
48
87
|
if (existingResults) {
|
|
49
88
|
for (const [sampleId, variants] of Object.entries(existingResults)) {
|
|
50
|
-
results
|
|
89
|
+
setOwnRecordValue(results, sampleId, { ...variants });
|
|
90
|
+
for (const result of Object.values(variants)) {
|
|
91
|
+
if (result.ok)
|
|
92
|
+
totalCostUSD += result.costUSD;
|
|
93
|
+
}
|
|
51
94
|
}
|
|
52
95
|
}
|
|
53
96
|
const cacheDir = DEFAULT_CACHE_DIR;
|
|
54
97
|
const cache = noCache ? null : createCache(cacheDir);
|
|
55
98
|
async function executeTask(task) {
|
|
56
99
|
// Skip if already have a successful result (--resume)
|
|
57
|
-
if (existingResults
|
|
100
|
+
if (ownRecordValue(ownRecordValue(existingResults ?? {}, task.sample_id) ?? {}, task.variant)?.ok) {
|
|
58
101
|
skipped++;
|
|
59
102
|
started++;
|
|
60
103
|
completed++;
|
|
@@ -76,7 +119,6 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
|
|
|
76
119
|
const total = tasks.length;
|
|
77
120
|
onProgress?.({ phase: 'start', completed: idx, total, sample_id: task.sample_id, variant: task.variant });
|
|
78
121
|
const executionPlan = resolveExecutionStrategy(task, model, timeoutMs, verbose, effort, samplesBaseDir);
|
|
79
|
-
const effectiveExecutorName = executorName || 'claude';
|
|
80
122
|
const executorRuntime = getExecutorRuntimeFingerprint(effectiveExecutorName, model, {
|
|
81
123
|
skillDir: executionPlan.input.skillDir,
|
|
82
124
|
});
|
|
@@ -88,7 +130,7 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
|
|
|
88
130
|
const key = cacheKey(model, executionPlan.cacheSystem, executionPlan.input.prompt, executionPlan.input.cwd, task.artifact.allowedSkills, effectiveExecutorName, executorRuntime.fingerprint, executionPlan.input.mocks, executionPlan.input.mocksStrict, effort,
|
|
89
131
|
// artifact 内容指纹进 key:本地 dir-skill 改 references/ 资产只动 contentHash,system 不变,
|
|
90
132
|
// 不进 key 会命中旧输出贴到新 artifactHashes(静默污染)。
|
|
91
|
-
task.artifact.contentHash);
|
|
133
|
+
task.artifact.contentHash, hashSampleExecutionDependencies(task._sample, samplesBaseDir));
|
|
92
134
|
const cached = cache?.get(key);
|
|
93
135
|
const execStart = Date.now();
|
|
94
136
|
if (cached) {
|
|
@@ -97,25 +139,60 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
|
|
|
97
139
|
else {
|
|
98
140
|
// Execute with retry on failure
|
|
99
141
|
const maxAttempts = 1 + Math.max(0, retry);
|
|
142
|
+
let attemptCostUSD = 0;
|
|
143
|
+
let attemptCostReported = true;
|
|
144
|
+
let attemptCount = 0;
|
|
100
145
|
for (let attempt = 1; attempt <= maxAttempts; attempt++) {
|
|
101
146
|
try {
|
|
102
|
-
execResult = await executor
|
|
147
|
+
execResult = await executeWithAttemptIsolation(executor, executionPlan.input, executionPlan.isolatedCwd === true);
|
|
103
148
|
}
|
|
104
149
|
catch (err) {
|
|
105
150
|
execResult = makeErrorResult(err);
|
|
106
151
|
}
|
|
107
|
-
|
|
152
|
+
const validationError = executorResultValidationError(execResult);
|
|
153
|
+
if (validationError) {
|
|
154
|
+
execResult = makeErrorResult(`executor returned invalid result: ${validationError}`);
|
|
155
|
+
}
|
|
156
|
+
else {
|
|
157
|
+
execResult = normalizeExecResultToolIdentities(execResult);
|
|
158
|
+
}
|
|
159
|
+
attemptCount += 1;
|
|
160
|
+
const nextAttemptCost = attemptCostUSD + execResult.costUSD;
|
|
161
|
+
if (Number.isFinite(nextAttemptCost)
|
|
162
|
+
&& nextAttemptCost >= 0
|
|
163
|
+
&& nextAttemptCost <= Number.MAX_SAFE_INTEGER) {
|
|
164
|
+
attemptCostUSD = nextAttemptCost;
|
|
165
|
+
}
|
|
166
|
+
else {
|
|
167
|
+
// Never turn overflow into a free execution. Saturation keeps persisted
|
|
168
|
+
// numbers valid and remains a conservative lower bound; the completeness
|
|
169
|
+
// flag tells reports that the exact amount is unavailable.
|
|
170
|
+
attemptCostUSD = Number.MAX_SAFE_INTEGER;
|
|
171
|
+
attemptCostReported = false;
|
|
172
|
+
}
|
|
173
|
+
if (execResult.costReportedByExecutor === false)
|
|
174
|
+
attemptCostReported = false;
|
|
175
|
+
const retryWouldExceedBudget = (budget?.perSampleUSD != null
|
|
176
|
+
&& attemptCostUSD > budget.perSampleUSD) || (budget?.totalUSD != null
|
|
177
|
+
&& totalCostUSD + attemptCostUSD > budget.totalUSD);
|
|
178
|
+
if (execResult.ok || attempt === maxAttempts || retryWouldExceedBudget)
|
|
108
179
|
break;
|
|
109
180
|
// Exponential backoff before retry
|
|
110
181
|
const backoffMs = Math.min(2 ** (attempt - 1) * 1000, 30000);
|
|
111
182
|
onProgress?.({ phase: 'retry', completed: idx, total, sample_id: task.sample_id, variant: task.variant, attempt, maxAttempts });
|
|
112
183
|
await sleep(backoffMs);
|
|
113
184
|
}
|
|
185
|
+
execResult = {
|
|
186
|
+
...execResult,
|
|
187
|
+
costUSD: attemptCostUSD,
|
|
188
|
+
...(attemptCostReported ? {} : { costReportedByExecutor: false }),
|
|
189
|
+
...(attemptCount > 1 ? { attemptCount } : {}),
|
|
190
|
+
};
|
|
114
191
|
if (cache && execResult.ok)
|
|
115
192
|
cache.set(key, execResult);
|
|
116
193
|
}
|
|
117
194
|
const execMs = Date.now() - execStart;
|
|
118
|
-
totalCostUSD
|
|
195
|
+
totalCostUSD = Math.min(Number.MAX_SAFE_INTEGER, totalCostUSD + execResult.costUSD);
|
|
119
196
|
if (verbose && onProgress) {
|
|
120
197
|
onProgress({
|
|
121
198
|
phase: 'exec_done',
|
|
@@ -184,35 +261,26 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
|
|
|
184
261
|
if (execResult.ok && execResult.output && task.cwd) {
|
|
185
262
|
factCheck = checkFacts(execResult.output, resolve(task.cwd));
|
|
186
263
|
}
|
|
187
|
-
|
|
188
|
-
results
|
|
264
|
+
const sampleResults = ownRecordValue(results, task.sample_id)
|
|
265
|
+
?? setOwnRecordValue(results, task.sample_id, {});
|
|
189
266
|
const variantResult = buildVariantResult(execResult, gradeResult, { execMs, gradeMs, factCheck });
|
|
190
267
|
// Diagnostic — 与 judge 完全独立的"哪错了 + skill 怎么改"诊断。
|
|
191
268
|
// 触发条件:noDiagnostic=false + 至少 1 条 assertion fail + sample 跑成功(有 fullOutput)。
|
|
192
269
|
//
|
|
193
|
-
// executor / model
|
|
194
|
-
//
|
|
195
|
-
// - 用户没配 claude judge(只配了 openai / gemini 等)时,沿用第一个 judge 的
|
|
196
|
-
// executor + 它对应的 model 名。硬写 'haiku' 会导致非 claude executor 拒绝
|
|
197
|
-
// (model not found),诊断整段挂掉。
|
|
198
|
-
// - 实在没有 judge executor 配置时回退主 executor(虽然不是 lean 也能跑)。
|
|
270
|
+
// executor / model 选择跟报告契约共用 resolveDiagnosticTarget:
|
|
271
|
+
// 跟随首位 judge;没有 judge 配置时才跟随主执行器。不得暗中偏爱某个 provider。
|
|
199
272
|
const failedDetails = (gradeResult?.assertions?.details || []).filter((d) => !d.passed);
|
|
200
273
|
const shouldDiagnose = !noDiagnostic && execResult.ok && failedDetails.length > 0;
|
|
201
274
|
if (shouldDiagnose) {
|
|
275
|
+
const diagnosticStart = Date.now();
|
|
202
276
|
try {
|
|
203
277
|
const { runDiagnostic } = await import('../grading/diagnostic.js');
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
// judgeModels 里配的 model(用户已经验证可用)。
|
|
211
|
-
let diagModel = 'haiku';
|
|
212
|
-
if (diagExecutorName && diagExecutorName !== 'claude') {
|
|
213
|
-
const judgeEntry = judgeModels.find((j) => j.executor === diagExecutorName);
|
|
214
|
-
if (judgeEntry)
|
|
215
|
-
diagModel = judgeEntry.model;
|
|
278
|
+
const diagnosticTarget = resolveDiagnosticTarget(judgeModels, effectiveExecutorName, model);
|
|
279
|
+
const diagExecutor = diagnosticTarget.executor === effectiveExecutorName
|
|
280
|
+
? executor
|
|
281
|
+
: ownRecordValue(judgeExecutors, diagnosticTarget.executor);
|
|
282
|
+
if (!diagExecutor) {
|
|
283
|
+
throw new Error(`diagnostic executor "${diagnosticTarget.executor}" is not registered`);
|
|
216
284
|
}
|
|
217
285
|
const diagnostic = await runDiagnostic({
|
|
218
286
|
sample: task._sample,
|
|
@@ -223,9 +291,12 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
|
|
|
223
291
|
fullOutput: execResult.output || undefined,
|
|
224
292
|
assertionDetails: gradeResult?.assertions?.details || [],
|
|
225
293
|
executor: diagExecutor,
|
|
226
|
-
model:
|
|
294
|
+
model: diagnosticTarget.model,
|
|
227
295
|
});
|
|
228
296
|
variantResult.diagnostic = diagnostic;
|
|
297
|
+
if (diagnostic.costReportedByExecutor === false) {
|
|
298
|
+
variantResult.judgeCostReportedByExecutor = false;
|
|
299
|
+
}
|
|
229
300
|
// diagnostic 成本三层对齐(reviewer PR#95 CR 2026-05-11 P2):
|
|
230
301
|
// - meta.totalCostUSD 累加(下面 totalCostUSD += 这一行)
|
|
231
302
|
// - variant summary 的 totalCostUSD / totalDiagnosticCostUSD 由 buildVariantSummary
|
|
@@ -240,6 +311,7 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
|
|
|
240
311
|
}
|
|
241
312
|
}
|
|
242
313
|
catch (err) {
|
|
314
|
+
variantResult.judgeCostReportedByExecutor = false;
|
|
243
315
|
// diagnostic 失败不影响主评测,降级成 minimal 错误对象
|
|
244
316
|
const msg = err instanceof Error ? err.message : String(err);
|
|
245
317
|
variantResult.diagnostic = {
|
|
@@ -252,6 +324,15 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
|
|
|
252
324
|
suggestion: { skill: '', sample: '', none: '' },
|
|
253
325
|
};
|
|
254
326
|
}
|
|
327
|
+
finally {
|
|
328
|
+
const diagnosticMs = Date.now() - diagnosticStart;
|
|
329
|
+
variantResult.timing = {
|
|
330
|
+
execMs,
|
|
331
|
+
gradeMs,
|
|
332
|
+
diagnosticMs,
|
|
333
|
+
totalMs: execMs + gradeMs + diagnosticMs,
|
|
334
|
+
};
|
|
335
|
+
}
|
|
255
336
|
}
|
|
256
337
|
// per-sample budget enforcement. If a sample's cost or latency
|
|
257
338
|
// exceeds the per-sample cap, the result is kept (so the user can see
|
|
@@ -260,11 +341,12 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
|
|
|
260
341
|
variantResult.ok = false;
|
|
261
342
|
variantResult.error = `budget overrun: per-sample cost $${variantResult.costUSD.toFixed(4)} > cap $${budget.perSampleUSD.toFixed(4)}`;
|
|
262
343
|
}
|
|
263
|
-
|
|
344
|
+
const sampleDurationMs = variantResult.timing?.totalMs ?? execMs + gradeMs;
|
|
345
|
+
if (budget?.perSampleMs != null && sampleDurationMs > budget.perSampleMs) {
|
|
264
346
|
variantResult.ok = false;
|
|
265
|
-
variantResult.error = `budget overrun: per-sample latency ${
|
|
347
|
+
variantResult.error = `budget overrun: per-sample latency ${sampleDurationMs}ms > cap ${budget.perSampleMs}ms`;
|
|
266
348
|
}
|
|
267
|
-
|
|
349
|
+
setOwnRecordValue(sampleResults, task.variant, variantResult);
|
|
268
350
|
completed++;
|
|
269
351
|
onProgress?.({
|
|
270
352
|
phase: 'done',
|
|
@@ -298,7 +380,10 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
|
|
|
298
380
|
}
|
|
299
381
|
return { results, totalCostUSD, skipped, budgetExhausted };
|
|
300
382
|
}
|
|
301
|
-
export
|
|
383
|
+
export function preflightRuntimeLabel(executorName, model) {
|
|
384
|
+
return PREFLIGHT_RUNTIME_LABEL_EXECUTORS.has(executorName) ? `${executorName}:${model}` : `custom:${model}`;
|
|
385
|
+
}
|
|
386
|
+
export async function preflight(executor, model, timeoutMs = 180000, label) {
|
|
302
387
|
const result = await executor({
|
|
303
388
|
model,
|
|
304
389
|
system: '',
|
|
@@ -307,7 +392,7 @@ export async function preflight(executor, model, timeoutMs = 180000) {
|
|
|
307
392
|
timeoutMs,
|
|
308
393
|
});
|
|
309
394
|
if (!result.ok) {
|
|
310
|
-
throw new Error(`preflight failed [${model}]: ${result.error}`);
|
|
395
|
+
throw new Error(`preflight failed [${label ?? model}]: ${result.error}`);
|
|
311
396
|
}
|
|
312
397
|
}
|
|
313
398
|
/**
|
|
@@ -329,10 +414,10 @@ export async function preflightAllJudges(judgeModels, judgeExecutors, timeoutMs)
|
|
|
329
414
|
if (seen.has(key))
|
|
330
415
|
continue;
|
|
331
416
|
seen.add(key);
|
|
332
|
-
const exec = judgeExecutors
|
|
417
|
+
const exec = ownRecordValue(judgeExecutors, jc.executor);
|
|
333
418
|
if (!exec) {
|
|
334
419
|
throw new Error(`preflight: no executor registered for "${jc.executor}" (judge "${key}"); pipeline must populate judgeExecutors before preflight`);
|
|
335
420
|
}
|
|
336
|
-
await preflight(exec, jc.model, timeoutMs);
|
|
421
|
+
await preflight(exec, jc.model, timeoutMs, preflightRuntimeLabel(jc.executor, jc.model));
|
|
337
422
|
}
|
|
338
423
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { Artifact, EvaluationErrorCategory, EvaluationJob, EvaluationRequest, EvaluationRun, JudgeConfig } from '../types/index.js';
|
|
2
|
-
export declare function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
|
|
2
|
+
export declare function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, retry, noDiagnostic, }: {
|
|
3
3
|
samplesPath: string;
|
|
4
4
|
skillDir: string;
|
|
5
5
|
artifacts: Artifact[];
|
|
@@ -22,7 +22,10 @@ export declare function buildEvaluationRequest({ samplesPath, skillDir, artifact
|
|
|
22
22
|
bootstrapSamples?: number;
|
|
23
23
|
lengthDebias?: boolean;
|
|
24
24
|
budget?: import('../types/index.js').EvalBudget;
|
|
25
|
+
strictBaseline?: boolean;
|
|
25
26
|
effort?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
|
|
27
|
+
retry?: number;
|
|
28
|
+
noDiagnostic?: boolean;
|
|
26
29
|
}): EvaluationRequest;
|
|
27
30
|
export declare function createEvaluationRun(runId: string, startedAt?: string): {
|
|
28
31
|
run: EvaluationRun;
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
function nowIso() {
|
|
2
2
|
return new Date().toISOString();
|
|
3
3
|
}
|
|
4
|
-
export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
|
|
4
|
+
export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, retry, noDiagnostic, }) {
|
|
5
5
|
return {
|
|
6
6
|
samplesPath,
|
|
7
7
|
skillDir,
|
|
@@ -25,7 +25,10 @@ export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model
|
|
|
25
25
|
bootstrapSamples,
|
|
26
26
|
lengthDebias,
|
|
27
27
|
budget,
|
|
28
|
+
strictBaseline,
|
|
28
29
|
...(effort ? { effort } : {}),
|
|
30
|
+
...(retry && retry > 0 ? { retry } : {}),
|
|
31
|
+
...(noDiagnostic ? { noDiagnostic: true } : {}),
|
|
29
32
|
};
|
|
30
33
|
}
|
|
31
34
|
export function createEvaluationRun(runId, startedAt = nowIso()) {
|
|
@@ -1,15 +1,19 @@
|
|
|
1
|
-
import
|
|
1
|
+
import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
|
|
2
|
+
import type { Artifact, Report, Sample, Task, VariantResult, GitInfo, EvaluationJob, EvaluationRequest, EvaluationRun, ReportDocument } from '../types/index.js';
|
|
2
3
|
export declare const DEFAULT_OUTPUT_DIR: string;
|
|
3
|
-
export declare const EVALUATION_REPORT_SCHEMA_VERSION =
|
|
4
|
+
export declare const EVALUATION_REPORT_SCHEMA_VERSION = 5;
|
|
4
5
|
export declare function hashString(str: string): string;
|
|
5
|
-
|
|
6
|
-
* Stable content hash of a sample. Hashes the prompt + assertions + dimensions/rubric
|
|
7
|
-
* (the parts that determine what's being measured). Two samples with the same hash
|
|
8
|
-
* across runs measure the same thing; mismatched hashes mean the sample changed.
|
|
9
|
-
*/
|
|
10
|
-
export declare function hashSample(sample: Sample): string;
|
|
6
|
+
export { hashSample } from './sample-fingerprint.js';
|
|
11
7
|
export declare function getCliVersion(): string;
|
|
12
8
|
export declare function getGitInfo(): GitInfo | null;
|
|
9
|
+
export declare function buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request, }: {
|
|
10
|
+
variants: string[];
|
|
11
|
+
model: string;
|
|
12
|
+
executorName: string;
|
|
13
|
+
tasks: Task[];
|
|
14
|
+
artifacts: Artifact[];
|
|
15
|
+
request?: Pick<EvaluationRequest, 'skillDir' | 'timeoutMs'>;
|
|
16
|
+
}): Record<string, ReturnType<typeof getExecutorRuntimeFingerprint>>;
|
|
13
17
|
interface AggregateReportOptions {
|
|
14
18
|
runId: string;
|
|
15
19
|
variants: string[];
|
|
@@ -18,6 +22,7 @@ interface AggregateReportOptions {
|
|
|
18
22
|
noJudge: boolean;
|
|
19
23
|
executorName: string;
|
|
20
24
|
samples: Sample[];
|
|
25
|
+
samplesBaseDir?: string;
|
|
21
26
|
tasks: Task[];
|
|
22
27
|
results: Record<string, Record<string, VariantResult>>;
|
|
23
28
|
totalCostUSD: number;
|
|
@@ -27,10 +32,8 @@ interface AggregateReportOptions {
|
|
|
27
32
|
job?: EvaluationJob;
|
|
28
33
|
layeredStats?: boolean;
|
|
29
34
|
}
|
|
30
|
-
export declare function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, }: AggregateReportOptions): Report;
|
|
31
|
-
export
|
|
32
|
-
id: string;
|
|
33
|
-
}
|
|
35
|
+
export declare function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, samplesBaseDir, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, }: AggregateReportOptions): Report;
|
|
36
|
+
export type PersistableReport = ReportDocument;
|
|
34
37
|
export declare function persistReport(report: PersistableReport, outputDir: string | null): string | null;
|
|
35
38
|
/**
|
|
36
39
|
* run id 的时间戳后缀 `YYYYMMDDTHHmmss-rand4`。
|
|
@@ -40,4 +43,3 @@ export declare function persistReport(report: PersistableReport, outputDir: stri
|
|
|
40
43
|
*/
|
|
41
44
|
export declare function runIdSuffix(): string;
|
|
42
45
|
export declare function generateRunId(variants: string[]): string;
|
|
43
|
-
export {};
|
|
@@ -1,17 +1,22 @@
|
|
|
1
|
-
import { readFileSync,
|
|
1
|
+
import { readFileSync, mkdirSync, existsSync } from 'node:fs';
|
|
2
2
|
import { join, dirname } from 'node:path';
|
|
3
3
|
import { execFileSync } from 'node:child_process';
|
|
4
4
|
import { createHash } from 'node:crypto';
|
|
5
5
|
import { fileURLToPath } from 'node:url';
|
|
6
6
|
import { DEFAULT_REPORTS_DIR } from './default-dirs.js';
|
|
7
7
|
import { indexReportWrite } from './artifact-index.js';
|
|
8
|
+
import { parseReportDocument } from './report-document.js';
|
|
8
9
|
import { randomRunToken, reportFilePath, runTimestamp } from './artifact-file-names.js';
|
|
9
10
|
import { persistEvalGraphSidecar } from '../artifact-graph/eval.js';
|
|
10
11
|
import { buildVariantSummary } from './schema.js';
|
|
11
12
|
import { buildVariantConfig, resolveExecutionStrategy } from './execution-strategy.js';
|
|
12
13
|
import { getJudgePromptHash } from '../grading/judge.js';
|
|
14
|
+
import { getDiagnosticPromptHash, resolveDiagnosticTarget, } from '../grading/diagnostic.js';
|
|
13
15
|
import { bootstrapMeanCI, bootstrapPairedDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES, } from './bootstrap.js';
|
|
14
16
|
import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
|
|
17
|
+
import { ownRecordValue, setOwnRecordValue, } from '../shared/record-count.js';
|
|
18
|
+
import { writeJsonFileAtomic } from '../shared/atomic-json.js';
|
|
19
|
+
import { hashSample } from './sample-fingerprint.js';
|
|
15
20
|
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
16
21
|
function findPackageJson(startDir) {
|
|
17
22
|
let dir = startDir;
|
|
@@ -26,39 +31,11 @@ function findPackageJson(startDir) {
|
|
|
26
31
|
const PKG = JSON.parse(readFileSync(findPackageJson(__dirname), 'utf-8'));
|
|
27
32
|
// 写报告的默认目录 = reports 单一来源(default-dirs)。保留 DEFAULT_OUTPUT_DIR 名给既有 16 处 import,值统一,杜绝写/读两端漂移。
|
|
28
33
|
export const DEFAULT_OUTPUT_DIR = DEFAULT_REPORTS_DIR;
|
|
29
|
-
export const EVALUATION_REPORT_SCHEMA_VERSION =
|
|
34
|
+
export const EVALUATION_REPORT_SCHEMA_VERSION = 5;
|
|
30
35
|
export function hashString(str) {
|
|
31
36
|
return createHash('sha256').update(str).digest('hex').slice(0, 12);
|
|
32
37
|
}
|
|
33
|
-
|
|
34
|
-
* Canonical (key-sorted, recursive) JSON serialization. Required for cross-run hash
|
|
35
|
-
* stability — JS object key iteration order is implementation-defined for objects
|
|
36
|
-
* built by spread / Object.assign / yaml.parse, so naive JSON.stringify can produce
|
|
37
|
-
* different bytes for the "same" sample on different runs.
|
|
38
|
-
*/
|
|
39
|
-
function canonicalStringify(value) {
|
|
40
|
-
if (value === null || typeof value !== 'object')
|
|
41
|
-
return JSON.stringify(value);
|
|
42
|
-
if (Array.isArray(value))
|
|
43
|
-
return '[' + value.map(canonicalStringify).join(',') + ']';
|
|
44
|
-
const entries = Object.keys(value).sort();
|
|
45
|
-
return '{' + entries.map((k) => JSON.stringify(k) + ':' + canonicalStringify(value[k])).join(',') + '}';
|
|
46
|
-
}
|
|
47
|
-
/**
|
|
48
|
-
* Stable content hash of a sample. Hashes the prompt + assertions + dimensions/rubric
|
|
49
|
-
* (the parts that determine what's being measured). Two samples with the same hash
|
|
50
|
-
* across runs measure the same thing; mismatched hashes mean the sample changed.
|
|
51
|
-
*/
|
|
52
|
-
export function hashSample(sample) {
|
|
53
|
-
const stableForm = canonicalStringify({
|
|
54
|
-
prompt: sample.prompt,
|
|
55
|
-
rubric: sample.rubric ?? null,
|
|
56
|
-
dimensions: sample.dimensions ?? null,
|
|
57
|
-
assertions: sample.assertions ?? null,
|
|
58
|
-
schema: sample.schema ?? null,
|
|
59
|
-
});
|
|
60
|
-
return hashString(stableForm);
|
|
61
|
-
}
|
|
38
|
+
export { hashSample } from './sample-fingerprint.js';
|
|
62
39
|
export function getCliVersion() {
|
|
63
40
|
return PKG.version;
|
|
64
41
|
}
|
|
@@ -87,18 +64,18 @@ function commonRuntime(runtimes) {
|
|
|
87
64
|
function representativeRuntime(runtimes) {
|
|
88
65
|
return Object.values(runtimes)[0];
|
|
89
66
|
}
|
|
90
|
-
function buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request, }) {
|
|
67
|
+
export function buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request, }) {
|
|
91
68
|
const runtimes = {};
|
|
92
69
|
for (const task of tasks) {
|
|
93
|
-
if (runtimes
|
|
70
|
+
if (ownRecordValue(runtimes, task.variant))
|
|
94
71
|
continue;
|
|
95
72
|
const executionPlan = resolveExecutionStrategy(task, model, request?.timeoutMs, false);
|
|
96
|
-
runtimes
|
|
73
|
+
setOwnRecordValue(runtimes, task.variant, getExecutorRuntimeFingerprint(executorName, model, {
|
|
97
74
|
skillDir: executionPlan.input.skillDir,
|
|
98
|
-
});
|
|
75
|
+
}));
|
|
99
76
|
}
|
|
100
77
|
for (const variant of variants) {
|
|
101
|
-
if (runtimes
|
|
78
|
+
if (ownRecordValue(runtimes, variant))
|
|
102
79
|
continue;
|
|
103
80
|
const artifact = artifacts.find((a) => a.name === variant);
|
|
104
81
|
// 与主路径 extractSkillDir 一致:dir-skill 优先隔离副本 execRoot(副本无 node_modules、PATH 不污染);
|
|
@@ -109,17 +86,19 @@ function buildExecutorRuntimesByVariant({ variants, model, executorName, tasks,
|
|
|
109
86
|
: artifact?.locator
|
|
110
87
|
? dirname(artifact.locator)
|
|
111
88
|
: request?.skillDir);
|
|
112
|
-
runtimes
|
|
89
|
+
setOwnRecordValue(runtimes, variant, getExecutorRuntimeFingerprint(executorName, model, {
|
|
113
90
|
skillDir: fallbackSkillDir,
|
|
114
|
-
});
|
|
91
|
+
}));
|
|
115
92
|
}
|
|
116
93
|
return runtimes;
|
|
117
94
|
}
|
|
118
|
-
export function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, }) {
|
|
95
|
+
export function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, samplesBaseDir, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, }) {
|
|
119
96
|
const summary = {};
|
|
120
97
|
for (const variant of variants) {
|
|
121
|
-
const entries = Object.values(results)
|
|
122
|
-
|
|
98
|
+
const entries = Object.values(results)
|
|
99
|
+
.map((result) => ownRecordValue(result, variant))
|
|
100
|
+
.filter((entry) => Boolean(entry));
|
|
101
|
+
setOwnRecordValue(summary, variant, buildVariantSummary(entries));
|
|
123
102
|
}
|
|
124
103
|
// Bootstrap CI (per-variant mean) when --bootstrap requested. Adds bootstrapCI to
|
|
125
104
|
// each VariantSummary; legacy t-interval (in summary's other fields) is preserved.
|
|
@@ -131,7 +110,9 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
|
|
|
131
110
|
// 当且仅当该样本**无任何可测层**(真·缺测,如纯评委样本且评委失败)。故 `> 0` 过滤精确剔除非测量、
|
|
132
111
|
// 绝不丢"低分内容"(评委失败已在上游当缺测,不会以 0 进 composite)。下同(control / treatment)。
|
|
133
112
|
for (const variant of variants) {
|
|
134
|
-
const entries = Object.values(results)
|
|
113
|
+
const entries = Object.values(results)
|
|
114
|
+
.map((r) => ownRecordValue(r, variant))
|
|
115
|
+
.filter((entry) => Boolean(entry));
|
|
135
116
|
const compositeScores = entries
|
|
136
117
|
.filter((e) => typeof e.compositeScore === 'number' && e.compositeScore > 0)
|
|
137
118
|
.map((e) => e.compositeScore);
|
|
@@ -155,8 +136,8 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
|
|
|
155
136
|
const treatmentName = variants[i];
|
|
156
137
|
const pairs = [];
|
|
157
138
|
for (const r of sampleRecords) {
|
|
158
|
-
const c = r
|
|
159
|
-
const t = r
|
|
139
|
+
const c = ownRecordValue(r, controlName);
|
|
140
|
+
const t = ownRecordValue(r, treatmentName);
|
|
160
141
|
const a = c && typeof c.compositeScore === 'number' && c.compositeScore > 0 ? c.compositeScore : undefined;
|
|
161
142
|
const b = t && typeof t.compositeScore === 'number' && t.compositeScore > 0 ? t.compositeScore : undefined;
|
|
162
143
|
if (a !== undefined && b !== undefined)
|
|
@@ -188,7 +169,7 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
|
|
|
188
169
|
// contentHash 落在同一空间——证据可绑定的前提,也修掉「只哈 SKILL.md 正文、改资产指纹不变」的资产瞎。
|
|
189
170
|
// baseline / 无 skill 记 'no-skill'。
|
|
190
171
|
const artifactHashes = Object.fromEntries(artifacts.map((artifact) => [artifact.name, artifact.contentHash ?? 'no-skill']));
|
|
191
|
-
const sampleHashes = Object.fromEntries(samples.map((
|
|
172
|
+
const sampleHashes = Object.fromEntries(samples.map((sample) => [sample.sample_id, hashSample(sample, samplesBaseDir)]));
|
|
192
173
|
const judgeRepeat = request?.judgeRepeat && request.judgeRepeat > 1 ? request.judgeRepeat : undefined;
|
|
193
174
|
const runtimeOptions = { skillDir: request?.skillDir };
|
|
194
175
|
const executorRuntimes = buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request });
|
|
@@ -204,6 +185,17 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
|
|
|
204
185
|
model: jc.model,
|
|
205
186
|
...(noJudge ? {} : { runtime: getExecutorRuntimeFingerprint(jc.executor, jc.model, runtimeOptions) }),
|
|
206
187
|
}));
|
|
188
|
+
const diagnosticEnabled = request?.noDiagnostic !== true;
|
|
189
|
+
const diagnosticTarget = resolveDiagnosticTarget(requestJudges, executorName, model);
|
|
190
|
+
const diagnostic = diagnosticEnabled
|
|
191
|
+
? {
|
|
192
|
+
enabled: true,
|
|
193
|
+
executor: diagnosticTarget.executor,
|
|
194
|
+
model: diagnosticTarget.model,
|
|
195
|
+
runtime: getExecutorRuntimeFingerprint(diagnosticTarget.executor, diagnosticTarget.model),
|
|
196
|
+
promptHash: getDiagnosticPromptHash(),
|
|
197
|
+
}
|
|
198
|
+
: { enabled: false };
|
|
207
199
|
// length-debias is on by default; the request only sets it
|
|
208
200
|
// false when the user passed --no-debias-length. The judgePromptHash differs between
|
|
209
201
|
// the length-debias-on and -off prompt variants so readers can detect the divergence.
|
|
@@ -235,6 +227,7 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
|
|
|
235
227
|
artifactHashes,
|
|
236
228
|
sampleHashes,
|
|
237
229
|
...(noJudge ? {} : { judgePromptHash: getJudgePromptHash(lengthDebiasOn) }),
|
|
230
|
+
diagnostic,
|
|
238
231
|
executorRuntime,
|
|
239
232
|
executorRuntimes,
|
|
240
233
|
judgeModels: judgeModelsMeta,
|
|
@@ -259,16 +252,22 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
|
|
|
259
252
|
sample_id,
|
|
260
253
|
variants: variantData,
|
|
261
254
|
})),
|
|
262
|
-
//
|
|
263
|
-
//
|
|
264
|
-
//
|
|
255
|
+
// 用例设计快照,供单测视角与证据审计使用。执行/评分语义字段必须保留:
|
|
256
|
+
// cwd / environment / mocksStrict / allowedTools 等会改变真实构造,不能只留一个
|
|
257
|
+
// 不可解释的 sampleHash。纯扩展字段仍不盲目全选,控制报告体积。
|
|
265
258
|
sampleSnapshots: Object.fromEntries(samples.map((s) => [s.sample_id, {
|
|
266
259
|
sample_id: s.sample_id,
|
|
267
260
|
prompt: s.prompt,
|
|
261
|
+
...(s.cwd ? { cwd: s.cwd } : {}),
|
|
268
262
|
...(s.rubric ? { rubric: s.rubric } : {}),
|
|
269
263
|
...(s.context ? { context: s.context } : {}),
|
|
264
|
+
...(s.dimensions && Object.keys(s.dimensions).length > 0 ? { dimensions: s.dimensions } : {}),
|
|
270
265
|
...(s.assertions && s.assertions.length > 0 ? { assertions: s.assertions } : {}),
|
|
271
266
|
...(s.mocks && s.mocks.length > 0 ? { mocks: s.mocks } : {}),
|
|
267
|
+
...(s.mocksStrict !== undefined ? { mocksStrict: s.mocksStrict } : {}),
|
|
268
|
+
...(s.environment ? { environment: s.environment } : {}),
|
|
269
|
+
...(s.allowedTools && s.allowedTools.length > 0 ? { allowedTools: s.allowedTools } : {}),
|
|
270
|
+
...(s.expectedTools && s.expectedTools.length > 0 ? { expectedTools: s.expectedTools } : {}),
|
|
272
271
|
...(s.capability && s.capability.length > 0 ? { capability: s.capability } : {}),
|
|
273
272
|
...(s.difficulty ? { difficulty: s.difficulty } : {}),
|
|
274
273
|
...(s.construct ? { construct: s.construct } : {}),
|
|
@@ -279,7 +278,7 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
|
|
|
279
278
|
};
|
|
280
279
|
}
|
|
281
280
|
function isEvaluationReport(report) {
|
|
282
|
-
return report
|
|
281
|
+
return report.kind === 'evaluation';
|
|
283
282
|
}
|
|
284
283
|
function persistEvalGraphSidecarSafely(report, outputDir, sourcePath) {
|
|
285
284
|
if (!isEvaluationReport(report))
|
|
@@ -298,11 +297,14 @@ export function persistReport(report, outputDir) {
|
|
|
298
297
|
if (!existsSync(outputDir))
|
|
299
298
|
mkdirSync(outputDir, { recursive: true });
|
|
300
299
|
const filePath = reportFilePath(outputDir, report.id);
|
|
301
|
-
|
|
302
|
-
|
|
300
|
+
const parsed = parseReportDocument(report, report.id, report.id);
|
|
301
|
+
if (!parsed)
|
|
302
|
+
throw new Error('invalid report');
|
|
303
|
+
writeJsonFileAtomic(filePath, parsed);
|
|
304
|
+
persistEvalGraphSidecarSafely(parsed, outputDir, filePath);
|
|
303
305
|
// 产物发现索引:报告落项目本地后,best-effort 追加全局轻卡片,让 omk studio 跨项目聚合成机器级总览。
|
|
304
306
|
// 永不抛、永不阻断报告落盘(正文是 source of truth)。
|
|
305
|
-
indexReportWrite(
|
|
307
|
+
indexReportWrite(parsed, filePath, outputDir);
|
|
306
308
|
return filePath;
|
|
307
309
|
}
|
|
308
310
|
/**
|
|
@@ -3,6 +3,8 @@ export interface ExecutionPlan {
|
|
|
3
3
|
strategy: ExecutionStrategyKind;
|
|
4
4
|
cacheSystem: string;
|
|
5
5
|
input: ExecutorInput;
|
|
6
|
+
/** Replace the logical cwd with a fresh empty directory for each attempt. */
|
|
7
|
+
isolatedCwd?: boolean;
|
|
6
8
|
}
|
|
7
9
|
export declare function resolveArtifactExecutionStrategy(artifact: Artifact): ExecutionStrategyKind;
|
|
8
10
|
export declare function resolveExperimentType(artifact: Artifact): ExperimentType;
|