oh-my-knowledge 0.48.0 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -18
- package/README.zh.md +55 -23
- package/dist/analysis/coverage-analyzer.d.ts +1 -0
- package/dist/analysis/coverage-analyzer.js +125 -62
- package/dist/analysis/failure-clusterer.js +2 -1
- package/dist/analysis/gap-analyzer.d.ts +2 -2
- package/dist/analysis/gap-analyzer.js +13 -3
- package/dist/analysis/hedging-classifier.d.ts +2 -2
- package/dist/analysis/hedging-classifier.js +3 -4
- package/dist/analysis/report-diagnostics.js +9 -7
- package/dist/analysis/sample-diagnostics.js +6 -6
- package/dist/artifact-graph/doctor.js +15 -7
- package/dist/assets/agent-skills/omk/SKILL.md +27 -7
- package/dist/assets/agent-skills/omk/references/commands.md +18 -17
- package/dist/authoring/evolver.d.ts +10 -6
- package/dist/authoring/evolver.js +496 -83
- package/dist/authoring/generator.d.ts +3 -3
- package/dist/authoring/generator.js +5 -10
- package/dist/authoring/sample-fixer.d.ts +8 -6
- package/dist/authoring/sample-fixer.js +76 -5
- package/dist/cli/commands/doctor.js +31 -14
- package/dist/cli/commands/eval/index.d.ts +3 -0
- package/dist/cli/commands/eval/index.js +163 -17
- package/dist/cli/commands/evolve.d.ts +4 -4
- package/dist/cli/commands/evolve.js +27 -13
- package/dist/cli/commands/init.js +16 -3
- package/dist/cli/commands/observe/inbox.js +28 -21
- package/dist/cli/commands/observe/index.js +20 -11
- package/dist/cli/commands/observe/ingest.d.ts +3 -0
- package/dist/cli/commands/observe/ingest.js +30 -2
- package/dist/cli/commands/sample.d.ts +6 -3
- package/dist/cli/commands/sample.js +72 -68
- package/dist/cli/lib/codex-model-hint.d.ts +9 -0
- package/dist/cli/lib/codex-model-hint.js +45 -0
- package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
- package/dist/cli/lib/generation-failure-hint.js +61 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +4 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +38 -6
- package/dist/cli/lib/i18n-dict/help.js +6 -6
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +13 -9
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +34 -2
- package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
- package/dist/cli/lib/llm-failure-classifier.js +8 -0
- package/dist/cli/lib/parse-run-config.d.ts +6 -5
- package/dist/cli/lib/parse-run-config.js +16 -9
- package/dist/cli/lib/runtime-defaults.d.ts +21 -0
- package/dist/cli/lib/runtime-defaults.js +79 -0
- package/dist/diagnosis/observe-mapper.js +14 -15
- package/dist/diagnosis/observe-producer.js +3 -1
- package/dist/diagnosis/studio-projection.js +14 -7
- package/dist/diagnosis/types.d.ts +2 -0
- package/dist/diagnosis/types.js +12 -0
- package/dist/doctor/endpoint-rule.js +2 -1
- package/dist/eval-core/artifact-file-names.js +18 -1
- package/dist/eval-core/artifact-index.d.ts +7 -11
- package/dist/eval-core/artifact-index.js +139 -80
- package/dist/eval-core/cache.d.ts +12 -3
- package/dist/eval-core/cache.js +89 -29
- package/dist/eval-core/comparability.js +10 -6
- package/dist/eval-core/evaluation-execution.d.ts +2 -1
- package/dist/eval-core/evaluation-execution.js +122 -37
- package/dist/eval-core/evaluation-job.d.ts +4 -1
- package/dist/eval-core/evaluation-job.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +15 -13
- package/dist/eval-core/evaluation-reporting.js +54 -52
- package/dist/eval-core/execution-strategy.d.ts +2 -0
- package/dist/eval-core/execution-strategy.js +11 -9
- package/dist/eval-core/fact-checker.js +15 -7
- package/dist/eval-core/holdout.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +2 -2
- package/dist/eval-core/mock-hook.cjs +23 -6
- package/dist/eval-core/mocks-runtime.js +30 -8
- package/dist/eval-core/report-document.d.ts +12 -0
- package/dist/eval-core/report-document.js +1151 -0
- package/dist/eval-core/report-extensions.d.ts +4 -0
- package/dist/eval-core/report-extensions.js +500 -0
- package/dist/eval-core/report-file-migration.js +7 -2
- package/dist/eval-core/resume-compatibility.d.ts +31 -0
- package/dist/eval-core/resume-compatibility.js +141 -0
- package/dist/eval-core/sample-fingerprint.d.ts +12 -0
- package/dist/eval-core/sample-fingerprint.js +193 -0
- package/dist/eval-core/schema.js +86 -31
- package/dist/eval-core/verdict.d.ts +8 -4
- package/dist/eval-core/verdict.js +24 -10
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
- package/dist/eval-workflows/evaluation-pipeline.js +12 -7
- package/dist/eval-workflows/run-evaluation.d.ts +9 -7
- package/dist/eval-workflows/run-evaluation.js +79 -51
- package/dist/executors/anthropic-api.js +65 -9
- package/dist/executors/claude-cli.js +16 -79
- package/dist/executors/claude-protocol.d.ts +28 -0
- package/dist/executors/claude-protocol.js +180 -0
- package/dist/executors/claude-sdk-trace.js +56 -28
- package/dist/executors/claude-sdk.d.ts +1 -0
- package/dist/executors/claude-sdk.js +39 -93
- package/dist/executors/codex-cli-trace.js +166 -31
- package/dist/executors/codex-cli.d.ts +6 -8
- package/dist/executors/codex-cli.js +49 -151
- package/dist/executors/codex-protocol.d.ts +24 -0
- package/dist/executors/codex-protocol.js +234 -0
- package/dist/executors/codex-sdk.js +68 -120
- package/dist/executors/gemini.js +88 -13
- package/dist/executors/index.d.ts +2 -3
- package/dist/executors/index.js +5 -3
- package/dist/executors/openai-api.js +70 -9
- package/dist/executors/runtime-fingerprint.js +88 -11
- package/dist/executors/script-command.d.ts +8 -0
- package/dist/executors/script-command.js +87 -0
- package/dist/executors/script.js +202 -29
- package/dist/executors/shared.d.ts +35 -3
- package/dist/executors/shared.js +113 -15
- package/dist/grading/assertions.d.ts +1 -1
- package/dist/grading/assertions.js +19 -9
- package/dist/grading/diagnostic.d.ts +9 -2
- package/dist/grading/diagnostic.js +25 -2
- package/dist/grading/index.js +10 -4
- package/dist/grading/judge.js +19 -6
- package/dist/grading/layered-scores.d.ts +2 -3
- package/dist/grading/layered-scores.js +2 -3
- package/dist/inputs/load-samples.d.ts +1 -2
- package/dist/inputs/load-samples.js +23 -1
- package/dist/inputs/mcp-resolver.js +6 -3
- package/dist/inputs/sample-document.d.ts +11 -0
- package/dist/inputs/sample-document.js +96 -0
- package/dist/managed/evidence.d.ts +1 -0
- package/dist/managed/evidence.js +1 -1
- package/dist/managed/store.js +200 -91
- package/dist/observability/codex-trace-adapter.d.ts +5 -0
- package/dist/observability/codex-trace-adapter.js +850 -0
- package/dist/observability/experience.d.ts +32 -6
- package/dist/observability/experience.js +2695 -459
- package/dist/observability/feedback-matchers.js +16 -1
- package/dist/observability/inbox-view-model.d.ts +1 -1
- package/dist/observability/inbox-view-model.js +19 -14
- package/dist/observability/inbox.d.ts +7 -1
- package/dist/observability/inbox.js +632 -124
- package/dist/observability/problem-patterns.js +2 -0
- package/dist/observability/review-state.d.ts +6 -0
- package/dist/observability/review-state.js +235 -63
- package/dist/observability/skill-chain-advisories.js +1 -1
- package/dist/observability/skill-chain.js +17 -4
- package/dist/observability/skill-health-analyzer.d.ts +32 -7
- package/dist/observability/skill-health-analyzer.js +194 -121
- package/dist/observability/skill-health-report.d.ts +10 -0
- package/dist/observability/skill-health-report.js +620 -0
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +1 -1
- package/dist/observability/soft-standards/index.js +1 -1
- package/dist/observability/soft-standards/llm-extractor.js +8 -10
- package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
- package/dist/observability/soft-standards/skill-standards-store.js +59 -18
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/observability/trace-adapter.d.ts +12 -7
- package/dist/observability/trace-adapter.js +11 -9
- package/dist/observability/trace-attribution.d.ts +13 -5
- package/dist/observability/trace-attribution.js +315 -21
- package/dist/observability/trace-ingestion.d.ts +9 -0
- package/dist/observability/trace-ingestion.js +80 -0
- package/dist/observability/trace-ir.d.ts +113 -0
- package/dist/observability/trace-ir.js +87 -0
- package/dist/observability/trace-segmenter.d.ts +19 -6
- package/dist/observability/trace-segmenter.js +377 -196
- package/dist/observability/trace-session-index.d.ts +19 -0
- package/dist/observability/trace-session-index.js +68 -0
- package/dist/observability/trace-source.d.ts +12 -4
- package/dist/observability/trace-source.js +939 -215
- package/dist/renderer/html-renderer.js +37 -6
- package/dist/renderer/icons.js +3 -0
- package/dist/renderer/observation-inbox-renderer.js +208 -90
- package/dist/renderer/skill-detail-renderer.js +452 -109
- package/dist/renderer/skill-health-renderer.js +69 -12
- package/dist/renderer/summary.js +28 -7
- package/dist/renderer/table.js +21 -4
- package/dist/renderer/test-view.d.ts +1 -0
- package/dist/renderer/test-view.js +44 -9
- package/dist/server/indexed-report-store.js +14 -18
- package/dist/server/job-store.js +64 -26
- package/dist/server/report-server.js +190 -78
- package/dist/server/report-store.js +57 -80
- package/dist/server/skill-index.js +143 -49
- package/dist/server/skill-insights.js +44 -5
- package/dist/shared/artifact-graph.d.ts +3 -0
- package/dist/shared/artifact-graph.js +224 -0
- package/dist/shared/assertion-types.d.ts +8 -0
- package/dist/shared/assertion-types.js +46 -0
- package/dist/shared/atomic-json.d.ts +8 -0
- package/dist/shared/atomic-json.js +33 -0
- package/dist/shared/diagnosis-schema.d.ts +9 -0
- package/dist/shared/diagnosis-schema.js +181 -0
- package/dist/shared/doctor-report.d.ts +3 -0
- package/dist/shared/doctor-report.js +103 -0
- package/dist/shared/evaluation-job.d.ts +6 -0
- package/dist/shared/evaluation-job.js +217 -0
- package/dist/shared/executor-result.d.ts +17 -0
- package/dist/shared/executor-result.js +221 -0
- package/dist/shared/file-lock.d.ts +12 -0
- package/dist/shared/file-lock.js +129 -0
- package/dist/shared/json-value.d.ts +5 -0
- package/dist/shared/json-value.js +36 -0
- package/dist/shared/keyed-mutex.d.ts +7 -0
- package/dist/shared/keyed-mutex.js +24 -0
- package/dist/shared/record-count.d.ts +8 -0
- package/dist/shared/record-count.js +43 -0
- package/dist/shared/sample-contract.d.ts +3 -0
- package/dist/shared/sample-contract.js +332 -0
- package/dist/shared/timestamp.d.ts +6 -0
- package/dist/shared/timestamp.js +64 -0
- package/dist/shared/token-usage.d.ts +19 -0
- package/dist/shared/token-usage.js +50 -0
- package/dist/shared/tool-call-status.d.ts +8 -0
- package/dist/shared/tool-call-status.js +28 -0
- package/dist/shared/tool-identity.d.ts +21 -0
- package/dist/shared/tool-identity.js +84 -0
- package/dist/shared/tool-search.js +73 -16
- package/dist/shared/trace-projection.d.ts +5 -0
- package/dist/shared/trace-projection.js +20 -0
- package/dist/shared/trace-source-kind.d.ts +3 -0
- package/dist/shared/trace-source-kind.js +12 -0
- package/dist/types/diagnosis.d.ts +2 -0
- package/dist/types/eval.d.ts +4 -0
- package/dist/types/executor.d.ts +32 -5
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/judge.d.ts +2 -0
- package/dist/types/observability.d.ts +116 -9
- package/dist/types/report.d.ts +58 -6
- package/dist/types/skill-index.d.ts +7 -0
- package/dist/types/trace.d.ts +2 -0
- package/dist/types/trace.js +1 -0
- package/package.json +9 -5
|
@@ -1,19 +1,21 @@
|
|
|
1
1
|
import { resolve } from 'node:path';
|
|
2
2
|
import { DEFAULT_OUTPUT_DIR, persistReport } from '../eval-core/evaluation-reporting.js';
|
|
3
|
-
import { createExecutor
|
|
3
|
+
import { createExecutor } from '../executors/index.js';
|
|
4
4
|
import { discoverBatchSkills } from '../inputs/skill-loader.js';
|
|
5
5
|
import { confidenceInterval, tTest, effectSize } from '../eval-core/statistics.js';
|
|
6
6
|
import { executeBatchEvaluationRuns, buildBatchVariantSpecs } from './batch-evaluation-workflow.js';
|
|
7
7
|
import { buildDryRunTaskReport, prepareEvaluationRun, } from './evaluation-preparation.js';
|
|
8
8
|
import { executeEvaluationPipeline } from './evaluation-pipeline.js';
|
|
9
|
+
import { checkResumeCompatibility } from '../eval-core/resume-compatibility.js';
|
|
9
10
|
import { findSaturationPoint } from '../analysis/saturation.js';
|
|
10
11
|
import { bootstrapMeanCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
|
|
11
|
-
|
|
12
|
+
import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
|
|
13
|
+
export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
|
|
12
14
|
// Unified judgeModels → derive single-judge fields for downstream pipeline / grading
|
|
13
15
|
// (which still operate on string `judgeModel` + `judgeExecutorName` fields per call).
|
|
14
16
|
const effectiveJudgeModels = judgeModels && judgeModels.length > 0
|
|
15
17
|
? judgeModels
|
|
16
|
-
: [{ executor: executorName, model
|
|
18
|
+
: [{ executor: executorName, model }];
|
|
17
19
|
const judgeModel = effectiveJudgeModels[0].model;
|
|
18
20
|
const judgeExecutorName = effectiveJudgeModels[0].executor;
|
|
19
21
|
const { samples, artifacts: resolvedArtifacts, tasks, variantNames, requires, samplesBaseDir, samplesSourceFiles } = await prepareEvaluationRun({
|
|
@@ -66,8 +68,6 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
66
68
|
}
|
|
67
69
|
}
|
|
68
70
|
}
|
|
69
|
-
// --resume 时自动跳过 LLM 连通性检测(原 run 已经验过, 重跑是浪费 LLM 调用)
|
|
70
|
-
const effectiveSkipConnectivity = resume ? true : skipConnectivity;
|
|
71
71
|
if (dryRun) {
|
|
72
72
|
// Emit power warnings during dry-run too — this is exactly when users
|
|
73
73
|
// preview the run, the right moment to flag "you might be wasting it".
|
|
@@ -75,7 +75,10 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
75
75
|
for (const w of buildPowerWarnings(samples.length, repeat ?? 1, lang)) {
|
|
76
76
|
process.stderr.write(`${w}\n`);
|
|
77
77
|
}
|
|
78
|
-
for (const w of buildIsolationWarnings(resolvedArtifacts, strictBaseline
|
|
78
|
+
for (const w of buildIsolationWarnings(resolvedArtifacts, strictBaseline, {
|
|
79
|
+
executorName,
|
|
80
|
+
lang,
|
|
81
|
+
})) {
|
|
79
82
|
process.stderr.write(`${w}\n`);
|
|
80
83
|
}
|
|
81
84
|
return {
|
|
@@ -101,27 +104,47 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
101
104
|
const store = createOverlayReportStore(resolve(outputDir || DEFAULT_OUTPUT_DIR), globalReportsDir());
|
|
102
105
|
const existing = await store.get(resume);
|
|
103
106
|
if (existing?.kind === 'evaluation') {
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
107
|
+
const compatibility = checkResumeCompatibility(existing, {
|
|
108
|
+
variants: variantNames,
|
|
109
|
+
model,
|
|
110
|
+
executorName,
|
|
111
|
+
effort,
|
|
112
|
+
noJudge,
|
|
113
|
+
judgeModels: effectiveJudgeModels,
|
|
114
|
+
judgeRepeat,
|
|
115
|
+
lengthDebias,
|
|
116
|
+
budget,
|
|
117
|
+
timeoutMs,
|
|
118
|
+
retry,
|
|
119
|
+
noDiagnostic,
|
|
120
|
+
skillDir,
|
|
121
|
+
samples,
|
|
122
|
+
samplesBaseDir,
|
|
123
|
+
tasks,
|
|
124
|
+
artifacts: resolvedArtifacts,
|
|
125
|
+
});
|
|
126
|
+
if (compatibility.compatible) {
|
|
127
|
+
const sourceBySample = new Map(existing.results.map((entry) => [entry.sample_id, entry.variants]));
|
|
128
|
+
const resumedResults = Object.create(null);
|
|
129
|
+
for (const task of tasks) {
|
|
130
|
+
const result = sourceBySample.get(task.sample_id)?.[task.variant];
|
|
131
|
+
if (!result?.ok)
|
|
132
|
+
continue;
|
|
133
|
+
const variants = ownRecordValue(resumedResults, task.sample_id)
|
|
134
|
+
?? setOwnRecordValue(resumedResults, task.sample_id, {});
|
|
135
|
+
setOwnRecordValue(variants, task.variant, result);
|
|
136
|
+
}
|
|
137
|
+
existingResults = resumedResults;
|
|
138
|
+
const count = Object.values(resumedResults).reduce((sum, variants) => sum + Object.keys(variants).length, 0);
|
|
139
|
+
process.stderr.write(lang === 'zh'
|
|
140
|
+
? `\n已从报告 ${resume} 恢复 ${count} 条兼容的成功结果。\n`
|
|
141
|
+
: `\nResumed ${count} compatible successful result(s) from report ${resume}.\n`);
|
|
121
142
|
}
|
|
122
|
-
else
|
|
123
|
-
|
|
124
|
-
|
|
143
|
+
else {
|
|
144
|
+
const fields = compatibility.mismatches.join(', ');
|
|
145
|
+
process.stderr.write(lang === 'zh'
|
|
146
|
+
? `\n报告 ${resume} 与当前评测契约不兼容,将从头运行。差异字段:${fields}。\n`
|
|
147
|
+
: `\nReport ${resume} is incompatible with the current evaluation contract; starting from scratch. Mismatched fields: ${fields}.\n`);
|
|
125
148
|
}
|
|
126
149
|
}
|
|
127
150
|
else if (existing?.kind === 'batch-evaluation') {
|
|
@@ -131,6 +154,10 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
131
154
|
process.stderr.write(`\n⚠️ report ${resume} not found, starting from scratch\n`);
|
|
132
155
|
}
|
|
133
156
|
}
|
|
157
|
+
// 只有通过完整契约校验、真正接受恢复结果时,才能沿用原 run 的连通性结论。
|
|
158
|
+
const effectiveSkipConnectivity = existingResults !== undefined
|
|
159
|
+
? true
|
|
160
|
+
: skipConnectivity;
|
|
134
161
|
const executor = createExecutor(executorName);
|
|
135
162
|
const judgeExecutor = createExecutor(judgeExecutorName || executorName);
|
|
136
163
|
return executeEvaluationPipeline({
|
|
@@ -201,7 +228,7 @@ const LAYER_EXTRACTORS = {
|
|
|
201
228
|
};
|
|
202
229
|
function buildMetricStats(runs, variant, extractor) {
|
|
203
230
|
const scores = runs
|
|
204
|
-
.map((run) => extractor(run.summary
|
|
231
|
+
.map((run) => extractor(ownRecordValue(run.summary, variant)))
|
|
205
232
|
.filter((x) => typeof x === 'number');
|
|
206
233
|
if (scores.length === 0)
|
|
207
234
|
return null;
|
|
@@ -236,8 +263,8 @@ function buildSaturationData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAMPLES,
|
|
|
236
263
|
if (variants.length === 0)
|
|
237
264
|
return undefined;
|
|
238
265
|
// Per-variant: cumulative composite scores after each repeat.
|
|
239
|
-
const cumulativeByVariant =
|
|
240
|
-
const tracesByVariant =
|
|
266
|
+
const cumulativeByVariant = Object.create(null);
|
|
267
|
+
const tracesByVariant = Object.create(null);
|
|
241
268
|
const checkpointSampleCounts = [];
|
|
242
269
|
const acc = Object.fromEntries(variants.map((v) => [v, []]));
|
|
243
270
|
for (let runIdx = 0; runIdx < runs.length; runIdx++) {
|
|
@@ -245,47 +272,48 @@ function buildSaturationData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAMPLES,
|
|
|
245
272
|
for (const variant of variants) {
|
|
246
273
|
const newScores = [];
|
|
247
274
|
for (const entry of run.results ?? []) {
|
|
248
|
-
const v = entry.variants
|
|
275
|
+
const v = ownRecordValue(entry.variants, variant);
|
|
249
276
|
if (!v || typeof v.compositeScore !== 'number' || v.compositeScore <= 0)
|
|
250
277
|
continue;
|
|
251
278
|
newScores.push(v.compositeScore);
|
|
252
279
|
}
|
|
253
|
-
acc
|
|
280
|
+
setOwnRecordValue(acc, variant, (ownRecordValue(acc, variant) ?? []).concat(newScores));
|
|
254
281
|
}
|
|
255
282
|
// Snapshot cumulative state for this checkpoint.
|
|
256
|
-
const checkpointN = acc
|
|
283
|
+
const checkpointN = ownRecordValue(acc, variants[0])?.length ?? 0;
|
|
257
284
|
checkpointSampleCounts.push(checkpointN);
|
|
258
285
|
for (const variant of variants) {
|
|
259
|
-
|
|
260
|
-
cumulativeByVariant
|
|
261
|
-
|
|
286
|
+
const cumulative = ownRecordValue(cumulativeByVariant, variant)
|
|
287
|
+
?? setOwnRecordValue(cumulativeByVariant, variant, []);
|
|
288
|
+
cumulative.push([...(ownRecordValue(acc, variant) ?? [])]);
|
|
262
289
|
// Per-checkpoint trace: bootstrap CI on cumulative scores.
|
|
263
|
-
const
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
290
|
+
const scores = ownRecordValue(acc, variant) ?? [];
|
|
291
|
+
const ci = bootstrapMeanCI(scores, DEFAULT_BOOTSTRAP_ALPHA, bootstrapSamples, seed);
|
|
292
|
+
const traces = ownRecordValue(tracesByVariant, variant)
|
|
293
|
+
?? setOwnRecordValue(tracesByVariant, variant, []);
|
|
294
|
+
traces.push({
|
|
295
|
+
n: scores.length,
|
|
268
296
|
mean: ci.estimate,
|
|
269
297
|
ciLow: ci.low,
|
|
270
298
|
ciHigh: ci.high,
|
|
271
299
|
});
|
|
272
300
|
}
|
|
273
301
|
}
|
|
274
|
-
const verdicts =
|
|
302
|
+
const verdicts = Object.create(null);
|
|
275
303
|
if (runs.length >= 5) {
|
|
276
304
|
for (const variant of variants) {
|
|
277
|
-
const cumulative = cumulativeByVariant
|
|
305
|
+
const cumulative = ownRecordValue(cumulativeByVariant, variant);
|
|
278
306
|
if (!cumulative)
|
|
279
307
|
continue;
|
|
280
308
|
const r = findSaturationPoint(cumulative, 'bootstrap-ci-width', undefined, undefined, bootstrapSamples, seed);
|
|
281
|
-
verdicts
|
|
309
|
+
setOwnRecordValue(verdicts, variant, {
|
|
282
310
|
saturated: r.saturated,
|
|
283
311
|
atN: r.atN,
|
|
284
312
|
confidence: r.confidence,
|
|
285
313
|
method: r.method,
|
|
286
314
|
threshold: r.threshold,
|
|
287
315
|
reason: r.reason,
|
|
288
|
-
};
|
|
316
|
+
});
|
|
289
317
|
}
|
|
290
318
|
}
|
|
291
319
|
return {
|
|
@@ -299,7 +327,7 @@ export function buildVarianceData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAM
|
|
|
299
327
|
return null;
|
|
300
328
|
}
|
|
301
329
|
const variants = runs[0].meta.variants || [];
|
|
302
|
-
const perVariant =
|
|
330
|
+
const perVariant = Object.create(null);
|
|
303
331
|
for (const variant of variants) {
|
|
304
332
|
// Composite lives on the legacy flat fields.
|
|
305
333
|
const composite = buildMetricStats(runs, variant, COMPOSITE_EXTRACTOR);
|
|
@@ -319,17 +347,17 @@ export function buildVarianceData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAM
|
|
|
319
347
|
if (layerStats)
|
|
320
348
|
byLayer[key] = layerStats;
|
|
321
349
|
}
|
|
322
|
-
perVariant
|
|
350
|
+
setOwnRecordValue(perVariant, variant, {
|
|
323
351
|
...composite,
|
|
324
352
|
...(Object.keys(byMetric).length > 0 ? { byMetric } : {}),
|
|
325
353
|
...(Object.keys(byLayer).length > 0 ? { byLayer } : {}),
|
|
326
|
-
};
|
|
354
|
+
});
|
|
327
355
|
}
|
|
328
356
|
const comparisons = [];
|
|
329
357
|
for (let i = 0; i < variants.length; i++) {
|
|
330
358
|
for (let j = i + 1; j < variants.length; j++) {
|
|
331
|
-
const vA = perVariant
|
|
332
|
-
const vB = perVariant
|
|
359
|
+
const vA = ownRecordValue(perVariant, variants[i]);
|
|
360
|
+
const vB = ownRecordValue(perVariant, variants[j]);
|
|
333
361
|
if (!vA || !vB)
|
|
334
362
|
continue;
|
|
335
363
|
const compositeComp = buildComparisonMetric(vA.scores, vB.scores, vA.mean, vB.mean);
|
|
@@ -367,12 +395,12 @@ export function buildVarianceData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAM
|
|
|
367
395
|
...(saturation ? { saturation } : {}),
|
|
368
396
|
};
|
|
369
397
|
}
|
|
370
|
-
export async function runBatchEvaluation({ skillDir, model
|
|
398
|
+
export async function runBatchEvaluation({ skillDir, model, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, executorName, jobStore = null, persistJob = true, onProgress = null, onSkillProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, repeat, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache = false, strictBaseline, variantAllowedSkills, }) {
|
|
371
399
|
// Same unified judge derivation as runEvaluation (downstream pipeline / report build
|
|
372
400
|
// still uses single judgeModel + judgeExecutorName per call).
|
|
373
401
|
const effectiveJudgeModels = judgeModels && judgeModels.length > 0
|
|
374
402
|
? judgeModels
|
|
375
|
-
: [{ executor: executorName, model
|
|
403
|
+
: [{ executor: executorName, model }];
|
|
376
404
|
const judgeModel = effectiveJudgeModels[0].model;
|
|
377
405
|
const judgeExecutorName = effectiveJudgeModels[0].executor;
|
|
378
406
|
const skillEntries = discoverBatchSkills(resolve(skillDir));
|
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import { asErrorLike, DEFAULT_TIMEOUT_MS, errorMessage } from './shared.js';
|
|
1
|
+
import { asErrorLike, DEFAULT_TIMEOUT_MS, errorMessage, readJsonResponse, responseBodyPreview, } from './shared.js';
|
|
2
|
+
import { optionalTokenCount } from '../shared/token-usage.js';
|
|
2
3
|
export async function anthropicApiExecutor({ model, system, prompt, timeoutMs = DEFAULT_TIMEOUT_MS }) {
|
|
3
4
|
const apiKey = process.env.ANTHROPIC_API_KEY;
|
|
4
5
|
if (!apiKey)
|
|
@@ -19,17 +20,72 @@ export async function anthropicApiExecutor({ model, system, prompt, timeoutMs =
|
|
|
19
20
|
body: JSON.stringify(reqBody),
|
|
20
21
|
signal: AbortSignal.timeout(timeoutMs),
|
|
21
22
|
});
|
|
22
|
-
const data = await res
|
|
23
|
+
const { data, rawBody } = await readJsonResponse(res);
|
|
23
24
|
const durationMs = Date.now() - start;
|
|
24
25
|
if (!res.ok) {
|
|
25
|
-
|
|
26
|
+
const bodyPreview = responseBodyPreview(rawBody);
|
|
27
|
+
return { ok: false, error: data?.error?.message || `API error ${res.status}${bodyPreview ? `: ${bodyPreview}` : ''}`, durationMs, durationApiMs: 0, inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheCreationTokens: 0, tokenUsageReportedByExecutor: false, costUSD: 0, costReportedByExecutor: false, output: null, stopReason: 'error', numTurns: 0 };
|
|
28
|
+
}
|
|
29
|
+
if (!data) {
|
|
30
|
+
return { ok: false, error: 'Anthropic API returned an empty or non-JSON response', durationMs, durationApiMs: 0, inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheCreationTokens: 0, tokenUsageReportedByExecutor: false, costUSD: 0, costReportedByExecutor: false, output: null, stopReason: 'error', numTurns: 0 };
|
|
31
|
+
}
|
|
32
|
+
const usage = data.usage;
|
|
33
|
+
const inputTokens = optionalTokenCount(usage?.input_tokens);
|
|
34
|
+
const outputTokens = optionalTokenCount(usage?.output_tokens);
|
|
35
|
+
const cacheReadTokens = usage?.cache_read_input_tokens === undefined
|
|
36
|
+
? 0
|
|
37
|
+
: optionalTokenCount(usage.cache_read_input_tokens);
|
|
38
|
+
const cacheCreationTokens = usage?.cache_creation_input_tokens === undefined
|
|
39
|
+
? 0
|
|
40
|
+
: optionalTokenCount(usage.cache_creation_input_tokens);
|
|
41
|
+
if (inputTokens === undefined
|
|
42
|
+
|| outputTokens === undefined
|
|
43
|
+
|| cacheReadTokens === undefined
|
|
44
|
+
|| cacheCreationTokens === undefined) {
|
|
45
|
+
return {
|
|
46
|
+
ok: false,
|
|
47
|
+
error: 'Anthropic response contained missing or invalid token usage',
|
|
48
|
+
durationMs,
|
|
49
|
+
durationApiMs: 0,
|
|
50
|
+
inputTokens: 0,
|
|
51
|
+
outputTokens: 0,
|
|
52
|
+
cacheReadTokens: 0,
|
|
53
|
+
cacheCreationTokens: 0,
|
|
54
|
+
tokenUsageReportedByExecutor: false,
|
|
55
|
+
costUSD: 0,
|
|
56
|
+
costReportedByExecutor: false,
|
|
57
|
+
output: null,
|
|
58
|
+
stopReason: 'error',
|
|
59
|
+
numTurns: 1,
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
const output = data.content
|
|
63
|
+
?.filter((block) => block.type === undefined || block.type === 'text')
|
|
64
|
+
.map((block) => block.text || '')
|
|
65
|
+
.join('') ?? '';
|
|
66
|
+
if (!output.trim()) {
|
|
67
|
+
return {
|
|
68
|
+
ok: false,
|
|
69
|
+
error: 'Anthropic response did not contain assistant text',
|
|
70
|
+
durationMs,
|
|
71
|
+
durationApiMs: 0,
|
|
72
|
+
inputTokens,
|
|
73
|
+
outputTokens,
|
|
74
|
+
cacheReadTokens,
|
|
75
|
+
cacheCreationTokens,
|
|
76
|
+
costUSD: 0,
|
|
77
|
+
costReportedByExecutor: false,
|
|
78
|
+
output: null,
|
|
79
|
+
stopReason: 'error',
|
|
80
|
+
numTurns: 1,
|
|
81
|
+
};
|
|
26
82
|
}
|
|
27
|
-
const usage = data.usage || {};
|
|
28
83
|
return {
|
|
29
|
-
ok: true, output
|
|
30
|
-
inputTokens
|
|
31
|
-
cacheReadTokens
|
|
32
|
-
costUSD: 0,
|
|
84
|
+
ok: true, output, durationMs, durationApiMs: 0,
|
|
85
|
+
inputTokens, outputTokens,
|
|
86
|
+
cacheReadTokens, cacheCreationTokens,
|
|
87
|
+
costUSD: 0, costReportedByExecutor: false,
|
|
88
|
+
stopReason: data.stop_reason || 'end_turn', numTurns: 1,
|
|
33
89
|
};
|
|
34
90
|
}
|
|
35
91
|
catch (err) {
|
|
@@ -37,6 +93,6 @@ export async function anthropicApiExecutor({ model, system, prompt, timeoutMs =
|
|
|
37
93
|
const details = asErrorLike(err);
|
|
38
94
|
const stopReason = details.name === 'TimeoutError' ? 'timeout' : 'error';
|
|
39
95
|
const error = details.name === 'TimeoutError' ? `API request timed out after ${timeoutMs / 1000}s` : errorMessage(err);
|
|
40
|
-
return { ok: false, error, durationMs, durationApiMs: 0, inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheCreationTokens: 0, costUSD: 0, output: null, stopReason, numTurns: 0 };
|
|
96
|
+
return { ok: false, error, durationMs, durationApiMs: 0, inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheCreationTokens: 0, tokenUsageReportedByExecutor: false, costUSD: 0, costReportedByExecutor: false, output: null, stopReason, numTurns: 0 };
|
|
41
97
|
}
|
|
42
98
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { extractAgentTrace, isClaudeSdkResultMessage } from './claude-sdk-trace.js';
|
|
2
1
|
import { buildExecEnv, DEFAULT_TIMEOUT_MS, errorMessage, interruptedExecResult, MAX_BUFFER, spawnWithSigintPropagation, timeoutExecResult, } from './shared.js';
|
|
3
2
|
import { materializeForCliConfigDir } from '../eval-core/mocks-runtime.js';
|
|
3
|
+
import { buildClaudeResult, parseClaudeStreamJson } from './claude-protocol.js';
|
|
4
4
|
// claude CLI 用 `--disable-slash-commands` (文档:"Disable all skills") +
|
|
5
5
|
// `--disallowedTools Skill` 实现与 SDK 等价的完全隔离。非空 skill 白名单不再支持
|
|
6
6
|
// (它无法真正隔离,见 claude-sdk.ts buildSdkIsolationOptions),所以 [name1, ...] 必须 throw。
|
|
@@ -19,19 +19,6 @@ function applySkillIsolationToCliArgs(args, allowedSkills) {
|
|
|
19
19
|
// 完全隔离:双堵 main session skill 发现 + subagent Skill 工具
|
|
20
20
|
args.push('--disable-slash-commands', '--disallowedTools', 'Skill');
|
|
21
21
|
}
|
|
22
|
-
function parseStreamJson(stdout) {
|
|
23
|
-
const messages = [];
|
|
24
|
-
for (const line of stdout.split('\n')) {
|
|
25
|
-
const trimmed = line.trim();
|
|
26
|
-
if (!trimmed)
|
|
27
|
-
continue;
|
|
28
|
-
try {
|
|
29
|
-
messages.push(JSON.parse(trimmed));
|
|
30
|
-
}
|
|
31
|
-
catch { /* skip non-JSON lines */ }
|
|
32
|
-
}
|
|
33
|
-
return messages;
|
|
34
|
-
}
|
|
35
22
|
export async function claudeCliExecutor({ model, system, prompt, cwd, skillDir, timeoutMs = DEFAULT_TIMEOUT_MS, allowedSkills, mocks, mocksBaseDir, mocksStrict, lean, effort }) {
|
|
36
23
|
const args = ['-p', prompt, '--output-format', 'stream-json', '--verbose', '--model', model,
|
|
37
24
|
// 评测必须 bypass permission,否则 Bash / Edit / Write 等工具调用会卡在交互式确认。
|
|
@@ -86,40 +73,15 @@ export async function claudeCliExecutor({ model, system, prompt, cwd, skillDir,
|
|
|
86
73
|
child.stdin?.end();
|
|
87
74
|
const { stdout } = await done;
|
|
88
75
|
const durationMs = Date.now() - start;
|
|
89
|
-
const
|
|
90
|
-
// 提取 result 消息
|
|
91
|
-
const resultMsgs = messages.filter(isClaudeSdkResultMessage);
|
|
92
|
-
if (resultMsgs.length === 0) {
|
|
93
|
-
const ms = captureMockStats();
|
|
94
|
-
return {
|
|
95
|
-
ok: false, error: 'no result message in stream-json output',
|
|
96
|
-
durationMs, durationApiMs: 0,
|
|
97
|
-
inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheCreationTokens: 0,
|
|
98
|
-
costUSD: 0, output: null, stopReason: 'error', numTurns: 0,
|
|
99
|
-
...(ms && { mockStats: ms }),
|
|
100
|
-
};
|
|
101
|
-
}
|
|
102
|
-
const last = resultMsgs[resultMsgs.length - 1];
|
|
103
|
-
const usage = last.usage || {};
|
|
104
|
-
// 提取 trace
|
|
105
|
-
const trace = extractAgentTrace(messages);
|
|
76
|
+
const parsed = parseClaudeStreamJson(stdout);
|
|
106
77
|
const ms = captureMockStats();
|
|
107
78
|
return {
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
cacheCreationTokens: usage.cache_creation_input_tokens || 0,
|
|
115
|
-
costUSD: last.total_cost_usd || 0,
|
|
116
|
-
output: last.result || '',
|
|
117
|
-
stopReason: last.subtype || 'unknown',
|
|
118
|
-
numTurns: last.num_turns || 1,
|
|
119
|
-
fullNumTurns: trace.fullNumTurns,
|
|
120
|
-
numSubAgents: trace.numSubAgents,
|
|
121
|
-
...(trace.turns.length > 0 && { turns: trace.turns }),
|
|
122
|
-
...(trace.toolCalls.length > 0 && { toolCalls: trace.toolCalls }),
|
|
79
|
+
...buildClaudeResult({
|
|
80
|
+
messages: parsed.messages,
|
|
81
|
+
malformedLineCount: parsed.malformedLineCount,
|
|
82
|
+
wallClockDurationMs: durationMs,
|
|
83
|
+
source: 'claude stream-json',
|
|
84
|
+
}),
|
|
123
85
|
...(ms && { mockStats: ms }),
|
|
124
86
|
};
|
|
125
87
|
}
|
|
@@ -134,41 +96,16 @@ export async function claudeCliExecutor({ model, system, prompt, cwd, skillDir,
|
|
|
134
96
|
const ms = captureMockStats();
|
|
135
97
|
return { ...interruptedExecResult(durationMs), ...(ms && { mockStats: ms }) };
|
|
136
98
|
}
|
|
137
|
-
|
|
138
|
-
const messages = parseStreamJson(details.stdout || '');
|
|
139
|
-
const resultMsgs = messages.filter(isClaudeSdkResultMessage);
|
|
140
|
-
if (resultMsgs.length > 0) {
|
|
141
|
-
const last = resultMsgs[resultMsgs.length - 1];
|
|
142
|
-
const usage = last.usage || {};
|
|
143
|
-
const trace = extractAgentTrace(messages);
|
|
144
|
-
const ms = captureMockStats();
|
|
145
|
-
return {
|
|
146
|
-
ok: false,
|
|
147
|
-
error: last.errors?.join('; ') || last.result || errorMessage(err),
|
|
148
|
-
durationMs: last.duration_ms || durationMs,
|
|
149
|
-
durationApiMs: last.duration_api_ms || 0,
|
|
150
|
-
inputTokens: usage.input_tokens || 0,
|
|
151
|
-
outputTokens: usage.output_tokens || 0,
|
|
152
|
-
cacheReadTokens: usage.cache_read_input_tokens || 0,
|
|
153
|
-
cacheCreationTokens: usage.cache_creation_input_tokens || 0,
|
|
154
|
-
costUSD: last.total_cost_usd || 0,
|
|
155
|
-
output: last.result || null,
|
|
156
|
-
stopReason: 'error',
|
|
157
|
-
numTurns: last.num_turns || 0,
|
|
158
|
-
fullNumTurns: trace.fullNumTurns,
|
|
159
|
-
numSubAgents: trace.numSubAgents,
|
|
160
|
-
...(trace.turns.length > 0 && { turns: trace.turns }),
|
|
161
|
-
...(trace.toolCalls.length > 0 && { toolCalls: trace.toolCalls }),
|
|
162
|
-
...(ms && { mockStats: ms }),
|
|
163
|
-
};
|
|
164
|
-
}
|
|
99
|
+
const parsed = parseClaudeStreamJson(details.stdout || '');
|
|
165
100
|
const ms = captureMockStats();
|
|
166
101
|
return {
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
102
|
+
...buildClaudeResult({
|
|
103
|
+
messages: parsed.messages,
|
|
104
|
+
malformedLineCount: parsed.malformedLineCount,
|
|
105
|
+
wallClockDurationMs: durationMs,
|
|
106
|
+
source: 'claude stream-json',
|
|
107
|
+
forcedError: details.stderr?.trim() || errorMessage(err),
|
|
108
|
+
}),
|
|
172
109
|
...(ms && { mockStats: ms }),
|
|
173
110
|
};
|
|
174
111
|
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import type { ExecResult } from '../types/index.js';
|
|
2
|
+
import type { ClaudeSdkBaseMessage, ClaudeSdkResultMessage } from './shared.js';
|
|
3
|
+
export interface ClaudeSdkMeasurements {
|
|
4
|
+
durationMs: number;
|
|
5
|
+
durationApiMs: number;
|
|
6
|
+
inputTokens: number;
|
|
7
|
+
outputTokens: number;
|
|
8
|
+
cacheReadTokens: number;
|
|
9
|
+
cacheCreationTokens: number;
|
|
10
|
+
costUSD: number;
|
|
11
|
+
numTurns: number;
|
|
12
|
+
}
|
|
13
|
+
export declare function normalizeClaudeSdkMeasurements(result: ClaudeSdkResultMessage): ClaudeSdkMeasurements | {
|
|
14
|
+
error: string;
|
|
15
|
+
};
|
|
16
|
+
export interface ClaudeStreamParseResult {
|
|
17
|
+
messages: ClaudeSdkBaseMessage[];
|
|
18
|
+
malformedLineCount: number;
|
|
19
|
+
}
|
|
20
|
+
export declare function parseClaudeStreamJson(stdout: string): ClaudeStreamParseResult;
|
|
21
|
+
export declare function buildClaudeResult(options: {
|
|
22
|
+
messages: ClaudeSdkBaseMessage[];
|
|
23
|
+
wallClockDurationMs: number;
|
|
24
|
+
source: 'claude stream-json' | 'claude-sdk';
|
|
25
|
+
malformedLineCount?: number;
|
|
26
|
+
forcedError?: string;
|
|
27
|
+
messageTimestamps?: number[];
|
|
28
|
+
}): ExecResult;
|