oh-my-knowledge 0.48.0 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -18
- package/README.zh.md +55 -23
- package/dist/analysis/coverage-analyzer.d.ts +1 -0
- package/dist/analysis/coverage-analyzer.js +125 -62
- package/dist/analysis/failure-clusterer.js +2 -1
- package/dist/analysis/gap-analyzer.d.ts +2 -2
- package/dist/analysis/gap-analyzer.js +13 -3
- package/dist/analysis/hedging-classifier.d.ts +2 -2
- package/dist/analysis/hedging-classifier.js +3 -4
- package/dist/analysis/report-diagnostics.js +9 -7
- package/dist/analysis/sample-diagnostics.js +6 -6
- package/dist/artifact-graph/doctor.js +15 -7
- package/dist/assets/agent-skills/omk/SKILL.md +27 -7
- package/dist/assets/agent-skills/omk/references/commands.md +18 -17
- package/dist/authoring/evolver.d.ts +10 -6
- package/dist/authoring/evolver.js +496 -83
- package/dist/authoring/generator.d.ts +3 -3
- package/dist/authoring/generator.js +5 -10
- package/dist/authoring/sample-fixer.d.ts +8 -6
- package/dist/authoring/sample-fixer.js +76 -5
- package/dist/cli/commands/doctor.js +31 -14
- package/dist/cli/commands/eval/index.d.ts +3 -0
- package/dist/cli/commands/eval/index.js +163 -17
- package/dist/cli/commands/evolve.d.ts +4 -4
- package/dist/cli/commands/evolve.js +27 -13
- package/dist/cli/commands/init.js +16 -3
- package/dist/cli/commands/observe/inbox.js +28 -21
- package/dist/cli/commands/observe/index.js +20 -11
- package/dist/cli/commands/observe/ingest.d.ts +3 -0
- package/dist/cli/commands/observe/ingest.js +30 -2
- package/dist/cli/commands/sample.d.ts +6 -3
- package/dist/cli/commands/sample.js +72 -68
- package/dist/cli/lib/codex-model-hint.d.ts +9 -0
- package/dist/cli/lib/codex-model-hint.js +45 -0
- package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
- package/dist/cli/lib/generation-failure-hint.js +61 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +4 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +38 -6
- package/dist/cli/lib/i18n-dict/help.js +6 -6
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +13 -9
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +34 -2
- package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
- package/dist/cli/lib/llm-failure-classifier.js +8 -0
- package/dist/cli/lib/parse-run-config.d.ts +6 -5
- package/dist/cli/lib/parse-run-config.js +16 -9
- package/dist/cli/lib/runtime-defaults.d.ts +21 -0
- package/dist/cli/lib/runtime-defaults.js +79 -0
- package/dist/diagnosis/observe-mapper.js +14 -15
- package/dist/diagnosis/observe-producer.js +3 -1
- package/dist/diagnosis/studio-projection.js +14 -7
- package/dist/diagnosis/types.d.ts +2 -0
- package/dist/diagnosis/types.js +12 -0
- package/dist/doctor/endpoint-rule.js +2 -1
- package/dist/eval-core/artifact-file-names.js +18 -1
- package/dist/eval-core/artifact-index.d.ts +7 -11
- package/dist/eval-core/artifact-index.js +139 -80
- package/dist/eval-core/cache.d.ts +12 -3
- package/dist/eval-core/cache.js +89 -29
- package/dist/eval-core/comparability.js +10 -6
- package/dist/eval-core/evaluation-execution.d.ts +2 -1
- package/dist/eval-core/evaluation-execution.js +122 -37
- package/dist/eval-core/evaluation-job.d.ts +4 -1
- package/dist/eval-core/evaluation-job.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +15 -13
- package/dist/eval-core/evaluation-reporting.js +54 -52
- package/dist/eval-core/execution-strategy.d.ts +2 -0
- package/dist/eval-core/execution-strategy.js +11 -9
- package/dist/eval-core/fact-checker.js +15 -7
- package/dist/eval-core/holdout.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +2 -2
- package/dist/eval-core/mock-hook.cjs +23 -6
- package/dist/eval-core/mocks-runtime.js +30 -8
- package/dist/eval-core/report-document.d.ts +12 -0
- package/dist/eval-core/report-document.js +1151 -0
- package/dist/eval-core/report-extensions.d.ts +4 -0
- package/dist/eval-core/report-extensions.js +500 -0
- package/dist/eval-core/report-file-migration.js +7 -2
- package/dist/eval-core/resume-compatibility.d.ts +31 -0
- package/dist/eval-core/resume-compatibility.js +141 -0
- package/dist/eval-core/sample-fingerprint.d.ts +12 -0
- package/dist/eval-core/sample-fingerprint.js +193 -0
- package/dist/eval-core/schema.js +86 -31
- package/dist/eval-core/verdict.d.ts +8 -4
- package/dist/eval-core/verdict.js +24 -10
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
- package/dist/eval-workflows/evaluation-pipeline.js +12 -7
- package/dist/eval-workflows/run-evaluation.d.ts +9 -7
- package/dist/eval-workflows/run-evaluation.js +79 -51
- package/dist/executors/anthropic-api.js +65 -9
- package/dist/executors/claude-cli.js +16 -79
- package/dist/executors/claude-protocol.d.ts +28 -0
- package/dist/executors/claude-protocol.js +180 -0
- package/dist/executors/claude-sdk-trace.js +56 -28
- package/dist/executors/claude-sdk.d.ts +1 -0
- package/dist/executors/claude-sdk.js +39 -93
- package/dist/executors/codex-cli-trace.js +166 -31
- package/dist/executors/codex-cli.d.ts +6 -8
- package/dist/executors/codex-cli.js +49 -151
- package/dist/executors/codex-protocol.d.ts +24 -0
- package/dist/executors/codex-protocol.js +234 -0
- package/dist/executors/codex-sdk.js +68 -120
- package/dist/executors/gemini.js +88 -13
- package/dist/executors/index.d.ts +2 -3
- package/dist/executors/index.js +5 -3
- package/dist/executors/openai-api.js +70 -9
- package/dist/executors/runtime-fingerprint.js +88 -11
- package/dist/executors/script-command.d.ts +8 -0
- package/dist/executors/script-command.js +87 -0
- package/dist/executors/script.js +202 -29
- package/dist/executors/shared.d.ts +35 -3
- package/dist/executors/shared.js +113 -15
- package/dist/grading/assertions.d.ts +1 -1
- package/dist/grading/assertions.js +19 -9
- package/dist/grading/diagnostic.d.ts +9 -2
- package/dist/grading/diagnostic.js +25 -2
- package/dist/grading/index.js +10 -4
- package/dist/grading/judge.js +19 -6
- package/dist/grading/layered-scores.d.ts +2 -3
- package/dist/grading/layered-scores.js +2 -3
- package/dist/inputs/load-samples.d.ts +1 -2
- package/dist/inputs/load-samples.js +23 -1
- package/dist/inputs/mcp-resolver.js +6 -3
- package/dist/inputs/sample-document.d.ts +11 -0
- package/dist/inputs/sample-document.js +96 -0
- package/dist/managed/evidence.d.ts +1 -0
- package/dist/managed/evidence.js +1 -1
- package/dist/managed/store.js +200 -91
- package/dist/observability/codex-trace-adapter.d.ts +5 -0
- package/dist/observability/codex-trace-adapter.js +850 -0
- package/dist/observability/experience.d.ts +32 -6
- package/dist/observability/experience.js +2695 -459
- package/dist/observability/feedback-matchers.js +16 -1
- package/dist/observability/inbox-view-model.d.ts +1 -1
- package/dist/observability/inbox-view-model.js +19 -14
- package/dist/observability/inbox.d.ts +7 -1
- package/dist/observability/inbox.js +632 -124
- package/dist/observability/problem-patterns.js +2 -0
- package/dist/observability/review-state.d.ts +6 -0
- package/dist/observability/review-state.js +235 -63
- package/dist/observability/skill-chain-advisories.js +1 -1
- package/dist/observability/skill-chain.js +17 -4
- package/dist/observability/skill-health-analyzer.d.ts +32 -7
- package/dist/observability/skill-health-analyzer.js +194 -121
- package/dist/observability/skill-health-report.d.ts +10 -0
- package/dist/observability/skill-health-report.js +620 -0
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +1 -1
- package/dist/observability/soft-standards/index.js +1 -1
- package/dist/observability/soft-standards/llm-extractor.js +8 -10
- package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
- package/dist/observability/soft-standards/skill-standards-store.js +59 -18
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/observability/trace-adapter.d.ts +12 -7
- package/dist/observability/trace-adapter.js +11 -9
- package/dist/observability/trace-attribution.d.ts +13 -5
- package/dist/observability/trace-attribution.js +315 -21
- package/dist/observability/trace-ingestion.d.ts +9 -0
- package/dist/observability/trace-ingestion.js +80 -0
- package/dist/observability/trace-ir.d.ts +113 -0
- package/dist/observability/trace-ir.js +87 -0
- package/dist/observability/trace-segmenter.d.ts +19 -6
- package/dist/observability/trace-segmenter.js +377 -196
- package/dist/observability/trace-session-index.d.ts +19 -0
- package/dist/observability/trace-session-index.js +68 -0
- package/dist/observability/trace-source.d.ts +12 -4
- package/dist/observability/trace-source.js +939 -215
- package/dist/renderer/html-renderer.js +37 -6
- package/dist/renderer/icons.js +3 -0
- package/dist/renderer/observation-inbox-renderer.js +208 -90
- package/dist/renderer/skill-detail-renderer.js +452 -109
- package/dist/renderer/skill-health-renderer.js +69 -12
- package/dist/renderer/summary.js +28 -7
- package/dist/renderer/table.js +21 -4
- package/dist/renderer/test-view.d.ts +1 -0
- package/dist/renderer/test-view.js +44 -9
- package/dist/server/indexed-report-store.js +14 -18
- package/dist/server/job-store.js +64 -26
- package/dist/server/report-server.js +190 -78
- package/dist/server/report-store.js +57 -80
- package/dist/server/skill-index.js +143 -49
- package/dist/server/skill-insights.js +44 -5
- package/dist/shared/artifact-graph.d.ts +3 -0
- package/dist/shared/artifact-graph.js +224 -0
- package/dist/shared/assertion-types.d.ts +8 -0
- package/dist/shared/assertion-types.js +46 -0
- package/dist/shared/atomic-json.d.ts +8 -0
- package/dist/shared/atomic-json.js +33 -0
- package/dist/shared/diagnosis-schema.d.ts +9 -0
- package/dist/shared/diagnosis-schema.js +181 -0
- package/dist/shared/doctor-report.d.ts +3 -0
- package/dist/shared/doctor-report.js +103 -0
- package/dist/shared/evaluation-job.d.ts +6 -0
- package/dist/shared/evaluation-job.js +217 -0
- package/dist/shared/executor-result.d.ts +17 -0
- package/dist/shared/executor-result.js +221 -0
- package/dist/shared/file-lock.d.ts +12 -0
- package/dist/shared/file-lock.js +129 -0
- package/dist/shared/json-value.d.ts +5 -0
- package/dist/shared/json-value.js +36 -0
- package/dist/shared/keyed-mutex.d.ts +7 -0
- package/dist/shared/keyed-mutex.js +24 -0
- package/dist/shared/record-count.d.ts +8 -0
- package/dist/shared/record-count.js +43 -0
- package/dist/shared/sample-contract.d.ts +3 -0
- package/dist/shared/sample-contract.js +332 -0
- package/dist/shared/timestamp.d.ts +6 -0
- package/dist/shared/timestamp.js +64 -0
- package/dist/shared/token-usage.d.ts +19 -0
- package/dist/shared/token-usage.js +50 -0
- package/dist/shared/tool-call-status.d.ts +8 -0
- package/dist/shared/tool-call-status.js +28 -0
- package/dist/shared/tool-identity.d.ts +21 -0
- package/dist/shared/tool-identity.js +84 -0
- package/dist/shared/tool-search.js +73 -16
- package/dist/shared/trace-projection.d.ts +5 -0
- package/dist/shared/trace-projection.js +20 -0
- package/dist/shared/trace-source-kind.d.ts +3 -0
- package/dist/shared/trace-source-kind.js +12 -0
- package/dist/types/diagnosis.d.ts +2 -0
- package/dist/types/eval.d.ts +4 -0
- package/dist/types/executor.d.ts +32 -5
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/judge.d.ts +2 -0
- package/dist/types/observability.d.ts +116 -9
- package/dist/types/report.d.ts +58 -6
- package/dist/types/skill-index.d.ts +7 -0
- package/dist/types/trace.d.ts +2 -0
- package/dist/types/trace.js +1 -0
- package/package.json +9 -5
|
@@ -1,16 +1,24 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { copyFileSync, existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, statSync, writeFileSync, } from 'node:fs';
|
|
2
|
+
import { tmpdir } from 'node:os';
|
|
2
3
|
import { resolve, join, dirname, basename } from 'node:path';
|
|
3
4
|
import { runEvaluation } from '../eval-workflows/run-evaluation.js';
|
|
4
|
-
import { createExecutor
|
|
5
|
-
import { persistReport, DEFAULT_OUTPUT_DIR, runIdSuffix, hashString } from '../eval-core/evaluation-reporting.js';
|
|
5
|
+
import { createExecutor } from '../executors/index.js';
|
|
6
|
+
import { persistReport, DEFAULT_OUTPUT_DIR, EVALUATION_REPORT_SCHEMA_VERSION, getCliVersion, hashSample, runIdSuffix, hashString, } from '../eval-core/evaluation-reporting.js';
|
|
6
7
|
import { createOverlayReportStore } from '../server/report-store.js';
|
|
7
8
|
import { projectReportsDir, globalReportsDir } from '../eval-core/measurement-dirs.js';
|
|
8
9
|
import { analyzeResults } from '../analysis/report-diagnostics.js';
|
|
9
10
|
import { loadSamples } from '../inputs/load-samples.js';
|
|
11
|
+
import { writeFixedSamplesToSources } from '../inputs/sample-document.js';
|
|
10
12
|
import { hashArtifactSource } from '../inputs/content-hash.js';
|
|
11
13
|
import { MIN_HOLDOUT_SUBSET, pickByStride, splitHoldout, subsetCompositeScore } from '../eval-core/holdout.js';
|
|
12
14
|
import { bootstrapDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
|
|
13
|
-
import { fixSamples } from './sample-fixer.js';
|
|
15
|
+
import { fixSamples, sampleFixWithinScope, stampFixMetadata } from './sample-fixer.js';
|
|
16
|
+
import { toolCallStatus } from '../shared/tool-call-status.js';
|
|
17
|
+
import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
|
|
18
|
+
import { getJudgePromptHash } from '../grading/judge.js';
|
|
19
|
+
import { getDiagnosticPromptHash, resolveDiagnosticTarget, } from '../grading/diagnostic.js';
|
|
20
|
+
import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
|
|
21
|
+
import { parseReportDocument } from '../eval-core/report-document.js';
|
|
14
22
|
const IMPROVE_SYSTEM_PROMPT = `你是一个 AI 提示词改进专家。你的任务是分析评测结果中的薄弱环节,针对性地改进 skill(系统提示词),使其在评测中获得更高的分数。
|
|
15
23
|
|
|
16
24
|
改进原则:
|
|
@@ -46,58 +54,146 @@ function canonicalStringify(value) {
|
|
|
46
54
|
const entries = Object.keys(value).sort();
|
|
47
55
|
return '{' + entries.map((k) => JSON.stringify(k) + ':' + canonicalStringify(value[k])).join(',') + '}';
|
|
48
56
|
}
|
|
49
|
-
function hashSampleForReuse(sample) {
|
|
50
|
-
return hashString(canonicalStringify({
|
|
51
|
-
prompt: sample.prompt,
|
|
52
|
-
rubric: sample.rubric ?? null,
|
|
53
|
-
dimensions: sample.dimensions ?? null,
|
|
54
|
-
assertions: sample.assertions ?? null,
|
|
55
|
-
schema: sample.schema ?? null,
|
|
56
|
-
}));
|
|
57
|
-
}
|
|
58
57
|
function sameJudgeModels(report, judges) {
|
|
59
58
|
const metaJudges = report.meta.judgeModels ?? [];
|
|
60
59
|
if (metaJudges.length !== judges.length)
|
|
61
60
|
return false;
|
|
62
61
|
return metaJudges.every((j, i) => j.executor === judges[i].executor && j.model === judges[i].model);
|
|
63
62
|
}
|
|
64
|
-
function sampleHashesMatch(report, samples) {
|
|
63
|
+
function sampleHashesMatch(report, samples, samplesBaseDir) {
|
|
65
64
|
const hashes = report.meta.sampleHashes;
|
|
66
|
-
if (!hashes
|
|
65
|
+
if (!hashes
|
|
66
|
+
|| report.meta.sampleCount !== samples.length
|
|
67
|
+
|| Object.keys(hashes).length !== samples.length)
|
|
67
68
|
return false;
|
|
68
|
-
return samples.every((s) => hashes
|
|
69
|
+
return samples.every((s) => ownRecordValue(hashes, s.sample_id) === hashSample(s, samplesBaseDir));
|
|
70
|
+
}
|
|
71
|
+
function omitKeys(value, keys) {
|
|
72
|
+
const copy = { ...value };
|
|
73
|
+
for (const key of keys)
|
|
74
|
+
Reflect.deleteProperty(copy, key);
|
|
75
|
+
return copy;
|
|
69
76
|
}
|
|
70
|
-
function singleVariantReport(report, variantKey) {
|
|
71
|
-
const summary = report.summary
|
|
77
|
+
export function singleVariantReport(report, variantKey) {
|
|
78
|
+
const summary = ownRecordValue(report.summary, variantKey);
|
|
79
|
+
if (!summary) {
|
|
80
|
+
throw new Error(`report is missing summary for variant "${variantKey}"`);
|
|
81
|
+
}
|
|
82
|
+
const artifactHash = report.meta.artifactHashes
|
|
83
|
+
? ownRecordValue(report.meta.artifactHashes, variantKey)
|
|
84
|
+
: undefined;
|
|
85
|
+
const isolation = report.meta.skillIsolation
|
|
86
|
+
? ownRecordValue(report.meta.skillIsolation, variantKey)
|
|
87
|
+
: undefined;
|
|
88
|
+
const results = report.results.map((entry) => {
|
|
89
|
+
const variant = ownRecordValue(entry.variants, variantKey);
|
|
90
|
+
if (!variant) {
|
|
91
|
+
throw new Error(`report is missing variant "${variantKey}" for sample "${entry.sample_id}"`);
|
|
92
|
+
}
|
|
93
|
+
return {
|
|
94
|
+
sample_id: entry.sample_id,
|
|
95
|
+
variants: { [variantKey]: variant },
|
|
96
|
+
};
|
|
97
|
+
});
|
|
98
|
+
const totalCostUSD = Number(results.reduce((total, entry) => total + entry.variants[variantKey].costUSD, 0).toFixed(6));
|
|
99
|
+
const executorRuntime = report.meta.executorRuntimes
|
|
100
|
+
? ownRecordValue(report.meta.executorRuntimes, variantKey)
|
|
101
|
+
: report.meta.executorRuntime;
|
|
102
|
+
const request = report.meta.request
|
|
103
|
+
? {
|
|
104
|
+
...report.meta.request,
|
|
105
|
+
artifacts: report.meta.request.artifacts.filter((artifact) => artifact.name === variantKey),
|
|
106
|
+
}
|
|
107
|
+
: undefined;
|
|
108
|
+
if (report.meta.request && request?.artifacts.length !== 1) {
|
|
109
|
+
throw new Error(`report request is missing artifact "${variantKey}"`);
|
|
110
|
+
}
|
|
111
|
+
const sourceHumanAgreement = report.meta.humanAgreement;
|
|
112
|
+
const sourceJob = report.meta.job;
|
|
113
|
+
const sharedMeta = omitKeys(report.meta, [
|
|
114
|
+
'variants',
|
|
115
|
+
'taskCount',
|
|
116
|
+
'totalCostUSD',
|
|
117
|
+
'totalCostReported',
|
|
118
|
+
'artifactHashes',
|
|
119
|
+
'executorRuntime',
|
|
120
|
+
'executorRuntimes',
|
|
121
|
+
'pairComparisons',
|
|
122
|
+
'humanAgreement',
|
|
123
|
+
'variantConfigs',
|
|
124
|
+
'skillIsolation',
|
|
125
|
+
'request',
|
|
126
|
+
'job',
|
|
127
|
+
'evolve',
|
|
128
|
+
]);
|
|
129
|
+
const sharedReport = omitKeys(report, ['analysis', 'variance']);
|
|
72
130
|
return {
|
|
73
|
-
...
|
|
131
|
+
...sharedReport,
|
|
74
132
|
meta: {
|
|
75
|
-
...
|
|
133
|
+
...sharedMeta,
|
|
76
134
|
variants: [variantKey],
|
|
77
|
-
|
|
78
|
-
|
|
135
|
+
taskCount: report.meta.sampleCount,
|
|
136
|
+
totalCostUSD,
|
|
137
|
+
...(summary.execCostReported === false || summary.judgeCostReported === false
|
|
138
|
+
? { totalCostReported: false }
|
|
139
|
+
: {}),
|
|
140
|
+
artifactHashes: artifactHash
|
|
141
|
+
? { [variantKey]: artifactHash }
|
|
79
142
|
: {},
|
|
143
|
+
...(executorRuntime
|
|
144
|
+
? {
|
|
145
|
+
executorRuntime,
|
|
146
|
+
executorRuntimes: { [variantKey]: executorRuntime },
|
|
147
|
+
}
|
|
148
|
+
: {}),
|
|
80
149
|
...(report.meta.variantConfigs ? { variantConfigs: report.meta.variantConfigs.filter((cfg) => cfg.variant === variantKey) } : {}),
|
|
81
|
-
...(report.meta.skillIsolation ? { skillIsolation: { [variantKey]:
|
|
150
|
+
...(report.meta.skillIsolation ? { skillIsolation: { [variantKey]: isolation ?? null } } : {}),
|
|
151
|
+
...(sourceHumanAgreement?.variant === variantKey
|
|
152
|
+
? { humanAgreement: sourceHumanAgreement }
|
|
153
|
+
: {}),
|
|
154
|
+
...(request ? { request } : {}),
|
|
155
|
+
...(sourceJob && request
|
|
156
|
+
? {
|
|
157
|
+
job: {
|
|
158
|
+
...sourceJob,
|
|
159
|
+
request: structuredClone(request),
|
|
160
|
+
},
|
|
161
|
+
}
|
|
162
|
+
: {}),
|
|
82
163
|
},
|
|
83
164
|
summary: { [variantKey]: summary },
|
|
84
|
-
results
|
|
85
|
-
sample_id: entry.sample_id,
|
|
86
|
-
variants: entry.variants[variantKey] ? { [variantKey]: entry.variants[variantKey] } : {},
|
|
87
|
-
})),
|
|
165
|
+
results,
|
|
88
166
|
};
|
|
89
167
|
}
|
|
90
168
|
async function findReusableBaselineReport(opts) {
|
|
91
169
|
// baseline 复用读 overlay(项目 .omk/reports ∪ 全局):eval 写默认翻项目后,复用既能命中 eval 新写的项目
|
|
92
170
|
// baseline,又继续覆盖全局(含 evolve 自身写到全局的合并报告),复用命中率与报告数字不降。
|
|
93
171
|
const store = createOverlayReportStore(projectReportsDir(), globalReportsDir());
|
|
94
|
-
const { samples } = loadSamples(opts.samplesPath);
|
|
172
|
+
const { samples, baseDir: samplesBaseDir } = loadSamples(opts.samplesPath);
|
|
95
173
|
const artifactHash = opts.artifactHash;
|
|
96
174
|
const reports = await store.findByArtifactHash(artifactHash);
|
|
175
|
+
const expectedExecutorRuntime = getExecutorRuntimeFingerprint(opts.executorName, opts.model);
|
|
176
|
+
const expectedJudgeRuntimes = opts.judgeModels.map((judge) => getExecutorRuntimeFingerprint(judge.executor, judge.model, {
|
|
177
|
+
skillDir: opts.skillDir,
|
|
178
|
+
}));
|
|
179
|
+
const expectedJudgePromptHash = getJudgePromptHash(true);
|
|
180
|
+
const expectedDiagnosticTarget = resolveDiagnosticTarget(opts.judgeModels, opts.executorName, opts.model);
|
|
181
|
+
const expectedDiagnostic = opts.noDiagnostic
|
|
182
|
+
? { enabled: false }
|
|
183
|
+
: {
|
|
184
|
+
enabled: true,
|
|
185
|
+
executor: expectedDiagnosticTarget.executor,
|
|
186
|
+
model: expectedDiagnosticTarget.model,
|
|
187
|
+
runtime: getExecutorRuntimeFingerprint(expectedDiagnosticTarget.executor, expectedDiagnosticTarget.model).fingerprint,
|
|
188
|
+
promptHash: getDiagnosticPromptHash(),
|
|
189
|
+
};
|
|
97
190
|
for (const report of reports) {
|
|
98
|
-
//
|
|
99
|
-
//
|
|
100
|
-
|
|
191
|
+
// Reuse is a measurement-contract decision, not just a content-hash lookup.
|
|
192
|
+
// Old schema / CLI / prompt/runtime evidence can produce numerically similar
|
|
193
|
+
// scores under a different construct, so fail closed and rerun.
|
|
194
|
+
if (report.meta.schemaVersion !== EVALUATION_REPORT_SCHEMA_VERSION)
|
|
195
|
+
continue;
|
|
196
|
+
if (report.meta.cliVersion !== getCliVersion())
|
|
101
197
|
continue;
|
|
102
198
|
if (report.meta.model !== opts.model || report.meta.executor !== opts.executorName)
|
|
103
199
|
continue;
|
|
@@ -107,13 +203,67 @@ async function findReusableBaselineReport(opts) {
|
|
|
107
203
|
continue;
|
|
108
204
|
if (report.meta.budgetExhausted)
|
|
109
205
|
continue;
|
|
206
|
+
if ((report.meta.request?.noDiagnostic === true)
|
|
207
|
+
!== (opts.noDiagnostic === true))
|
|
208
|
+
continue;
|
|
209
|
+
const actualDiagnostic = report.meta.diagnostic
|
|
210
|
+
? {
|
|
211
|
+
enabled: report.meta.diagnostic.enabled,
|
|
212
|
+
...(report.meta.diagnostic.executor
|
|
213
|
+
? { executor: report.meta.diagnostic.executor }
|
|
214
|
+
: {}),
|
|
215
|
+
...(report.meta.diagnostic.model
|
|
216
|
+
? { model: report.meta.diagnostic.model }
|
|
217
|
+
: {}),
|
|
218
|
+
...(report.meta.diagnostic.runtime
|
|
219
|
+
? { runtime: report.meta.diagnostic.runtime.fingerprint }
|
|
220
|
+
: {}),
|
|
221
|
+
...(report.meta.diagnostic.promptHash
|
|
222
|
+
? { promptHash: report.meta.diagnostic.promptHash }
|
|
223
|
+
: {}),
|
|
224
|
+
}
|
|
225
|
+
: undefined;
|
|
226
|
+
if (canonicalStringify(actualDiagnostic) !== canonicalStringify(expectedDiagnostic))
|
|
227
|
+
continue;
|
|
110
228
|
if (!sameJudgeModels(report, opts.judgeModels))
|
|
111
229
|
continue;
|
|
112
|
-
if (
|
|
230
|
+
if (report.meta.judgePromptHash !== expectedJudgePromptHash)
|
|
231
|
+
continue;
|
|
232
|
+
if (report.meta.judgeModels.some((judge, index) => !judge.runtime
|
|
233
|
+
|| judge.runtime.fingerprint !== expectedJudgeRuntimes[index]?.fingerprint))
|
|
234
|
+
continue;
|
|
235
|
+
if (!sampleHashesMatch(report, samples, samplesBaseDir))
|
|
236
|
+
continue;
|
|
237
|
+
const variantKey = report.meta.variants.find((name) => report.meta.artifactHashes
|
|
238
|
+
&& ownRecordValue(report.meta.artifactHashes, name) === artifactHash);
|
|
239
|
+
if (!variantKey || !ownRecordValue(report.summary, variantKey))
|
|
240
|
+
continue;
|
|
241
|
+
const executorRuntime = report.meta.executorRuntimes
|
|
242
|
+
? ownRecordValue(report.meta.executorRuntimes, variantKey)
|
|
243
|
+
: report.meta.executorRuntime;
|
|
244
|
+
if (executorRuntime?.fingerprint !== expectedExecutorRuntime.fingerprint)
|
|
113
245
|
continue;
|
|
114
|
-
const
|
|
115
|
-
if (!
|
|
246
|
+
const config = report.meta.variantConfigs?.find((candidate) => candidate.variant === variantKey);
|
|
247
|
+
if (!config
|
|
248
|
+
|| config.artifactKind !== 'skill'
|
|
249
|
+
|| config.executionStrategy !== 'system-prompt'
|
|
250
|
+
|| config.experimentType !== 'artifact-injection'
|
|
251
|
+
|| config.hasArtifactContent !== true
|
|
252
|
+
|| config.allowedSkills !== undefined)
|
|
116
253
|
continue;
|
|
254
|
+
if (!report.meta.skillIsolation
|
|
255
|
+
|| !Object.hasOwn(report.meta.skillIsolation, variantKey)
|
|
256
|
+
|| ownRecordValue(report.meta.skillIsolation, variantKey) !== null)
|
|
257
|
+
continue;
|
|
258
|
+
if (opts.noDiagnostic !== true) {
|
|
259
|
+
const missingDiagnostic = report.results.some((entry) => {
|
|
260
|
+
const variant = ownRecordValue(entry.variants, variantKey);
|
|
261
|
+
return variant?.assertions?.details.some((detail) => !detail.passed) === true
|
|
262
|
+
&& variant.diagnostic === undefined;
|
|
263
|
+
});
|
|
264
|
+
if (missingDiagnostic)
|
|
265
|
+
continue;
|
|
266
|
+
}
|
|
117
267
|
return singleVariantReport(report, variantKey);
|
|
118
268
|
}
|
|
119
269
|
return null;
|
|
@@ -280,11 +430,20 @@ function createSampleFixExecutor(executorName) {
|
|
|
280
430
|
lean: opts.lean,
|
|
281
431
|
...(opts.cwd && { cwd: opts.cwd }),
|
|
282
432
|
});
|
|
283
|
-
return {
|
|
433
|
+
return {
|
|
434
|
+
ok: result.ok,
|
|
435
|
+
text: result.output ?? '',
|
|
436
|
+
costUSD: result.costUSD,
|
|
437
|
+
costReported: result.costReportedByExecutor !== false,
|
|
438
|
+
};
|
|
284
439
|
};
|
|
285
440
|
}
|
|
441
|
+
export function agentSampleEditWithinScope(previous, next) {
|
|
442
|
+
return sampleFixWithinScope(previous, next);
|
|
443
|
+
}
|
|
286
444
|
async function autoFixSamplesAgent(opts) {
|
|
287
|
-
const
|
|
445
|
+
const loaded = loadSamples(opts.samplesPath);
|
|
446
|
+
const { samples } = loaded;
|
|
288
447
|
const sampleMap = new Map(samples.map((s) => [s.sample_id, s]));
|
|
289
448
|
const fixContexts = [];
|
|
290
449
|
for (const entry of opts.report.results) {
|
|
@@ -310,7 +469,7 @@ async function autoFixSamplesAgent(opts) {
|
|
|
310
469
|
const toolCalls = (variantObj.toolCalls ?? []);
|
|
311
470
|
const failedList = assertionDetails.filter((a) => !a.passed).map((a) => `${a.type}: ${a.value}`).join('\n');
|
|
312
471
|
const toolSummary = toolCalls.length > 0
|
|
313
|
-
? toolCalls.map((tc, i) => `[${i}] ${tc.tool}
|
|
472
|
+
? toolCalls.map((tc, i) => `[${i}] ${tc.tool} status=${toolCallStatus(tc)}`).join('\n')
|
|
314
473
|
: '(无工具调用)';
|
|
315
474
|
fixContexts.push({
|
|
316
475
|
sampleId: sid,
|
|
@@ -318,13 +477,30 @@ async function autoFixSamplesAgent(opts) {
|
|
|
318
477
|
failedAssertions: `失败断言:\n${failedList}\n\n实际工具调用:\n${toolSummary}`,
|
|
319
478
|
});
|
|
320
479
|
}
|
|
321
|
-
if (fixContexts.length === 0)
|
|
322
|
-
return { fixedCount: 0, costUSD: 0 };
|
|
480
|
+
if (fixContexts.length === 0) {
|
|
481
|
+
return { fixedCount: 0, costUSD: 0, costReported: true };
|
|
482
|
+
}
|
|
323
483
|
const skillPreview = opts.skillContent.length > 4000
|
|
324
484
|
? opts.skillContent.slice(0, 4000) + '\n\n... (truncated)'
|
|
325
485
|
: opts.skillContent;
|
|
326
486
|
const sampleSections = fixContexts.map((ctx) => `### ${ctx.sampleId}\n\n${ctx.diag}\n\n${ctx.failedAssertions}`).join('\n\n---\n\n');
|
|
327
|
-
const
|
|
487
|
+
const tempRoot = mkdtempSync(join(tmpdir(), 'omk-sample-fix-'));
|
|
488
|
+
const sourceIsDirectory = statSync(opts.samplesPath).isDirectory();
|
|
489
|
+
const tempSamplesPath = sourceIsDirectory
|
|
490
|
+
? join(tempRoot, 'samples')
|
|
491
|
+
: join(tempRoot, basename(opts.samplesPath));
|
|
492
|
+
if (sourceIsDirectory)
|
|
493
|
+
mkdirSync(tempSamplesPath);
|
|
494
|
+
for (const sourceFile of loaded.sourceFiles) {
|
|
495
|
+
const targetFile = sourceIsDirectory
|
|
496
|
+
? join(tempSamplesPath, basename(sourceFile))
|
|
497
|
+
: tempSamplesPath;
|
|
498
|
+
copyFileSync(sourceFile, targetFile);
|
|
499
|
+
}
|
|
500
|
+
try {
|
|
501
|
+
const tempLoaded = loadSamples(tempSamplesPath);
|
|
502
|
+
const editableFiles = tempLoaded.sourceFiles.map((file) => `- ${file}`).join('\n');
|
|
503
|
+
const prompt = `以下有 ${fixContexts.length} 条失败的评测用例需要分析修复。
|
|
328
504
|
|
|
329
505
|
## Skill 原文(参考,不可修改)
|
|
330
506
|
|
|
@@ -334,23 +510,67 @@ ${skillPreview}
|
|
|
334
510
|
|
|
335
511
|
${sampleSections}
|
|
336
512
|
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
513
|
+
请只使用 Edit 工具修改以下临时副本:
|
|
514
|
+
${editableFiles}
|
|
515
|
+
|
|
516
|
+
只改有问题 sample 的 assertions / mocks / mocksStrict / environment 字段。如果判断是 LLM 行为问题(低分合理),不要改该 sample。`;
|
|
517
|
+
const executor = createExecutor(opts.executorName);
|
|
518
|
+
const result = await executor({
|
|
519
|
+
model: opts.model,
|
|
520
|
+
system: FIX_AGENT_SYSTEM_PROMPT,
|
|
521
|
+
prompt,
|
|
522
|
+
cwd: tempRoot,
|
|
523
|
+
timeoutMs: 300_000,
|
|
524
|
+
});
|
|
525
|
+
const costReported = result.costReportedByExecutor !== false;
|
|
526
|
+
if (!result.ok) {
|
|
527
|
+
return { fixedCount: 0, costUSD: result.costUSD, costReported };
|
|
528
|
+
}
|
|
529
|
+
let edited;
|
|
530
|
+
try {
|
|
531
|
+
edited = loadSamples(tempSamplesPath).samples;
|
|
532
|
+
}
|
|
533
|
+
catch (error) {
|
|
534
|
+
process.stderr.write(`[omk] auto-fix-samples 已拒绝非法评测用例修改:`
|
|
535
|
+
+ `${error instanceof Error ? error.message : String(error)}\n`);
|
|
536
|
+
return { fixedCount: 0, costUSD: result.costUSD, costReported };
|
|
537
|
+
}
|
|
538
|
+
const originalById = new Map(samples.map((sample) => [sample.sample_id, sample]));
|
|
539
|
+
const editedById = new Map(edited.map((sample) => [sample.sample_id, sample]));
|
|
540
|
+
if (editedById.size !== originalById.size
|
|
541
|
+
|| [...originalById.keys()].some((sampleId) => !editedById.has(sampleId))) {
|
|
542
|
+
process.stderr.write('[omk] auto-fix-samples 已拒绝新增、删除或改名评测用例。\n');
|
|
543
|
+
return { fixedCount: 0, costUSD: result.costUSD, costReported };
|
|
544
|
+
}
|
|
545
|
+
const fixableIds = new Set(fixContexts.map((context) => context.sampleId));
|
|
546
|
+
const changedIds = new Set();
|
|
547
|
+
for (const [sampleId, nextSample] of editedById.entries()) {
|
|
548
|
+
const previousSample = originalById.get(sampleId);
|
|
549
|
+
if (canonicalStringify(nextSample) === canonicalStringify(previousSample))
|
|
550
|
+
continue;
|
|
551
|
+
if (!fixableIds.has(sampleId)
|
|
552
|
+
|| !agentSampleEditWithinScope(previousSample, nextSample)) {
|
|
553
|
+
process.stderr.write(`[omk] auto-fix-samples 已拒绝越界修改 sample「${sampleId}」;`
|
|
554
|
+
+ '只允许修改待修复用例的 assertions / mocks / mocksStrict / environment。\n');
|
|
555
|
+
return { fixedCount: 0, costUSD: result.costUSD, costReported };
|
|
556
|
+
}
|
|
557
|
+
stampFixMetadata(nextSample, opts.report.id);
|
|
558
|
+
changedIds.add(sampleId);
|
|
559
|
+
}
|
|
560
|
+
writeFixedSamplesToSources(loaded, edited, changedIds);
|
|
561
|
+
return {
|
|
562
|
+
fixedCount: changedIds.size,
|
|
563
|
+
costUSD: result.costUSD,
|
|
564
|
+
costReported,
|
|
565
|
+
};
|
|
566
|
+
}
|
|
567
|
+
finally {
|
|
568
|
+
rmSync(tempRoot, { recursive: true, force: true });
|
|
569
|
+
}
|
|
351
570
|
}
|
|
352
571
|
async function autoFixSamplesAfterSkillRound(opts) {
|
|
353
|
-
const
|
|
572
|
+
const loaded = loadSamples(opts.samplesPath);
|
|
573
|
+
const { samples } = loaded;
|
|
354
574
|
if (opts.improveMode === 'agent') {
|
|
355
575
|
return autoFixSamplesAgent(opts);
|
|
356
576
|
}
|
|
@@ -364,9 +584,14 @@ async function autoFixSamplesAfterSkillRound(opts) {
|
|
|
364
584
|
maxAttemptsPerSample: opts.maxAttemptsPerSample,
|
|
365
585
|
});
|
|
366
586
|
if (result.fixedCount > 0) {
|
|
367
|
-
|
|
587
|
+
const changedIds = new Set(result.fixes.filter((fix) => fix.changed).map((fix) => fix.sampleId));
|
|
588
|
+
writeFixedSamplesToSources(loaded, result.samples, changedIds);
|
|
368
589
|
}
|
|
369
|
-
return {
|
|
590
|
+
return {
|
|
591
|
+
fixedCount: result.fixedCount,
|
|
592
|
+
costUSD: result.costUSD,
|
|
593
|
+
costReported: result.costReported,
|
|
594
|
+
};
|
|
370
595
|
}
|
|
371
596
|
/** Number of skill lines below which the edit budget never trips — so a tiny skill
|
|
372
597
|
* isn't frozen by a percentage threshold that a few lines already blow past. */
|
|
@@ -458,16 +683,130 @@ function parseImprovedSkill(output) {
|
|
|
458
683
|
}
|
|
459
684
|
return content;
|
|
460
685
|
}
|
|
461
|
-
|
|
686
|
+
function sourceVariantForRound({ round, report }) {
|
|
687
|
+
if (report.meta.variants.length !== 1) {
|
|
688
|
+
throw new Error(`evolve 第 ${round} 轮报告必须且只能包含一个 variant。`);
|
|
689
|
+
}
|
|
690
|
+
const variant = report.meta.variants[0];
|
|
691
|
+
if (!ownRecordValue(report.summary, variant)
|
|
692
|
+
|| report.results.some((entry) => !ownRecordValue(entry.variants, variant))) {
|
|
693
|
+
throw new Error(`evolve 第 ${round} 轮报告缺少 variant「${variant}」的完整结果。`);
|
|
694
|
+
}
|
|
695
|
+
return variant;
|
|
696
|
+
}
|
|
697
|
+
function assertComparableEvolveRounds(roundReports, sourceVariants) {
|
|
698
|
+
const first = roundReports[0].report;
|
|
699
|
+
const firstSampleIds = first.results.map((entry) => entry.sample_id);
|
|
700
|
+
const firstSampleSet = new Set(firstSampleIds);
|
|
701
|
+
if (firstSampleSet.size !== firstSampleIds.length) {
|
|
702
|
+
throw new Error('evolve 首轮报告包含重复 sample_id。');
|
|
703
|
+
}
|
|
704
|
+
const comparableMeta = (report, variant) => {
|
|
705
|
+
const config = report.meta.variantConfigs?.find((candidate) => candidate.variant === variant);
|
|
706
|
+
const runtime = report.meta.executorRuntimes
|
|
707
|
+
? ownRecordValue(report.meta.executorRuntimes, variant)
|
|
708
|
+
: report.meta.executorRuntime;
|
|
709
|
+
const hasIsolation = Boolean(report.meta.skillIsolation
|
|
710
|
+
&& Object.hasOwn(report.meta.skillIsolation, variant));
|
|
711
|
+
return {
|
|
712
|
+
model: report.meta.model,
|
|
713
|
+
executor: report.meta.executor,
|
|
714
|
+
effort: report.meta.effort,
|
|
715
|
+
schemaVersion: report.meta.schemaVersion,
|
|
716
|
+
cliVersion: report.meta.cliVersion,
|
|
717
|
+
nodeVersion: report.meta.nodeVersion,
|
|
718
|
+
judgeRepeat: report.meta.judgeRepeat,
|
|
719
|
+
noJudge: report.meta.noJudge,
|
|
720
|
+
judgePromptHash: report.meta.judgePromptHash,
|
|
721
|
+
diagnostic: report.meta.diagnostic
|
|
722
|
+
? {
|
|
723
|
+
enabled: report.meta.diagnostic.enabled,
|
|
724
|
+
executor: report.meta.diagnostic.executor,
|
|
725
|
+
model: report.meta.diagnostic.model,
|
|
726
|
+
runtime: report.meta.diagnostic.runtime?.fingerprint,
|
|
727
|
+
promptHash: report.meta.diagnostic.promptHash,
|
|
728
|
+
}
|
|
729
|
+
: undefined,
|
|
730
|
+
evaluationFramework: report.meta.evaluationFramework,
|
|
731
|
+
debiasMode: report.meta.debiasMode,
|
|
732
|
+
judgeModels: report.meta.judgeModels,
|
|
733
|
+
sampleHashes: report.meta.sampleHashes,
|
|
734
|
+
requestProtocol: report.meta.request
|
|
735
|
+
? {
|
|
736
|
+
timeoutMs: report.meta.request.timeoutMs,
|
|
737
|
+
retry: report.meta.request.retry ?? 0,
|
|
738
|
+
budget: report.meta.request.budget,
|
|
739
|
+
noDiagnostic: report.meta.request.noDiagnostic === true,
|
|
740
|
+
}
|
|
741
|
+
: undefined,
|
|
742
|
+
executorRuntime: runtime
|
|
743
|
+
? {
|
|
744
|
+
fingerprint: runtime.fingerprint,
|
|
745
|
+
capabilities: runtime.capabilities,
|
|
746
|
+
}
|
|
747
|
+
: null,
|
|
748
|
+
execution: config
|
|
749
|
+
? {
|
|
750
|
+
artifactKind: config.artifactKind,
|
|
751
|
+
executionStrategy: config.executionStrategy,
|
|
752
|
+
experimentType: config.experimentType,
|
|
753
|
+
hasArtifactContent: config.hasArtifactContent,
|
|
754
|
+
cwd: config.cwd,
|
|
755
|
+
allowedSkills: config.allowedSkills,
|
|
756
|
+
}
|
|
757
|
+
: null,
|
|
758
|
+
skillIsolation: {
|
|
759
|
+
present: hasIsolation,
|
|
760
|
+
value: hasIsolation
|
|
761
|
+
? ownRecordValue(report.meta.skillIsolation, variant)
|
|
762
|
+
: undefined,
|
|
763
|
+
},
|
|
764
|
+
};
|
|
765
|
+
};
|
|
766
|
+
const expectedMeta = canonicalStringify(comparableMeta(first, sourceVariants[0]));
|
|
767
|
+
for (let index = 0; index < roundReports.length; index++) {
|
|
768
|
+
const { round, report } = roundReports[index];
|
|
769
|
+
if (!Number.isSafeInteger(round) || round < 0) {
|
|
770
|
+
throw new Error(`evolve 轮次必须是非负安全整数,收到「${String(round)}」。`);
|
|
771
|
+
}
|
|
772
|
+
if (index > 0 && round <= roundReports[index - 1].round) {
|
|
773
|
+
throw new Error('evolve 轮次必须严格递增且不能重复。');
|
|
774
|
+
}
|
|
775
|
+
if (!parseReportDocument(report, report.id, report.id)) {
|
|
776
|
+
throw new Error(`evolve 第 ${round} 轮报告不符合持久化契约。`);
|
|
777
|
+
}
|
|
778
|
+
if (report.meta.budgetExhausted === true) {
|
|
779
|
+
throw new Error(`evolve 第 ${round} 轮报告因预算耗尽而不完整。`);
|
|
780
|
+
}
|
|
781
|
+
if (report.meta.sampleCount !== first.meta.sampleCount
|
|
782
|
+
|| report.results.length !== first.results.length
|
|
783
|
+
|| canonicalStringify(comparableMeta(report, sourceVariants[index])) !== expectedMeta) {
|
|
784
|
+
throw new Error(`evolve 第 ${round} 轮与首轮的测量配置或用例集合不可比。`);
|
|
785
|
+
}
|
|
786
|
+
const sampleIds = new Set(report.results.map((entry) => entry.sample_id));
|
|
787
|
+
if (sampleIds.size !== firstSampleSet.size
|
|
788
|
+
|| firstSampleIds.some((sampleId) => !sampleIds.has(sampleId))) {
|
|
789
|
+
throw new Error(`evolve 第 ${round} 轮的 sample_id 集合与首轮不一致。`);
|
|
790
|
+
}
|
|
791
|
+
if (!ownRecordValue(report.meta.artifactHashes, sourceVariants[index])) {
|
|
792
|
+
throw new Error(`evolve 第 ${round} 轮缺少知识载体内容指纹。`);
|
|
793
|
+
}
|
|
794
|
+
}
|
|
795
|
+
}
|
|
796
|
+
export function mergeEvolveReports(roundReports, skillName, processCostUSD, samples, skillPath, processCostReported = true) {
|
|
797
|
+
if (roundReports.length === 0) {
|
|
798
|
+
throw new Error('evolve 合并报告至少需要一轮评测结果。');
|
|
799
|
+
}
|
|
462
800
|
const firstReport = roundReports[0].report;
|
|
801
|
+
const sourceVariants = roundReports.map(sourceVariantForRound);
|
|
802
|
+
assertComparableEvolveRounds(roundReports, sourceVariants);
|
|
463
803
|
// Build variant labels: "round-0", "round-1", "round-2", ...
|
|
464
804
|
const variantLabels = roundReports.map(({ round }) => `round-${round}`);
|
|
465
805
|
// Build summary: map each variant label to its round's summary
|
|
466
806
|
const summary = {};
|
|
467
807
|
for (let i = 0; i < roundReports.length; i++) {
|
|
468
808
|
const { report } = roundReports[i];
|
|
469
|
-
|
|
470
|
-
summary[variantLabels[i]] = report.summary[originalKey];
|
|
809
|
+
setOwnRecordValue(summary, variantLabels[i], ownRecordValue(report.summary, sourceVariants[i]));
|
|
471
810
|
}
|
|
472
811
|
// Build results: merge per-sample variant data across rounds
|
|
473
812
|
const sampleIds = firstReport.results.map((r) => r.sample_id);
|
|
@@ -475,41 +814,108 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
|
|
|
475
814
|
const variants = {};
|
|
476
815
|
for (let i = 0; i < roundReports.length; i++) {
|
|
477
816
|
const entry = roundReports[i].report.results.find((r) => r.sample_id === sampleId);
|
|
478
|
-
|
|
479
|
-
const originalKey = Object.keys(entry.variants)[0];
|
|
480
|
-
variants[variantLabels[i]] = entry.variants[originalKey];
|
|
481
|
-
}
|
|
817
|
+
setOwnRecordValue(variants, variantLabels[i], ownRecordValue(entry.variants, sourceVariants[i]));
|
|
482
818
|
}
|
|
483
819
|
return { sample_id: sampleId, variants };
|
|
484
820
|
});
|
|
485
|
-
//
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
variant
|
|
489
|
-
|
|
490
|
-
|
|
821
|
+
// round-0 是当前知识载体版本的 control,不是「无知识载体」的 baseline kind。
|
|
822
|
+
// 保留每轮真实 artifact / execution 语义,只重标实验角色。
|
|
823
|
+
const variantConfigs = roundReports.map(({ report }, i) => {
|
|
824
|
+
const cfg = report.meta.variantConfigs?.find((candidate) => candidate.variant === sourceVariants[i]);
|
|
825
|
+
return cfg ? {
|
|
826
|
+
...cfg,
|
|
827
|
+
variant: variantLabels[i],
|
|
828
|
+
experimentRole: i === 0 ? 'control' : 'treatment',
|
|
829
|
+
} : undefined;
|
|
830
|
+
});
|
|
491
831
|
// Build artifactHashes: collect from each round, relabel key
|
|
492
832
|
const artifactHashes = {};
|
|
493
833
|
for (let i = 0; i < roundReports.length; i++) {
|
|
494
|
-
|
|
495
|
-
const originalKey = Object.keys(hashes)[0];
|
|
496
|
-
if (originalKey)
|
|
497
|
-
artifactHashes[variantLabels[i]] = hashes[originalKey];
|
|
834
|
+
setOwnRecordValue(artifactHashes, variantLabels[i], ownRecordValue(roundReports[i].report.meta.artifactHashes, sourceVariants[i]));
|
|
498
835
|
}
|
|
836
|
+
const executorRuntimes = {};
|
|
837
|
+
let hasCompleteExecutorRuntimes = true;
|
|
838
|
+
const sourceExecutorRuntimes = roundReports.map(({ report }, index) => {
|
|
839
|
+
const runtime = report.meta.executorRuntimes
|
|
840
|
+
? ownRecordValue(report.meta.executorRuntimes, sourceVariants[index])
|
|
841
|
+
: report.meta.executorRuntime;
|
|
842
|
+
if (!runtime) {
|
|
843
|
+
hasCompleteExecutorRuntimes = false;
|
|
844
|
+
return undefined;
|
|
845
|
+
}
|
|
846
|
+
setOwnRecordValue(executorRuntimes, variantLabels[index], runtime);
|
|
847
|
+
return runtime;
|
|
848
|
+
});
|
|
849
|
+
const commonExecutorRuntime = sourceExecutorRuntimes[0]
|
|
850
|
+
&& sourceExecutorRuntimes.every((runtime) => runtime !== undefined
|
|
851
|
+
&& canonicalStringify(runtime) === canonicalStringify(sourceExecutorRuntimes[0]))
|
|
852
|
+
? sourceExecutorRuntimes[0]
|
|
853
|
+
: undefined;
|
|
854
|
+
const skillIsolation = {};
|
|
855
|
+
let hasCompleteSkillIsolation = true;
|
|
856
|
+
for (let index = 0; index < roundReports.length; index++) {
|
|
857
|
+
const source = roundReports[index].report.meta.skillIsolation;
|
|
858
|
+
if (!source || !Object.hasOwn(source, sourceVariants[index])) {
|
|
859
|
+
hasCompleteSkillIsolation = false;
|
|
860
|
+
break;
|
|
861
|
+
}
|
|
862
|
+
setOwnRecordValue(skillIsolation, variantLabels[index], ownRecordValue(source, sourceVariants[index]) ?? null);
|
|
863
|
+
}
|
|
864
|
+
// request / run / job 描述的是某一次源评测,不能伪装成 aggregate 的生命周期。
|
|
865
|
+
// pairComparisons / humanAgreement / budget 也绑定源报告的 variant 或单轮预算,
|
|
866
|
+
// 合并后必须重新计算才有意义,因此不从首轮继承。
|
|
867
|
+
const sharedMeta = omitKeys(firstReport.meta, [
|
|
868
|
+
'variants',
|
|
869
|
+
'taskCount',
|
|
870
|
+
'totalCostUSD',
|
|
871
|
+
'totalCostReported',
|
|
872
|
+
'timestamp',
|
|
873
|
+
'artifactHashes',
|
|
874
|
+
'executorRuntime',
|
|
875
|
+
'executorRuntimes',
|
|
876
|
+
'pairComparisons',
|
|
877
|
+
'humanAgreement',
|
|
878
|
+
'budget',
|
|
879
|
+
'budgetExhausted',
|
|
880
|
+
'variantConfigs',
|
|
881
|
+
'skillIsolation',
|
|
882
|
+
'request',
|
|
883
|
+
'run',
|
|
884
|
+
'job',
|
|
885
|
+
'evolve',
|
|
886
|
+
]);
|
|
499
887
|
const runId = `evolve-${skillName}-${runIdSuffix()}`;
|
|
888
|
+
const measurementCostUSD = Number(results.reduce((total, entry) => total + variantLabels.reduce((sampleTotal, variant) => sampleTotal + (ownRecordValue(entry.variants, variant)?.costUSD ?? 0), 0), 0).toFixed(6));
|
|
889
|
+
const measurementCostReported = roundReports.every(({ report: source }) => Object.values(source.summary).every((variant) => variant.execCostReported !== false
|
|
890
|
+
&& variant.judgeCostReported !== false));
|
|
500
891
|
const report = {
|
|
501
892
|
kind: 'evaluation',
|
|
502
893
|
id: runId,
|
|
503
894
|
meta: {
|
|
504
|
-
...
|
|
895
|
+
...sharedMeta,
|
|
505
896
|
variants: variantLabels,
|
|
506
|
-
|
|
897
|
+
taskCount: firstReport.meta.sampleCount * variantLabels.length,
|
|
898
|
+
...(variantConfigs.every((config) => config !== undefined)
|
|
899
|
+
? { variantConfigs: variantConfigs }
|
|
900
|
+
: {}),
|
|
507
901
|
artifactHashes,
|
|
508
|
-
|
|
902
|
+
...(hasCompleteExecutorRuntimes ? { executorRuntimes } : {}),
|
|
903
|
+
...(commonExecutorRuntime ? { executorRuntime: commonExecutorRuntime } : {}),
|
|
904
|
+
...(hasCompleteSkillIsolation ? { skillIsolation } : {}),
|
|
905
|
+
totalCostUSD: measurementCostUSD,
|
|
906
|
+
...(measurementCostReported ? {} : { totalCostReported: false }),
|
|
509
907
|
timestamp: new Date().toISOString(),
|
|
510
908
|
evolve: {
|
|
511
909
|
skillName,
|
|
512
910
|
...(skillPath ? { skillPath } : {}),
|
|
911
|
+
processCostUSD: Number(processCostUSD.toFixed(6)),
|
|
912
|
+
...(processCostReported ? {} : { processCostReported: false }),
|
|
913
|
+
sourceReports: roundReports.map(({ round, accepted, report: source }, index) => ({
|
|
914
|
+
round,
|
|
915
|
+
accepted,
|
|
916
|
+
reportId: source.id,
|
|
917
|
+
variant: sourceVariants[index],
|
|
918
|
+
})),
|
|
513
919
|
},
|
|
514
920
|
},
|
|
515
921
|
summary,
|
|
@@ -522,15 +928,15 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
|
|
|
522
928
|
report.analysis = analyzeResults(report, { samples });
|
|
523
929
|
return report;
|
|
524
930
|
}
|
|
525
|
-
export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model
|
|
931
|
+
export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model, judgeModels, improveModel = model, improveMode = 'agent', executorName, concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, holdoutRatio = 0, significanceGate = true, significanceAlpha = DEFAULT_BOOTSTRAP_ALPHA, testRatio = 0, editBudget = 0.2, rejectMemory = true, writeBackToSource = true, onProgress = null, onRoundProgress = null, }) {
|
|
526
932
|
if (judgeModels && judgeModels.length > 1) {
|
|
527
933
|
throw new Error('evolveSkill does not support multi-judge ensemble (received '
|
|
528
934
|
+ `${judgeModels.length} judges). Pass a single-judge array, e.g. `
|
|
529
|
-
+ `[{ executor:
|
|
935
|
+
+ `[{ executor: executorName, model }]`);
|
|
530
936
|
}
|
|
531
937
|
const effectiveJudgeModels = judgeModels && judgeModels.length > 0
|
|
532
938
|
? judgeModels
|
|
533
|
-
: [{ executor: executorName, model
|
|
939
|
+
: [{ executor: executorName, model }];
|
|
534
940
|
const absSkillPath = resolve(skillPath);
|
|
535
941
|
const absSamplesPath = resolve(samplesPath);
|
|
536
942
|
const skillDir = dirname(absSkillPath);
|
|
@@ -635,6 +1041,7 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
635
1041
|
? await findReusableBaselineReport({
|
|
636
1042
|
artifactHash: baselineArtifactHash,
|
|
637
1043
|
samplesPath: absSamplesPath,
|
|
1044
|
+
skillDir,
|
|
638
1045
|
model,
|
|
639
1046
|
executorName,
|
|
640
1047
|
judgeModels: effectiveJudgeModels,
|
|
@@ -667,7 +1074,7 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
667
1074
|
if (writeBackToSource)
|
|
668
1075
|
writeFileSync(absSkillPath, currentBest);
|
|
669
1076
|
const { samples } = loadSamples(absSamplesPath);
|
|
670
|
-
const mergedReport = mergeEvolveReports(roundReports, skillName, totalCostUSD, samples, absSkillPath);
|
|
1077
|
+
const mergedReport = mergeEvolveReports(roundReports, skillName, totalCostUSD, samples, absSkillPath, totalCostReported);
|
|
671
1078
|
persistReport(mergedReport, DEFAULT_OUTPUT_DIR);
|
|
672
1079
|
return {
|
|
673
1080
|
startScore: bestScore,
|
|
@@ -784,6 +1191,7 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
784
1191
|
continue;
|
|
785
1192
|
}
|
|
786
1193
|
let preEvalSampleFixCost = 0;
|
|
1194
|
+
let preEvalSampleFixCostReported = true;
|
|
787
1195
|
if (autoFixSamples) {
|
|
788
1196
|
const sampleFix = await autoFixSamplesAfterSkillRound({
|
|
789
1197
|
samplesPath: absSamplesPath,
|
|
@@ -801,7 +1209,10 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
801
1209
|
sampleFixes.push({ round, fixedCount: sampleFix.fixedCount, costUSD: sampleFix.costUSD });
|
|
802
1210
|
}
|
|
803
1211
|
preEvalSampleFixCost = sampleFix.costUSD;
|
|
1212
|
+
preEvalSampleFixCostReported = sampleFix.costReported;
|
|
804
1213
|
totalCostUSD += sampleFix.costUSD;
|
|
1214
|
+
if (!sampleFix.costReported)
|
|
1215
|
+
totalCostReported = false;
|
|
805
1216
|
}
|
|
806
1217
|
// Evaluate candidate with any sample fixes already applied.
|
|
807
1218
|
const candidateReport = await evaluate(candidatePath, {
|
|
@@ -810,7 +1221,9 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
810
1221
|
const candidateVariantKey = Object.keys(candidateReport.summary)[0];
|
|
811
1222
|
const candidateScore = decisionScore(candidateReport, candidateVariantKey);
|
|
812
1223
|
const roundCost = improveCostUSD + preEvalSampleFixCost + candidateReport.meta.totalCostUSD;
|
|
813
|
-
const roundCostReported = improveCostReported
|
|
1224
|
+
const roundCostReported = improveCostReported
|
|
1225
|
+
&& preEvalSampleFixCostReported
|
|
1226
|
+
&& !reportHasUnreportedCost(candidateReport);
|
|
814
1227
|
if (!roundCostReported)
|
|
815
1228
|
totalCostReported = false;
|
|
816
1229
|
totalCostUSD += improveCostUSD + candidateReport.meta.totalCostUSD;
|
|
@@ -875,7 +1288,7 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
875
1288
|
if (roundReports.length > 0) {
|
|
876
1289
|
// load samples once to enable analysis.sampleQuality on the merged report.
|
|
877
1290
|
const { samples } = loadSamples(absSamplesPath);
|
|
878
|
-
const mergedReport = mergeEvolveReports(roundReports, skillName, totalCostUSD, samples, absSkillPath);
|
|
1291
|
+
const mergedReport = mergeEvolveReports(roundReports, skillName, totalCostUSD, samples, absSkillPath, totalCostReported);
|
|
879
1292
|
persistReport(mergedReport, DEFAULT_OUTPUT_DIR);
|
|
880
1293
|
reportId = mergedReport.id;
|
|
881
1294
|
}
|