@tangle-network/agent-eval 0.175.0 → 0.176.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/dist/adapters/http.d.ts +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js +374 -0
- package/dist/agent-profile-cell-0gSi5ffD.js.map +1 -0
- package/dist/analyst/index.d.ts +2 -2
- package/dist/analyst/index.js +3 -3
- package/dist/{benchmark-command-D_5xG9LG.js → benchmark-command-yPqjcZnC.js} +7 -7
- package/dist/{benchmark-command-D_5xG9LG.js.map → benchmark-command-yPqjcZnC.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +3 -3
- package/dist/campaign/index.js +9 -9
- package/dist/{campaign-BzMSCejE.js → campaign-85igdlgG.js} +12 -12
- package/dist/{campaign-BzMSCejE.js.map → campaign-85igdlgG.js.map} +1 -1
- package/dist/campaign-evidence-D8DBLqLI.js +2083 -0
- package/dist/campaign-evidence-D8DBLqLI.js.map +1 -0
- package/dist/cli.js +1 -1
- package/dist/contract/index.d.ts +4 -4
- package/dist/contract/index.js +8 -8
- package/dist/{define-agent-eval-V1jQyCDR.d.ts → define-agent-eval-CCbl8k2E.d.ts} +11 -4
- package/dist/define-agent-eval-CCbl8k2E.d.ts.map +1 -0
- package/dist/{define-agent-eval-ox5McL6e.js → define-agent-eval-DEMsu5eA.js} +54 -36
- package/dist/define-agent-eval-DEMsu5eA.js.map +1 -0
- package/dist/{dspy-rlm-engine-Caz2pl4L.js → dspy-rlm-engine-DqjER2sV.js} +2 -2
- package/dist/{dspy-rlm-engine-Caz2pl4L.js.map → dspy-rlm-engine-DqjER2sV.js.map} +1 -1
- package/dist/{eval-campaign-BeAjdhzC.js → eval-campaign-Cs-7MiCs.js} +4 -5
- package/dist/{eval-campaign-BeAjdhzC.js.map → eval-campaign-Cs-7MiCs.js.map} +1 -1
- package/dist/experiment/index.d.ts +3 -68
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +6 -128
- package/dist/experiment/index.js.map +1 -1
- package/dist/{attestation-XSUpbc4o.js → experiment-tracker-BKEumQug.js} +2 -96
- package/dist/experiment-tracker-BKEumQug.js.map +1 -0
- package/dist/{attestation-c1QvaBdX.d.ts → experiment-tracker-CNwqCZFD.d.ts} +2 -78
- package/dist/experiment-tracker-CNwqCZFD.d.ts.map +1 -0
- package/dist/{external-optimizer-process-CxnFL1hd.js → external-optimizer-process-Dlz8YxrT.js} +3 -3
- package/dist/{external-optimizer-process-CxnFL1hd.js.map → external-optimizer-process-Dlz8YxrT.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-CQi27uEI.js → external-optimizer-subprocess-q3VzlGAO.js} +2 -2
- package/dist/{external-optimizer-subprocess-CQi27uEI.js.map → external-optimizer-subprocess-q3VzlGAO.js.map} +1 -1
- package/dist/{index-DKXuBPXf.d.ts → index-Bg6OT2Dd.d.ts} +23 -10
- package/dist/{index-DKXuBPXf.d.ts.map → index-Bg6OT2Dd.d.ts.map} +1 -1
- package/dist/{index-BTrx5s8m.d.ts → index-DBkcm_9H.d.ts} +4 -4
- package/dist/{index-BTrx5s8m.d.ts.map → index-DBkcm_9H.d.ts.map} +1 -1
- package/dist/{index-D-UdhAmg.d.ts → index-u0d1Jp4F.d.ts} +4 -2
- package/dist/{index-D-UdhAmg.d.ts.map → index-u0d1Jp4F.d.ts.map} +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.js +14 -15
- package/dist/index.js.map +1 -1
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-PIfjCbKn.js → ledger-core-Cs9f7385.js} +60 -47
- package/dist/{ledger-core-PIfjCbKn.js.map → ledger-core-Cs9f7385.js.map} +1 -1
- package/dist/{llm-judge-DmNaBrXB.js → llm-judge-DliimmRb.js} +994 -1517
- package/dist/llm-judge-DliimmRb.js.map +1 -0
- package/dist/{mint-vWOdD8Ae.js → mint-Cc1_zwRQ.js} +2 -2
- package/dist/{mint-vWOdD8Ae.js.map → mint-Cc1_zwRQ.js.map} +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-B8mw6zj9.js → produced-state-DrMqa2HD.js} +3 -2
- package/dist/{produced-state-B8mw6zj9.js.map → produced-state-DrMqa2HD.js.map} +1 -1
- package/dist/profile-cell.js +1 -268
- package/dist/{promotion-policy-LY9mVQ7W.js → promotion-policy-DWOm70gx.js} +2 -2
- package/dist/{promotion-policy-LY9mVQ7W.js.map → promotion-policy-DWOm70gx.js.map} +1 -1
- package/dist/{release-confidence-BsGEg_xg.js → release-confidence-BcGCclTB.js} +2 -2
- package/dist/{release-confidence-BsGEg_xg.js.map → release-confidence-BcGCclTB.js.map} +1 -1
- package/dist/reporting.js +2 -2
- package/dist/{reward-hacking-CKW4teig.js → reward-hacking-D0XwhVWE.js} +2 -215
- package/dist/reward-hacking-D0XwhVWE.js.map +1 -0
- package/dist/rl.js +5 -4
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-CGlDq1GI.js → rollout-DmoJVqrF.js} +2 -2
- package/dist/{rollout-CGlDq1GI.js.map → rollout-DmoJVqrF.js.map} +1 -1
- package/dist/run-record-CR63CpHK.js +216 -0
- package/dist/run-record-CR63CpHK.js.map +1 -0
- package/dist/{run-record-ZIsR9Fif.js → run-record-DQpSf7t-.js} +2 -2
- package/dist/{run-record-ZIsR9Fif.js.map → run-record-DQpSf7t-.js.map} +1 -1
- package/dist/{semantic-concept-judge-E3s_fEjB.js → semantic-concept-judge-Dw-f7TEs.js} +3 -3
- package/dist/{semantic-concept-judge-E3s_fEjB.js.map → semantic-concept-judge-Dw-f7TEs.js.map} +1 -1
- package/dist/{sequential-B51qAYE4.js → sequential-B5gXgcyp.js} +3 -3
- package/dist/{sequential-B51qAYE4.js.map → sequential-B5gXgcyp.js.map} +1 -1
- package/dist/{skillopt-optimization-method-f7399oGb.js → skillopt-optimization-method-CV7go7ex.js} +6 -749
- package/dist/skillopt-optimization-method-CV7go7ex.js.map +1 -0
- package/dist/{statistical-heldout-Cqb73yE9.d.ts → statistical-heldout-Z9NROFFS.d.ts} +156 -3
- package/dist/statistical-heldout-Z9NROFFS.d.ts.map +1 -0
- package/dist/{summary-report-Bgh8CpNK.js → summary-report-B16xy9Kd.js} +2 -2
- package/dist/{summary-report-Bgh8CpNK.js.map → summary-report-B16xy9Kd.js.map} +1 -1
- package/dist/traces.js +1 -1
- package/docs/public-api.md +62 -39
- package/docs/search-history-receipts.md +48 -1
- package/package.json +1 -1
- package/dist/attestation-XSUpbc4o.js.map +0 -1
- package/dist/attestation-c1QvaBdX.d.ts.map +0 -1
- package/dist/define-agent-eval-V1jQyCDR.d.ts.map +0 -1
- package/dist/define-agent-eval-ox5McL6e.js.map +0 -1
- package/dist/llm-judge-DmNaBrXB.js.map +0 -1
- package/dist/power-preflight-CFXm0Vjo.js +0 -502
- package/dist/power-preflight-CFXm0Vjo.js.map +0 -1
- package/dist/pre-registration-D94b7Of5.js +0 -110
- package/dist/pre-registration-D94b7Of5.js.map +0 -1
- package/dist/profile-cell.js.map +0 -1
- package/dist/reward-hacking-CKW4teig.js.map +0 -1
- package/dist/skillopt-optimization-method-f7399oGb.js.map +0 -1
- package/dist/statistical-heldout-Cqb73yE9.d.ts.map +0 -1
|
@@ -0,0 +1,2083 @@
|
|
|
1
|
+
import { t as AgentEvalError } from "./errors-Dngq5h35.js";
|
|
2
|
+
import { a as hashCanonical, i as compareCodeUnits, r as canonicalString } from "./canonical-DPyQ_rpt.js";
|
|
3
|
+
import { g as pairedRiskDifferenceScore, h as pairedRiskDifferenceExact, p as pairedBinaryScale } from "./paired-arms-D4aeIHUy.js";
|
|
4
|
+
import { a as pairedSignTest, i as pairedDeltaTieFraction, r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
|
|
5
|
+
import { n as contentHash } from "./verdict-cache-B3eCVQtY.js";
|
|
6
|
+
import { n as campaignCellExecutionEvidence, o as projectCampaignCellQuality } from "./run-record-CR63CpHK.js";
|
|
7
|
+
import { createHash } from "node:crypto";
|
|
8
|
+
import { join } from "node:path";
|
|
9
|
+
//#region src/campaign/coverage.ts
|
|
10
|
+
/** Reject campaign designs whose denominator cannot be identified exactly. */
|
|
11
|
+
function assertCampaignDesign(scenarios, reps) {
|
|
12
|
+
if (!Number.isSafeInteger(reps) || reps < 1) throw new Error("campaign design requires reps to be a positive safe integer");
|
|
13
|
+
const scenarioIds = /* @__PURE__ */ new Set();
|
|
14
|
+
for (const scenario of scenarios) {
|
|
15
|
+
if (typeof scenario.id !== "string" || scenario.id.trim().length === 0) throw new Error("campaign design requires every scenario to have a non-empty id");
|
|
16
|
+
if (scenarioIds.has(scenario.id)) throw new Error(`campaign design contains duplicate scenario id '${scenario.id}'`);
|
|
17
|
+
if (typeof scenario.kind !== "string" || scenario.kind.trim().length === 0) throw new Error("campaign design requires every scenario to have a non-empty kind");
|
|
18
|
+
if (scenario.seedGroup !== void 0 && (typeof scenario.seedGroup !== "string" || scenario.seedGroup.trim().length === 0)) throw new Error("campaign design requires seedGroup to be a non-empty string when set");
|
|
19
|
+
scenarioIds.add(scenario.id);
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
/** Redacted but independently verifiable identity of one complete scenario. */
|
|
23
|
+
function campaignScenarioIdentity(scenario) {
|
|
24
|
+
assertCampaignDesign([scenario], 1);
|
|
25
|
+
return {
|
|
26
|
+
id: scenario.id,
|
|
27
|
+
kind: scenario.kind,
|
|
28
|
+
scenarioDigest: `sha256:${contentHash(scenario)}`
|
|
29
|
+
};
|
|
30
|
+
}
|
|
31
|
+
/** Canonical split identity reconstructed from redacted scenario identities. */
|
|
32
|
+
function campaignSplitDigestFromIdentities(scenarios, reps) {
|
|
33
|
+
assertCampaignDesign(scenarios, reps);
|
|
34
|
+
for (const scenario of scenarios) if (!/^sha256:[a-f0-9]{64}$/.test(scenario.scenarioDigest)) throw new Error(`campaign scenario '${scenario.id}' has an invalid digest`);
|
|
35
|
+
return `sha256:${contentHash({
|
|
36
|
+
schema: "tangle.campaign-split",
|
|
37
|
+
scenarios: scenarios.map(({ id, kind, scenarioDigest }) => ({
|
|
38
|
+
id,
|
|
39
|
+
kind,
|
|
40
|
+
scenarioDigest
|
|
41
|
+
})),
|
|
42
|
+
reps
|
|
43
|
+
})}`;
|
|
44
|
+
}
|
|
45
|
+
/** Canonical identity of the exact scenario payloads and replicate count. */
|
|
46
|
+
function campaignSplitDigest(scenarios, reps) {
|
|
47
|
+
assertCampaignDesign(scenarios, reps);
|
|
48
|
+
return campaignSplitDigestFromIdentities(scenarios.map(campaignScenarioIdentity), reps);
|
|
49
|
+
}
|
|
50
|
+
/** Refuse a campaign whose retained task identities contradict its split digest. */
|
|
51
|
+
function assertCampaignSplitIdentity(scenarios, reps, splitDigest) {
|
|
52
|
+
if (campaignSplitDigestFromIdentities(scenarios, reps) !== splitDigest) throw new Error("campaign split digest does not match its retained scenario identities");
|
|
53
|
+
}
|
|
54
|
+
/** Exact designed-denominator receipt for one campaign. */
|
|
55
|
+
function campaignCoverage(cells, scenarios, reps, requireJudgeScore) {
|
|
56
|
+
assertCampaignDesign(scenarios, reps);
|
|
57
|
+
const expectedCellIds = designedCellIds(scenarios, reps);
|
|
58
|
+
const cellsById = /* @__PURE__ */ new Map();
|
|
59
|
+
for (const cell of cells) {
|
|
60
|
+
const matches = cellsById.get(cell.cellId) ?? [];
|
|
61
|
+
matches.push(cell);
|
|
62
|
+
cellsById.set(cell.cellId, matches);
|
|
63
|
+
}
|
|
64
|
+
const scorableCellIds = [];
|
|
65
|
+
const unscorableCells = [];
|
|
66
|
+
for (const cellId of expectedCellIds) {
|
|
67
|
+
const matches = cellsById.get(cellId) ?? [];
|
|
68
|
+
if (matches.length === 0) {
|
|
69
|
+
unscorableCells.push({
|
|
70
|
+
cellId,
|
|
71
|
+
reason: "missing campaign cell"
|
|
72
|
+
});
|
|
73
|
+
continue;
|
|
74
|
+
}
|
|
75
|
+
if (matches.length > 1) {
|
|
76
|
+
unscorableCells.push({
|
|
77
|
+
cellId,
|
|
78
|
+
reason: `duplicate campaign cell (${matches.length})`
|
|
79
|
+
});
|
|
80
|
+
continue;
|
|
81
|
+
}
|
|
82
|
+
const cell = matches[0];
|
|
83
|
+
const scoreEntries = Object.entries(cell.judgeScores);
|
|
84
|
+
const successfulScores = scoreEntries.map(([, score]) => score).filter((score) => score.failed !== true && Number.isFinite(score.composite));
|
|
85
|
+
const nonFiniteScores = scoreEntries.filter(([, score]) => score.failed !== true && (!Number.isFinite(score.composite) || Object.values(score.dimensions).some((value) => !Number.isFinite(value))));
|
|
86
|
+
const reasons = [];
|
|
87
|
+
if (cell.error) reasons.push(cell.error);
|
|
88
|
+
if (cell.artifact === null || cell.artifact === void 0) reasons.push("missing artifact");
|
|
89
|
+
if (!cell.error && requireJudgeScore && successfulScores.length === 0) reasons.push("no successful finite judge score");
|
|
90
|
+
if (scoreEntries.some(([, score]) => score.failed === true)) reasons.push("judge score marked failed");
|
|
91
|
+
const failedPanelJudges = [...new Set(scoreEntries.flatMap(([, score]) => score.failedJudges ?? []))].sort();
|
|
92
|
+
if (failedPanelJudges.length > 0) reasons.push(`judge panel incomplete: ${failedPanelJudges.join(", ")}`);
|
|
93
|
+
if (nonFiniteScores.length > 0) reasons.push(`non-finite judge score: ${nonFiniteScores.map(([name]) => name).sort().join(", ")}`);
|
|
94
|
+
if (reasons.length > 0) unscorableCells.push({
|
|
95
|
+
cellId,
|
|
96
|
+
reason: reasons.join("; ")
|
|
97
|
+
});
|
|
98
|
+
else scorableCellIds.push(cellId);
|
|
99
|
+
}
|
|
100
|
+
const expected = new Set(expectedCellIds);
|
|
101
|
+
for (const cell of cells) {
|
|
102
|
+
if (cell.cellId !== `${cell.scenarioId}:${cell.rep}`) {
|
|
103
|
+
unscorableCells.push({
|
|
104
|
+
cellId: cell.cellId,
|
|
105
|
+
reason: "campaign cell id does not match scenario id and rep"
|
|
106
|
+
});
|
|
107
|
+
continue;
|
|
108
|
+
}
|
|
109
|
+
if (!expected.has(cell.cellId)) unscorableCells.push({
|
|
110
|
+
cellId: cell.cellId,
|
|
111
|
+
reason: "unexpected campaign cell"
|
|
112
|
+
});
|
|
113
|
+
}
|
|
114
|
+
return {
|
|
115
|
+
complete: unscorableCells.length === 0 && scorableCellIds.length === expectedCellIds.length,
|
|
116
|
+
expectedCellIds,
|
|
117
|
+
scorableCellIds,
|
|
118
|
+
unscorableCells
|
|
119
|
+
};
|
|
120
|
+
}
|
|
121
|
+
function formatCoverageFailures(coverage) {
|
|
122
|
+
const shown = coverage.unscorableCells.slice(0, 3).map((cell) => `${cell.cellId}: ${cell.reason}`).join("; ");
|
|
123
|
+
const remainder = coverage.unscorableCells.length - Math.min(3, coverage.unscorableCells.length);
|
|
124
|
+
return remainder > 0 ? `${shown}; +${remainder} more` : shown || "unknown coverage failure";
|
|
125
|
+
}
|
|
126
|
+
function designedCellIds(scenarios, reps) {
|
|
127
|
+
const ids = [];
|
|
128
|
+
for (const scenario of scenarios) for (let rep = 0; rep < reps; rep++) ids.push(`${scenario.id}:${rep}`);
|
|
129
|
+
return ids;
|
|
130
|
+
}
|
|
131
|
+
/** Require the complete designed denominator before a final comparison. */
|
|
132
|
+
function assertCompleteCampaign(campaign, scenarios, reps, requireJudgeScore, label) {
|
|
133
|
+
const coverage = campaignCoverage(campaign.cells, scenarios, reps, requireJudgeScore);
|
|
134
|
+
if (!coverage.complete) throw new Error(`${label} is incomplete (${coverage.scorableCellIds.length}/${coverage.expectedCellIds.length} designed cells scorable) — ${formatCoverageFailures(coverage)}. Refusing to compare unequal results.`);
|
|
135
|
+
}
|
|
136
|
+
//#endregion
|
|
137
|
+
//#region src/integrity/backend-integrity.ts
|
|
138
|
+
/**
|
|
139
|
+
* Error thrown when an integrity assertion fails. Caller can pattern-match
|
|
140
|
+
* by `code === 'AGENT_EVAL_BACKEND_STUB'` to differentiate from other
|
|
141
|
+
* errors.
|
|
142
|
+
*/
|
|
143
|
+
var BackendIntegrityError = class extends AgentEvalError {
|
|
144
|
+
report;
|
|
145
|
+
constructor(message, report) {
|
|
146
|
+
super("backend_integrity", message);
|
|
147
|
+
this.report = report;
|
|
148
|
+
}
|
|
149
|
+
};
|
|
150
|
+
/**
|
|
151
|
+
* Inspect a batch of RunRecords and return an integrity report. Pure
|
|
152
|
+
* function — no I/O, no logging. The caller decides what to do with the
|
|
153
|
+
* verdict (print warning, throw, gate CI, etc.).
|
|
154
|
+
*/
|
|
155
|
+
function summarizeBackendIntegrity(records) {
|
|
156
|
+
return summarizeBackendUsage(records.map((record) => ({
|
|
157
|
+
inputTokens: record.tokenUsage.input,
|
|
158
|
+
outputTokens: record.tokenUsage.output,
|
|
159
|
+
costUsd: record.costUsd
|
|
160
|
+
})));
|
|
161
|
+
}
|
|
162
|
+
/** Inspect settled agent calls from the canonical cost ledger. */
|
|
163
|
+
function summarizeAgentReceiptIntegrity(receipts) {
|
|
164
|
+
return summarizeBackendUsage(receipts.filter((receipt) => receipt.channel === "agent").map((receipt) => ({
|
|
165
|
+
inputTokens: receipt.inputTokens,
|
|
166
|
+
outputTokens: receipt.outputTokens,
|
|
167
|
+
costUsd: receipt.costUsd
|
|
168
|
+
})));
|
|
169
|
+
}
|
|
170
|
+
function summarizeBackendUsage(records) {
|
|
171
|
+
const totalRecords = records.length;
|
|
172
|
+
let stubRecords = 0;
|
|
173
|
+
let realRecords = 0;
|
|
174
|
+
let uncostedRecords = 0;
|
|
175
|
+
let totalInputTokens = 0;
|
|
176
|
+
let totalOutputTokens = 0;
|
|
177
|
+
let totalCostUsd = 0;
|
|
178
|
+
for (const rec of records) {
|
|
179
|
+
totalInputTokens += rec.inputTokens;
|
|
180
|
+
totalOutputTokens += rec.outputTokens;
|
|
181
|
+
totalCostUsd += rec.costUsd ?? 0;
|
|
182
|
+
if (rec.inputTokens === 0 && rec.outputTokens === 0) stubRecords++;
|
|
183
|
+
else realRecords++;
|
|
184
|
+
if (rec.outputTokens > 0 && (rec.costUsd === null || rec.costUsd === 0)) uncostedRecords++;
|
|
185
|
+
}
|
|
186
|
+
const verdict = totalRecords === 0 ? "stub" : stubRecords === totalRecords ? "stub" : stubRecords === 0 ? "real" : "mixed";
|
|
187
|
+
const diagnosis = buildDiagnosis({
|
|
188
|
+
totalRecords,
|
|
189
|
+
stubRecords,
|
|
190
|
+
realRecords,
|
|
191
|
+
uncostedRecords,
|
|
192
|
+
totalInputTokens,
|
|
193
|
+
totalOutputTokens,
|
|
194
|
+
totalCostUsd,
|
|
195
|
+
verdict
|
|
196
|
+
});
|
|
197
|
+
return {
|
|
198
|
+
totalRecords,
|
|
199
|
+
stubRecords,
|
|
200
|
+
realRecords,
|
|
201
|
+
uncostedRecords,
|
|
202
|
+
totalInputTokens,
|
|
203
|
+
totalOutputTokens,
|
|
204
|
+
totalCostUsd,
|
|
205
|
+
verdict,
|
|
206
|
+
diagnosis
|
|
207
|
+
};
|
|
208
|
+
}
|
|
209
|
+
function buildDiagnosis(r) {
|
|
210
|
+
if (r.totalRecords === 0) return "no records — eval produced zero runs; backend likely failed before first turn";
|
|
211
|
+
if (r.verdict === "stub") return [
|
|
212
|
+
`all ${r.totalRecords} records have zero token usage — the LLM backend was never called.`,
|
|
213
|
+
"common causes: --backend sandbox without a sandbox bridge running; stub model returning hard-coded strings;",
|
|
214
|
+
"auth misconfigured so requests were silently dropped before the LLM. Re-run with --backend tcloud and TANGLE_API_KEY set,",
|
|
215
|
+
"or boot the cli-bridge / sandbox before invoking the eval."
|
|
216
|
+
].join(" ");
|
|
217
|
+
if (r.verdict === "mixed") {
|
|
218
|
+
const pct = (r.stubRecords / r.totalRecords * 100).toFixed(0);
|
|
219
|
+
return [
|
|
220
|
+
`${r.stubRecords}/${r.totalRecords} records (${pct}%) have zero token usage — the backend partially failed.`,
|
|
221
|
+
"common causes: rate-limit cascade (429s after the first N personas);",
|
|
222
|
+
"transient auth expiry mid-run; provider outage. Treat the affected records as missing data, not agent failures."
|
|
223
|
+
].join(" ");
|
|
224
|
+
}
|
|
225
|
+
if (r.uncostedRecords > 0) {
|
|
226
|
+
const pct = (r.uncostedRecords / r.totalRecords * 100).toFixed(0);
|
|
227
|
+
return [
|
|
228
|
+
`${r.totalRecords} records with real LLM activity (in=${r.totalInputTokens}, out=${r.totalOutputTokens} tokens).`,
|
|
229
|
+
`${r.uncostedRecords} (${pct}%) have output tokens but costUsd=0. Two distinct roots:`,
|
|
230
|
+
"(a) cost ledger mis-wired — no usage propagation from the runtime stream into RunRecord; or",
|
|
231
|
+
"(b) the model is unpriced at the source (sandbox/router returned $0 despite real tokens).",
|
|
232
|
+
"For (b), price the measured tokens against the substrate table (estimateCost) instead of leaving $0."
|
|
233
|
+
].join(" ");
|
|
234
|
+
}
|
|
235
|
+
return `${r.totalRecords} records with real LLM activity (in=${r.totalInputTokens}, out=${r.totalOutputTokens} tokens, $${r.totalCostUsd.toFixed(4)}).`;
|
|
236
|
+
}
|
|
237
|
+
/**
|
|
238
|
+
* Throw BackendIntegrityError if the verdict is 'stub' — i.e. every record
|
|
239
|
+
* shows zero LLM activity. Non-strict callers can pass `{ allowMixed: false }`
|
|
240
|
+
* to also reject mixed verdicts (recommended for CI gates).
|
|
241
|
+
*
|
|
242
|
+
* Real backends pass through silently.
|
|
243
|
+
*/
|
|
244
|
+
function assertRealBackend(records, opts = {}) {
|
|
245
|
+
return assertBackendReport(summarizeBackendIntegrity(records), opts);
|
|
246
|
+
}
|
|
247
|
+
function assertBackendReport(report, opts) {
|
|
248
|
+
const allowMixed = opts.allowMixed ?? true;
|
|
249
|
+
if (report.verdict === "stub") throw new BackendIntegrityError(`backend-integrity: ran against a stub or unconfigured backend — ${report.diagnosis}`, report);
|
|
250
|
+
if (!allowMixed && report.verdict === "mixed") throw new BackendIntegrityError(`backend-integrity: partial backend failure rejected — ${report.diagnosis}`, report);
|
|
251
|
+
return report;
|
|
252
|
+
}
|
|
253
|
+
//#endregion
|
|
254
|
+
//#region src/campaign/surface-identity.ts
|
|
255
|
+
const GIT_OBJECT_ID = /^(?:[a-f0-9]{40}|[a-f0-9]{64})$/;
|
|
256
|
+
const SHA256 = /^sha256:[a-f0-9]{64}$/;
|
|
257
|
+
/** Validate the immutable identity shape; the owning executor verifies the Git objects and patch. */
|
|
258
|
+
function assertCodeSurfaceIdentity(surface) {
|
|
259
|
+
if (!surface || typeof surface !== "object") throw new TypeError("CodeSurface must be an object");
|
|
260
|
+
const candidate = surface;
|
|
261
|
+
if (candidate.kind !== "code") throw new TypeError("CodeSurface.kind must be \"code\"");
|
|
262
|
+
if (typeof candidate.worktreeRef !== "string" || candidate.worktreeRef.trim().length === 0) throw new TypeError("CodeSurface.worktreeRef must be a non-empty locator");
|
|
263
|
+
if (typeof candidate.baseRef !== "string" || candidate.baseRef.trim().length === 0) throw new TypeError("CodeSurface.baseRef must be a non-empty ref label");
|
|
264
|
+
for (const [field, value] of [
|
|
265
|
+
["baseCommit", candidate.baseCommit],
|
|
266
|
+
["baseTree", candidate.baseTree],
|
|
267
|
+
["candidateCommit", candidate.candidateCommit],
|
|
268
|
+
["candidateTree", candidate.candidateTree]
|
|
269
|
+
]) if (typeof value !== "string" || !GIT_OBJECT_ID.test(value)) throw new TypeError(`CodeSurface.${field} must be a full Git object id`);
|
|
270
|
+
const patch = candidate.patch;
|
|
271
|
+
if (!patch || typeof patch !== "object" || patch.format !== "git-diff-binary") throw new TypeError("CodeSurface.patch.format must be \"git-diff-binary\"");
|
|
272
|
+
if (typeof patch.sha256 !== "string" || !SHA256.test(patch.sha256)) throw new TypeError("CodeSurface.patch.sha256 must be a sha256 digest");
|
|
273
|
+
if (!Number.isSafeInteger(patch.byteLength) || patch.byteLength < 0) throw new TypeError("CodeSurface.patch.byteLength must be a non-negative safe integer");
|
|
274
|
+
}
|
|
275
|
+
/** Assert that a value is a valid non-empty component surface. */
|
|
276
|
+
function assertComponentSurface(surface) {
|
|
277
|
+
if (!surface || typeof surface !== "object") throw new TypeError("ComponentSurface must be an object");
|
|
278
|
+
const candidate = surface;
|
|
279
|
+
if (candidate.kind !== "components") throw new TypeError("ComponentSurface.kind must be \"components\"");
|
|
280
|
+
if (!candidate.components || typeof candidate.components !== "object" || Array.isArray(candidate.components)) throw new TypeError("ComponentSurface.components must be an object");
|
|
281
|
+
const entries = Object.entries(candidate.components);
|
|
282
|
+
if (entries.length === 0) throw new TypeError("ComponentSurface.components must not be empty");
|
|
283
|
+
for (const [name, content] of entries) {
|
|
284
|
+
if (!name.trim() || name.trim() !== name) throw new TypeError("ComponentSurface component names must be trimmed and non-empty");
|
|
285
|
+
if (typeof content !== "string") throw new TypeError(`ComponentSurface component '${name}' must be a string`);
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
/**
|
|
289
|
+
* Deterministic identity material for a component surface.
|
|
290
|
+
*
|
|
291
|
+
* `canonicalString` orders keys by UTF-16 code unit (RFC 8785), which is a
|
|
292
|
+
* property of the value alone. The previous material ordered them with
|
|
293
|
+
* `localeCompare`, which reads the host's collation — so the same surface
|
|
294
|
+
* could produce two different identities on two machines, and the stored
|
|
295
|
+
* identity would stop matching a recomputation of the identical surface.
|
|
296
|
+
*/
|
|
297
|
+
function componentSurfaceIdentityMaterial(surface) {
|
|
298
|
+
assertComponentSurface(surface);
|
|
299
|
+
return canonicalString({
|
|
300
|
+
schema: "tangle.component-surface",
|
|
301
|
+
components: surface.components
|
|
302
|
+
});
|
|
303
|
+
}
|
|
304
|
+
/**
|
|
305
|
+
* The retired material builder, kept PRIVATE and reachable only from
|
|
306
|
+
* {@link surfaceHashMatches}.
|
|
307
|
+
*
|
|
308
|
+
* A surface identity recorded before this release was minted from these bytes.
|
|
309
|
+
* The verify path tries the current material first and falls back to this one,
|
|
310
|
+
* so a stored identity still matches its own surface; nothing mints from it.
|
|
311
|
+
*
|
|
312
|
+
* Every component surface's identity moves, not only one whose names sort
|
|
313
|
+
* differently under the host's collation: RFC 8785 also orders the two
|
|
314
|
+
* top-level keys, so `components` precedes `schema` where this builder emitted
|
|
315
|
+
* them in literal order. The retention window therefore covers every stored
|
|
316
|
+
* component-surface identity, which is why this builder is kept rather than
|
|
317
|
+
* scoped to the mixed-case case.
|
|
318
|
+
*/
|
|
319
|
+
function retiredComponentSurfaceIdentityMaterial(surface) {
|
|
320
|
+
assertComponentSurface(surface);
|
|
321
|
+
return JSON.stringify({
|
|
322
|
+
schema: "tangle.component-surface",
|
|
323
|
+
components: Object.fromEntries(Object.entries(surface.components).sort(([left], [right]) => left.localeCompare(right)))
|
|
324
|
+
});
|
|
325
|
+
}
|
|
326
|
+
/** Canonical, location-independent identity of a finalized code candidate.
|
|
327
|
+
* Commit metadata is excluded: two commits with the same base, final tree,
|
|
328
|
+
* and patch bytes are the same executable candidate. */
|
|
329
|
+
function codeSurfaceIdentityMaterial(surface) {
|
|
330
|
+
assertCodeSurfaceIdentity(surface);
|
|
331
|
+
return JSON.stringify({
|
|
332
|
+
schema: "tangle.code-surface",
|
|
333
|
+
baseCommit: surface.baseCommit,
|
|
334
|
+
baseTree: surface.baseTree,
|
|
335
|
+
candidateTree: surface.candidateTree,
|
|
336
|
+
patch: {
|
|
337
|
+
format: surface.patch.format,
|
|
338
|
+
sha256: surface.patch.sha256,
|
|
339
|
+
byteLength: surface.patch.byteLength
|
|
340
|
+
}
|
|
341
|
+
});
|
|
342
|
+
}
|
|
343
|
+
/** Full SHA-256 content identity for a prompt or finalized code surface. */
|
|
344
|
+
function surfaceContentHash(surface) {
|
|
345
|
+
const material = typeof surface === "string" ? surface : surface.kind === "components" ? componentSurfaceIdentityMaterial(surface) : codeSurfaceIdentityMaterial(surface);
|
|
346
|
+
return `sha256:${createHash("sha256").update(material).digest("hex")}`;
|
|
347
|
+
}
|
|
348
|
+
/** Short loop key derived from the same content identity as provenance. */
|
|
349
|
+
function surfaceHash(surface) {
|
|
350
|
+
return surfaceContentHash(surface).slice(7, 23);
|
|
351
|
+
}
|
|
352
|
+
/**
|
|
353
|
+
* Whether `storedHash` is the loop key of `surface`, under the current identity
|
|
354
|
+
* material or the retired one.
|
|
355
|
+
*
|
|
356
|
+
* A stored key is 16 hex characters with no room for a scheme tag, so the
|
|
357
|
+
* scheme cannot be read off the value the way an `agent-profile-cell` id names
|
|
358
|
+
* its own. The verify path therefore tries both, which gives the same property:
|
|
359
|
+
* a key minted by an earlier release still matches its own surface, and a
|
|
360
|
+
* surface that was actually edited matches neither.
|
|
361
|
+
*
|
|
362
|
+
* Only a component surface can differ between the two; a prompt or code surface
|
|
363
|
+
* produces identical material under both, so the second comparison is a no-op
|
|
364
|
+
* for them.
|
|
365
|
+
*/
|
|
366
|
+
function surfaceHashMatches(surface, storedHash) {
|
|
367
|
+
if (surfaceHash(surface) === storedHash) return true;
|
|
368
|
+
if (typeof surface === "string" || surface.kind !== "components") return false;
|
|
369
|
+
return createHash("sha256").update(retiredComponentSurfaceIdentityMaterial(surface)).digest("hex").slice(0, 16) === storedHash;
|
|
370
|
+
}
|
|
371
|
+
/** Canonical customer-visible description of the exact before/after surfaces. */
|
|
372
|
+
function renderSurfaceDiff(winnerSurface, baselineSurface) {
|
|
373
|
+
if (typeof winnerSurface === "string" && typeof baselineSurface === "string") return [
|
|
374
|
+
"--- baseline",
|
|
375
|
+
"+++ winner",
|
|
376
|
+
...baselineSurface.split("\n").map((line) => `- ${line}`),
|
|
377
|
+
...winnerSurface.split("\n").map((line) => `+ ${line}`)
|
|
378
|
+
].join("\n");
|
|
379
|
+
const describe = (surface) => {
|
|
380
|
+
if (typeof surface === "string") return "(prompt surface)";
|
|
381
|
+
if (surface.kind === "components") {
|
|
382
|
+
assertComponentSurface(surface);
|
|
383
|
+
return Object.entries(surface.components).sort(([left], [right]) => left.localeCompare(right)).map(([name, content]) => `[${name}]\n${content}`).join("\n\n");
|
|
384
|
+
}
|
|
385
|
+
assertCodeSurfaceIdentity(surface);
|
|
386
|
+
return [
|
|
387
|
+
`baseCommit=${surface.baseCommit}`,
|
|
388
|
+
`baseTree=${surface.baseTree}`,
|
|
389
|
+
`candidateCommit=${surface.candidateCommit}`,
|
|
390
|
+
`candidateTree=${surface.candidateTree}`,
|
|
391
|
+
`patch=${surface.patch.sha256}`,
|
|
392
|
+
`patchBytes=${surface.patch.byteLength}`,
|
|
393
|
+
...surface.summary ? [surface.summary] : []
|
|
394
|
+
].join("\n");
|
|
395
|
+
};
|
|
396
|
+
return `--- baseline\n${describe(baselineSurface)}\n+++ winner\n${describe(winnerSurface)}`;
|
|
397
|
+
}
|
|
398
|
+
/** Bind a campaign cache entry to the exact surface and caller-owned execution revision. */
|
|
399
|
+
function surfaceDispatchRef(surface, executionRef = "anonymous") {
|
|
400
|
+
if (!executionRef.trim() || executionRef.trim() !== executionRef) throw new Error("surfaceDispatchRef: executionRef must be trimmed and non-empty");
|
|
401
|
+
return `surface:${executionRef}:${surfaceContentHash(surface)}`;
|
|
402
|
+
}
|
|
403
|
+
//#endregion
|
|
404
|
+
//#region src/paired-delta-test.ts
|
|
405
|
+
/** Smallest all-positive sample that can clear a one-sided exact sign test. */
|
|
406
|
+
function minimumPairsForPairedDeltaTest(confidence = .95) {
|
|
407
|
+
if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`);
|
|
408
|
+
const oneSidedAlpha = (1 - confidence) / 2;
|
|
409
|
+
return Math.ceil(Math.log2(1 / oneSidedAlpha));
|
|
410
|
+
}
|
|
411
|
+
/**
|
|
412
|
+
* Tests whether a paired candidate-minus-baseline delta clears a threshold.
|
|
413
|
+
*
|
|
414
|
+
* At 20 or more pairs, the percentile bootstrap lower bound carries the
|
|
415
|
+
* decision. Below that point the interval is descriptive only, so the function
|
|
416
|
+
* switches to a pre-registered one-sided exact sign test. The exact path is
|
|
417
|
+
* deliberately conservative: it requires both a point estimate above the
|
|
418
|
+
* threshold and enough consistently positive paired differences.
|
|
419
|
+
*
|
|
420
|
+
* ## A zero-width interval is never significant
|
|
421
|
+
*
|
|
422
|
+
* When every paired delta is identical the resample distribution is a point
|
|
423
|
+
* mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n
|
|
424
|
+
* identical deltas of g. Neither says the effect is certain — both say the
|
|
425
|
+
* sample carries no information about how far the estimate could be wrong, and
|
|
426
|
+
* `low > threshold` then answers on the point estimate alone. It fails in both
|
|
427
|
+
* directions: `[0, 0]` clears every NEGATIVE threshold, which is how a
|
|
428
|
+
* tie-dominated pass/fail comparison laundered a regression into a
|
|
429
|
+
* noninferiority pass, and `[g, g]` clears every threshold below g with no
|
|
430
|
+
* spread behind it. Under a bounded asymmetric null whose true mean paired
|
|
431
|
+
* delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —
|
|
432
|
+
* every sample that misses the drop is exactly that shape, and deciding on
|
|
433
|
+
* `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.
|
|
434
|
+
*
|
|
435
|
+
* So `indeterminate` is reported and `significant` is false whenever the
|
|
436
|
+
* interval has zero width, on BOTH paths: at small n the exact sign test is a
|
|
437
|
+
* test of the MEDIAN and a zero-spread sample is precisely where it stops
|
|
438
|
+
* saying anything about the mean the caller is thresholding.
|
|
439
|
+
*
|
|
440
|
+
* `threshold` may be negative — that is a noninferiority margin, and it is the
|
|
441
|
+
* regime the zero-width hole is worst in. For a two-point (pass/fail) outcome
|
|
442
|
+
* the percentile bootstrap is not a valid interval at a nonzero margin at all;
|
|
443
|
+
* use {@link decidePairedPromotion}, which routes those to Tango's score
|
|
444
|
+
* interval, rather than thresholding this function's bootstrap directly.
|
|
445
|
+
*/
|
|
446
|
+
function pairedDeltaTest(before, after, options = {}) {
|
|
447
|
+
const threshold = options.threshold ?? 0;
|
|
448
|
+
if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
|
|
449
|
+
const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
|
|
450
|
+
const requestedMinimum = options.minPairs ?? exactMinimum;
|
|
451
|
+
if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
|
|
452
|
+
const minimumPairs = Math.max(requestedMinimum, exactMinimum);
|
|
453
|
+
const bootstrap = pairedBootstrap(before, after, options);
|
|
454
|
+
const sufficient = bootstrap.n >= minimumPairs;
|
|
455
|
+
const indeterminate = bootstrap.n > 0 && (!Number.isFinite(bootstrap.low) || !Number.isFinite(bootstrap.high) || bootstrap.low === bootstrap.high);
|
|
456
|
+
if (bootstrap.gateEligible) return {
|
|
457
|
+
bootstrap,
|
|
458
|
+
method: "bootstrap-ci",
|
|
459
|
+
pValue: null,
|
|
460
|
+
minimumPairs,
|
|
461
|
+
sufficient,
|
|
462
|
+
indeterminate,
|
|
463
|
+
significant: sufficient && !indeterminate && bootstrap.low > threshold
|
|
464
|
+
};
|
|
465
|
+
const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
|
|
466
|
+
const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
|
|
467
|
+
return {
|
|
468
|
+
bootstrap,
|
|
469
|
+
method: "exact-sign",
|
|
470
|
+
pValue: exact.pValue,
|
|
471
|
+
minimumPairs,
|
|
472
|
+
sufficient,
|
|
473
|
+
indeterminate,
|
|
474
|
+
significant: sufficient && !indeterminate && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
|
|
475
|
+
};
|
|
476
|
+
}
|
|
477
|
+
//#endregion
|
|
478
|
+
//#region src/paired-promotion-decision.ts
|
|
479
|
+
/**
|
|
480
|
+
* @module
|
|
481
|
+
* ONE rule for "does this paired interval clear a promotion threshold".
|
|
482
|
+
*
|
|
483
|
+
* The rule below was derived on `HeldOutGate` (#479) after the same estimator
|
|
484
|
+
* bug shipped twice. It then turned out that a SECOND gate — the composable
|
|
485
|
+
* `heldOutGate`, plus everything else routed through `heldoutSignificance` —
|
|
486
|
+
* still carried the original defect, because the rule had been written into one
|
|
487
|
+
* gate's method body rather than into a shared function. Two copies of a
|
|
488
|
+
* statistical rule is how a defect survives in one of them, so there is now
|
|
489
|
+
* exactly one copy and both gates call it.
|
|
490
|
+
*
|
|
491
|
+
* Three things the rule does that a bare `pairedBootstrap(...).low > threshold`
|
|
492
|
+
* does not:
|
|
493
|
+
*
|
|
494
|
+
* 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a
|
|
495
|
+
* pass/fail eval the paired delta vector is dominated by ties, so the
|
|
496
|
+
* bootstrap of the mean is a resample of a lattice with three atoms and its
|
|
497
|
+
* percentile interval is not valid at a nonzero margin. The score interval
|
|
498
|
+
* (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under
|
|
499
|
+
* each hypothesised margin instead of fixing it at the observed value, which
|
|
500
|
+
* is the only construction that stays a confidence interval as the margin
|
|
501
|
+
* moves off zero — the regime every noninferiority threshold lives in.
|
|
502
|
+
* Measured on the composable gate before this change, at a true risk
|
|
503
|
+
* difference sitting exactly on the production caller's -0.05 margin and a
|
|
504
|
+
* nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.
|
|
505
|
+
* 2. **McNemar's exact test holds a VETO at every non-negative threshold.**
|
|
506
|
+
* Redundant with the interval by construction and kept anyway, so that
|
|
507
|
+
* swapping the estimator for one without that duality cannot silently
|
|
508
|
+
* reintroduce "promotes what the exact test refuses". Witness: n = 6, b = 5,
|
|
509
|
+
* c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs
|
|
510
|
+
* (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE
|
|
511
|
+
* threshold is a noninferiority question, which McNemar's test of "no
|
|
512
|
+
* difference" is not the right test for, so the veto does not apply there.
|
|
513
|
+
* 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it
|
|
514
|
+
* cannot tell a gain from a regression and clears every negative threshold.
|
|
515
|
+
* Away from zero it fails the opposite way: n identical positive deltas give
|
|
516
|
+
* [g, g], which clears threshold 0 on no spread at all. Both are an absence
|
|
517
|
+
* of evidence. Measured on the composable gate before this change, under a
|
|
518
|
+
* bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %
|
|
519
|
+
* false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.
|
|
520
|
+
*
|
|
521
|
+
* Orthogonal to the small-sample switch inside {@link pairedDeltaTest}: that
|
|
522
|
+
* picks the TEST from the sample size (bootstrap CI at n >= 20, pre-registered
|
|
523
|
+
* exact sign test below it), this picks the ESTIMATOR from the outcome's shape.
|
|
524
|
+
* Both are needed — an exact sign test applied to a tie-pinned median is still
|
|
525
|
+
* blind, and a mean bootstrap CI at n = 6 is still not a valid test.
|
|
526
|
+
*/
|
|
527
|
+
/**
|
|
528
|
+
* Which estimator {@link decidePairedPromotion} would use on this data, and the
|
|
529
|
+
* shape facts behind it — for callers that must report the shape on a path
|
|
530
|
+
* where no interval is computed at all (an early rejection, or zero pairs).
|
|
531
|
+
* Cheap: no bootstrap, no interval.
|
|
532
|
+
*/
|
|
533
|
+
function pairedDecisionShape(before, after, statistic = "mean") {
|
|
534
|
+
const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after);
|
|
535
|
+
if (statistic === "median") return {
|
|
536
|
+
statistic: "median_bootstrap",
|
|
537
|
+
binaryScale: null,
|
|
538
|
+
tieFraction
|
|
539
|
+
};
|
|
540
|
+
const binaryScale = pairedBinaryScale(before, after);
|
|
541
|
+
if (binaryScale !== null) return {
|
|
542
|
+
statistic: "paired_risk_difference",
|
|
543
|
+
binaryScale,
|
|
544
|
+
tieFraction
|
|
545
|
+
};
|
|
546
|
+
return {
|
|
547
|
+
statistic: "mean_bootstrap",
|
|
548
|
+
binaryScale: null,
|
|
549
|
+
tieFraction
|
|
550
|
+
};
|
|
551
|
+
}
|
|
552
|
+
/**
|
|
553
|
+
* Decide whether a paired candidate-minus-baseline delta clears a promotion
|
|
554
|
+
* threshold. `before` is the baseline arm, `after` the candidate arm, paired by
|
|
555
|
+
* position. Throws on unequal lengths.
|
|
556
|
+
*/
|
|
557
|
+
function decidePairedPromotion(before, after, options = {}) {
|
|
558
|
+
if (before.length !== after.length) throw new Error(`decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`);
|
|
559
|
+
const threshold = options.threshold ?? 0;
|
|
560
|
+
if (!Number.isFinite(threshold)) throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`);
|
|
561
|
+
const confidence = options.confidence ?? .95;
|
|
562
|
+
const exactMinimum = minimumPairsForPairedDeltaTest(confidence);
|
|
563
|
+
const requestedMinimum = options.minPairs ?? exactMinimum;
|
|
564
|
+
if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`);
|
|
565
|
+
const minimumPairs = Math.max(requestedMinimum, exactMinimum);
|
|
566
|
+
const n = before.length;
|
|
567
|
+
const sufficient = n >= minimumPairs;
|
|
568
|
+
const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic);
|
|
569
|
+
let core;
|
|
570
|
+
if (binaryScale !== null) {
|
|
571
|
+
const unitControl = before.map((v) => v / binaryScale);
|
|
572
|
+
const unitTreatment = after.map((v) => v / binaryScale);
|
|
573
|
+
const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence);
|
|
574
|
+
const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence);
|
|
575
|
+
const low = score.lower * binaryScale;
|
|
576
|
+
core = {
|
|
577
|
+
statistic: "paired_risk_difference",
|
|
578
|
+
method: "score-interval",
|
|
579
|
+
delta: score.riskDifference * binaryScale,
|
|
580
|
+
low,
|
|
581
|
+
high: score.upper * binaryScale,
|
|
582
|
+
bootstrap: null,
|
|
583
|
+
mcnemar: {
|
|
584
|
+
b: exact.b,
|
|
585
|
+
c: exact.c,
|
|
586
|
+
nDiscordant: exact.nDiscordant,
|
|
587
|
+
pValue: exact.pValue
|
|
588
|
+
},
|
|
589
|
+
pValue: null,
|
|
590
|
+
clearsThreshold: low > threshold,
|
|
591
|
+
label: "success-rate",
|
|
592
|
+
methodDetail: ""
|
|
593
|
+
};
|
|
594
|
+
} else {
|
|
595
|
+
const bootstrapStatistic = options.statistic === "median" ? "median" : "mean";
|
|
596
|
+
const test = pairedDeltaTest(before, after, {
|
|
597
|
+
confidence,
|
|
598
|
+
resamples: options.resamples,
|
|
599
|
+
statistic: bootstrapStatistic,
|
|
600
|
+
seed: options.seed,
|
|
601
|
+
threshold,
|
|
602
|
+
minPairs: options.minPairs
|
|
603
|
+
});
|
|
604
|
+
const ci = test.bootstrap;
|
|
605
|
+
core = {
|
|
606
|
+
statistic: bootstrapStatistic === "mean" ? "mean_bootstrap" : "median_bootstrap",
|
|
607
|
+
method: test.method,
|
|
608
|
+
delta: bootstrapStatistic === "mean" ? ci.mean : ci.median,
|
|
609
|
+
low: ci.low,
|
|
610
|
+
high: ci.high,
|
|
611
|
+
bootstrap: ci,
|
|
612
|
+
mcnemar: null,
|
|
613
|
+
pValue: test.pValue,
|
|
614
|
+
clearsThreshold: test.significant,
|
|
615
|
+
label: bootstrapStatistic,
|
|
616
|
+
methodDetail: test.method === "exact-sign" ? ` Below ${test.minimumPairs} pairs the interval is descriptive only; the decision is the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.` : ""
|
|
617
|
+
};
|
|
618
|
+
}
|
|
619
|
+
const indeterminate = !Number.isFinite(core.low) || !Number.isFinite(core.high) || core.low === core.high;
|
|
620
|
+
const indeterminateCause = !indeterminate ? "" : tieFraction === 1 ? "every paired delta is an exact tie" : core.mcnemar !== null && core.mcnemar.nDiscordant === 0 ? "every pair is concordant (0 discordant pairs)" : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`;
|
|
621
|
+
const exactTestVetoes = core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence);
|
|
622
|
+
return {
|
|
623
|
+
n,
|
|
624
|
+
threshold,
|
|
625
|
+
confidence,
|
|
626
|
+
binaryScale,
|
|
627
|
+
tieFraction,
|
|
628
|
+
minimumPairs,
|
|
629
|
+
sufficient,
|
|
630
|
+
indeterminate,
|
|
631
|
+
indeterminateCause,
|
|
632
|
+
exactTestVetoes,
|
|
633
|
+
promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,
|
|
634
|
+
...core
|
|
635
|
+
};
|
|
636
|
+
}
|
|
637
|
+
function fmt(x) {
|
|
638
|
+
return x.toFixed(4);
|
|
639
|
+
}
|
|
640
|
+
//#endregion
|
|
641
|
+
//#region src/campaign/gates/statistical-heldout.ts
|
|
642
|
+
/**
|
|
643
|
+
* Statistical held-out promotion machinery — the trustworthy core the
|
|
644
|
+
* point-estimate `heldout-delta` gate lacked.
|
|
645
|
+
*
|
|
646
|
+
* The shipped false positive it prevents: a winner re-scored against the
|
|
647
|
+
* baseline on the holdout read run-to-run model NOISE (e.g. 91 vs 95) as a
|
|
648
|
+
* "+4 lift" and shipped, because the gate compared point estimates with no
|
|
649
|
+
* confidence interval. Here we pair candidate vs baseline holdout observations
|
|
650
|
+
* and bootstrap a CI on the paired delta — a candidate ships only when the CI
|
|
651
|
+
* lower bound clears the effect-size threshold (the gain is real at the
|
|
652
|
+
* confidence level, not noise), and is blocked when a critical dimension
|
|
653
|
+
* (e.g. `hallucination_free` for a legal agent) significantly regresses even if
|
|
654
|
+
* the net composite rose (anti-Goodhart).
|
|
655
|
+
*
|
|
656
|
+
* Two traps this module is built around (both produce a NEW false positive if
|
|
657
|
+
* gotten wrong):
|
|
658
|
+
* 1. PAIRING GRANULARITY — pairs by FULL `cellId` (`scenario:rep`), never by
|
|
659
|
+
* `scenarioId` (which averages reps away and destroys the within-pair
|
|
660
|
+
* variance reduction that makes a paired bootstrap tighter than unpaired).
|
|
661
|
+
* One paired observation per cell ⇒ reps multiply n.
|
|
662
|
+
* 2. SCALE — a judge may emit composites/dimensions on [0,1] or 0-100. The
|
|
663
|
+
* threshold + tolerance are interpreted in the judge's NATIVE scale; the
|
|
664
|
+
* per-dimension tolerance auto-scales off the observed baseline magnitudes
|
|
665
|
+
* so `-0.10` on [0,1] doesn't silently become a no-op on a 0-100 dimension.
|
|
666
|
+
*/
|
|
667
|
+
/** Tie fraction at/above which a gate annotates its verdict with the tie share.
|
|
668
|
+
* Tie-domination of the median bites structurally at >= 0.5 (the median is then
|
|
669
|
+
* 0 by construction); 0.4 is a softer warn threshold that flags a run APPROACHING
|
|
670
|
+
* that regime, so an operator sees it before the median goes fully blind. */
|
|
671
|
+
const TIE_WARN_FRACTION = .4;
|
|
672
|
+
/**
|
|
673
|
+
* Pair candidate vs baseline holdout observations by FULL cellId. `select`
|
|
674
|
+
* pulls the scalar from a cell's judge reports (composite, or a named
|
|
675
|
+
* dimension); a cell contributes the mean of `select` across its judges. Cells
|
|
676
|
+
* whose scenario is not in `scenarioIds`, or where `select` is undefined for
|
|
677
|
+
* every judge on either side, are skipped on BOTH sides so the arrays stay
|
|
678
|
+
* paired. Throws when the two maps disagree on which holdout cells exist — a
|
|
679
|
+
* load-bearing invariant: the baseline + winner holdout campaigns run the same
|
|
680
|
+
* scenarios with the same seed base, so their cellIds MUST align; a mismatch
|
|
681
|
+
* means a silent pairing bug, not a soft fallback.
|
|
682
|
+
*/
|
|
683
|
+
function pairHoldout(candidate, baseline, scenarioIds, select) {
|
|
684
|
+
const cellValue = (byCell, cellId) => {
|
|
685
|
+
const scores = byCell.get(cellId);
|
|
686
|
+
if (!scores) return void 0;
|
|
687
|
+
const vals = [];
|
|
688
|
+
for (const s of Object.values(scores)) {
|
|
689
|
+
if (s.failed === true) throw new Error(`pairHoldout: cell '${cellId}' contains a failed judge score`);
|
|
690
|
+
const v = select(s);
|
|
691
|
+
if (typeof v === "number" && !Number.isFinite(v)) throw new Error(`pairHoldout: cell '${cellId}' contains a non-finite selected score`);
|
|
692
|
+
if (typeof v === "number") vals.push(v);
|
|
693
|
+
}
|
|
694
|
+
if (vals.length === 0) return void 0;
|
|
695
|
+
return vals.reduce((a, b) => a + b, 0) / vals.length;
|
|
696
|
+
};
|
|
697
|
+
const inScope = (cellId) => scenarioIds.has(cellId.split(":")[0] ?? "");
|
|
698
|
+
const candCells = [...candidate.keys()].filter(inScope).sort();
|
|
699
|
+
const baseCells = [...baseline.keys()].filter(inScope).sort();
|
|
700
|
+
if (candCells.length !== baseCells.length || candCells.some((c, i) => c !== baseCells[i])) throw new Error(`pairHoldout: candidate/baseline holdout cells do not align — candidate=[${candCells.join(",")}] baseline=[${baseCells.join(",")}]. Both holdout campaigns must run the same scenarios with the same seed base.`);
|
|
701
|
+
const before = [];
|
|
702
|
+
const after = [];
|
|
703
|
+
const cellIds = [];
|
|
704
|
+
for (const cellId of candCells) {
|
|
705
|
+
const b = cellValue(baseline, cellId);
|
|
706
|
+
const a = cellValue(candidate, cellId);
|
|
707
|
+
if (b === void 0 && a === void 0) continue;
|
|
708
|
+
if (b === void 0 || a === void 0) throw new Error(`pairHoldout: cell '${cellId}' has a selected score on only one arm`);
|
|
709
|
+
before.push(b);
|
|
710
|
+
after.push(a);
|
|
711
|
+
cellIds.push(cellId);
|
|
712
|
+
}
|
|
713
|
+
return {
|
|
714
|
+
before,
|
|
715
|
+
after,
|
|
716
|
+
cellIds
|
|
717
|
+
};
|
|
718
|
+
}
|
|
719
|
+
/**
|
|
720
|
+
* Significance of the held-out composite lift: ship only when the lower bound
|
|
721
|
+
* of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
|
|
722
|
+
* 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
|
|
723
|
+
* scale.
|
|
724
|
+
*
|
|
725
|
+
* The decision is delegated whole to {@link decidePairedPromotion}, the one
|
|
726
|
+
* copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
|
|
727
|
+
* also calls. That module's header carries the measurements; the short version
|
|
728
|
+
* is three guards a bare `bootstrap.low > threshold` does not have:
|
|
729
|
+
*
|
|
730
|
+
* - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
|
|
731
|
+
* only paired-binary construction that stays valid at a nonzero margin;
|
|
732
|
+
* - McNemar's exact test VETOES at any non-negative threshold;
|
|
733
|
+
* - a ZERO-WIDTH interval is refused rather than promoted, in either
|
|
734
|
+
* direction — [0,0] clears every negative threshold and [g,g] clears every
|
|
735
|
+
* threshold below g, and both are an absence of evidence, not a result.
|
|
736
|
+
*
|
|
737
|
+
* Measured on this function before those guards landed, at a nominal 5 %:
|
|
738
|
+
* 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
|
|
739
|
+
* and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
|
|
740
|
+
* delta is exactly 0.
|
|
741
|
+
*
|
|
742
|
+
* At small n, where the percentile bootstrap is descriptive only, a
|
|
743
|
+
* pre-registered exact sign test still carries the bootstrap path.
|
|
744
|
+
*/
|
|
745
|
+
function heldoutSignificance(paired, opts = {}) {
|
|
746
|
+
const deltaThreshold = opts.deltaThreshold ?? 0;
|
|
747
|
+
const confidence = opts.confidence ?? .95;
|
|
748
|
+
const resamples = opts.resamples ?? 2e3;
|
|
749
|
+
const seed = opts.seed ?? 1337;
|
|
750
|
+
const statistic = opts.statistic ?? "mean";
|
|
751
|
+
const decision = decidePairedPromotion(paired.before, paired.after, {
|
|
752
|
+
confidence,
|
|
753
|
+
resamples,
|
|
754
|
+
statistic,
|
|
755
|
+
seed,
|
|
756
|
+
threshold: deltaThreshold,
|
|
757
|
+
minPairs: opts.minProductiveRuns
|
|
758
|
+
});
|
|
759
|
+
const bootstrap = decision.bootstrap ?? pairedBootstrap(paired.before, paired.after, {
|
|
760
|
+
confidence,
|
|
761
|
+
resamples,
|
|
762
|
+
statistic,
|
|
763
|
+
seed
|
|
764
|
+
});
|
|
765
|
+
const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
|
|
766
|
+
confidence,
|
|
767
|
+
resamples,
|
|
768
|
+
statistic: "median",
|
|
769
|
+
seed
|
|
770
|
+
});
|
|
771
|
+
const n = paired.before.length;
|
|
772
|
+
let ties = 0;
|
|
773
|
+
for (let i = 0; i < n; i += 1) {
|
|
774
|
+
const after = paired.after[i] ?? 0;
|
|
775
|
+
const before = paired.before[i] ?? 0;
|
|
776
|
+
if (Math.abs(after - before) < 1e-9) ties += 1;
|
|
777
|
+
}
|
|
778
|
+
const tieFraction = n === 0 ? 0 : ties / n;
|
|
779
|
+
return {
|
|
780
|
+
paired,
|
|
781
|
+
bootstrap,
|
|
782
|
+
medianBootstrap,
|
|
783
|
+
decision,
|
|
784
|
+
decisionStatistic: decision.statistic,
|
|
785
|
+
mcnemar: decision.mcnemar,
|
|
786
|
+
tieFraction,
|
|
787
|
+
n,
|
|
788
|
+
minimumRequired: decision.minimumPairs,
|
|
789
|
+
decisionMethod: decision.method,
|
|
790
|
+
pValue: decision.pValue,
|
|
791
|
+
significant: decision.promote,
|
|
792
|
+
fewRuns: !decision.sufficient
|
|
793
|
+
};
|
|
794
|
+
}
|
|
795
|
+
/** Detect the native scale of a set of scores: 0-100 when any magnitude clears
|
|
796
|
+
* 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
|
|
797
|
+
* expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
|
|
798
|
+
function detectScale(values) {
|
|
799
|
+
return values.some((v) => Math.abs(v) > 1.5) ? 100 : 1;
|
|
800
|
+
}
|
|
801
|
+
/** Per-critical-dimension regression guard. For each dimension, pair the
|
|
802
|
+
* candidate vs baseline values by full cellId and bootstrap the paired delta;
|
|
803
|
+
* a dimension is "regressed" when the CI lower bound < −tolerance (conservative
|
|
804
|
+
* — blocks if the credible worst case exceeds tolerance, which is the right
|
|
805
|
+
* posture for safety dimensions like `hallucination_free`). When `tolerance`
|
|
806
|
+
* is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
|
|
807
|
+
*
|
|
808
|
+
* The interval comes from {@link decidePairedPromotion}, so a pass/fail
|
|
809
|
+
* dimension is judged on Tango's score interval rather than a percentile
|
|
810
|
+
* bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
|
|
811
|
+
* is not a valid interval at one. That matters most here because this guard
|
|
812
|
+
* fails OPEN by construction: `tolerance` is positive, so an interval pinned at
|
|
813
|
+
* [0,0] never satisfies `low < −tolerance` and a real regression on a safety
|
|
814
|
+
* dimension would be reported as `regressed: false`. On the median it fails the
|
|
815
|
+
* same way for the same reason — when most pairs tie, which is automatic for a
|
|
816
|
+
* pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
|
|
817
|
+
* to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
|
|
818
|
+
* restore the pre-0.134 behaviour. */
|
|
819
|
+
function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensions, opts = {}) {
|
|
820
|
+
const out = [];
|
|
821
|
+
for (const dim of criticalDimensions) {
|
|
822
|
+
const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim]);
|
|
823
|
+
if (paired.before.length === 0) continue;
|
|
824
|
+
const tolerance = opts.tolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
|
|
825
|
+
const bootstrapStatistic = opts.statistic ?? "mean";
|
|
826
|
+
const shared = {
|
|
827
|
+
confidence: opts.confidence ?? .95,
|
|
828
|
+
resamples: opts.resamples ?? 2e3,
|
|
829
|
+
statistic: bootstrapStatistic,
|
|
830
|
+
seed: opts.seed ?? 1337
|
|
831
|
+
};
|
|
832
|
+
const guard = decidePairedPromotion(paired.before, paired.after, shared);
|
|
833
|
+
const regression = decidePairedPromotion(paired.after, paired.before, {
|
|
834
|
+
...shared,
|
|
835
|
+
threshold: tolerance
|
|
836
|
+
});
|
|
837
|
+
const bootstrap = guard.bootstrap ?? pairedBootstrap(paired.before, paired.after, shared);
|
|
838
|
+
out.push({
|
|
839
|
+
dimension: dim,
|
|
840
|
+
bootstrap,
|
|
841
|
+
bootstrapStatistic,
|
|
842
|
+
ci: {
|
|
843
|
+
low: guard.low,
|
|
844
|
+
high: guard.high
|
|
845
|
+
},
|
|
846
|
+
decisionStatistic: guard.statistic,
|
|
847
|
+
mcnemar: guard.mcnemar,
|
|
848
|
+
indeterminate: guard.indeterminate,
|
|
849
|
+
regressed: bootstrap.low < -tolerance || regression.promote,
|
|
850
|
+
tolerance,
|
|
851
|
+
n: paired.before.length
|
|
852
|
+
});
|
|
853
|
+
}
|
|
854
|
+
return out;
|
|
855
|
+
}
|
|
856
|
+
//#endregion
|
|
857
|
+
//#region src/campaign/gates/power-preflight.ts
|
|
858
|
+
/** Two-sided z for the common confidence levels; interpolation is overkill here. */
|
|
859
|
+
function zFor(confidence) {
|
|
860
|
+
if (confidence >= .99) return 2.576;
|
|
861
|
+
if (confidence >= .95) return 1.96;
|
|
862
|
+
if (confidence >= .9) return 1.645;
|
|
863
|
+
return 1.282;
|
|
864
|
+
}
|
|
865
|
+
/** Estimate the minimum detectable lift a paired-holdout improvement run can
|
|
866
|
+
* ship at a given budget, from the baseline holdout composites — call it BEFORE
|
|
867
|
+
* spending a search to learn whether the effect you are hunting is even
|
|
868
|
+
* observable at this holdout size and worker variance. */
|
|
869
|
+
function powerPreflight(opts) {
|
|
870
|
+
const composites = opts.baselineComposites.filter((v) => Number.isFinite(v));
|
|
871
|
+
if (composites.length < 3) throw new Error(`powerPreflight: need >= 3 finite baseline composites to estimate variance, got ${composites.length}`);
|
|
872
|
+
const deltaThreshold = opts.deltaThreshold ?? .05;
|
|
873
|
+
const confidence = opts.confidence ?? .95;
|
|
874
|
+
const n = opts.pairedN ?? composites.length;
|
|
875
|
+
if (n < 2) throw new Error(`powerPreflight: pairedN must be >= 2, got ${n}`);
|
|
876
|
+
const mean = composites.reduce((a, b) => a + b, 0) / composites.length;
|
|
877
|
+
const variance = composites.reduce((a, b) => a + (b - mean) * (b - mean), 0) / (composites.length - 1);
|
|
878
|
+
const sd = Math.sqrt(variance);
|
|
879
|
+
const z = zFor(confidence);
|
|
880
|
+
const mde = deltaThreshold + z * Math.SQRT2 * sd / Math.sqrt(n);
|
|
881
|
+
const scaleAssumed = composites.every((v) => v >= -.001 && v <= 1.5);
|
|
882
|
+
const headroom = Math.max(0, 1 - mean);
|
|
883
|
+
const underpowered = scaleAssumed && mde > headroom;
|
|
884
|
+
const sharedChannelCaveat = opts.sharedScorerChannel ? "Holdout and gate share one scoring channel: raising n/reps reduces only idiosyncratic noise — systematic judge bias remains and this MDE is a lower bound. Full debiasing needs an independent second scoring channel (different judge/benchmark family)." : void 0;
|
|
885
|
+
const recommendation = underpowered ? `UNDERPOWERED: minimum detectable lift ${mde.toFixed(3)} exceeds the ${headroom.toFixed(3)} headroom above the baseline (${mean.toFixed(3)}) — no achievable effect can ship at this budget. Raise paired n (scenarios x reps) to ~${Math.ceil((z * Math.SQRT2 * sd / Math.max(headroom - deltaThreshold, .01)) ** 2)} or reduce worker variance before searching.` : `Minimum detectable lift at n=${n}: ${mde.toFixed(3)} (baseline sd ${sd.toFixed(3)}). Effects smaller than this cannot clear the gate; budget the search for effects you believe exceed it.`;
|
|
886
|
+
return {
|
|
887
|
+
n,
|
|
888
|
+
sd,
|
|
889
|
+
mde,
|
|
890
|
+
baselineMean: mean,
|
|
891
|
+
headroom,
|
|
892
|
+
underpowered,
|
|
893
|
+
scaleAssumed,
|
|
894
|
+
deltaThreshold,
|
|
895
|
+
confidence,
|
|
896
|
+
...sharedChannelCaveat ? { sharedChannelCaveat } : {},
|
|
897
|
+
recommendation: sharedChannelCaveat ? `${recommendation} ${sharedChannelCaveat}` : recommendation
|
|
898
|
+
};
|
|
899
|
+
}
|
|
900
|
+
//#endregion
|
|
901
|
+
//#region src/json-recovery.ts
|
|
902
|
+
/**
|
|
903
|
+
* Truncation-tolerant JSON recovery — shared by every parser that reads JSON
|
|
904
|
+
* out of a model response (reflective-mutation proposals, judge scores, the
|
|
905
|
+
* completion-correctness checker).
|
|
906
|
+
*
|
|
907
|
+
* LLMs routinely hit a max_tokens cap mid-emission, leaving a JSON prefix
|
|
908
|
+
* with an unclosed string / object / array and often a dangling key or
|
|
909
|
+
* trailing comma. Throwing on that prefix — and letting the throw fold into
|
|
910
|
+
* a fabricated zero score downstream — is the bug class this module exists
|
|
911
|
+
* to prevent (see `JudgeParseError`'s contract: a synthetic zero is
|
|
912
|
+
* indistinguishable from a real low score). Recovering the complete prefix
|
|
913
|
+
* turns a would-be fabricated zero into a real measurement.
|
|
914
|
+
*/
|
|
915
|
+
/**
|
|
916
|
+
* Walk the input as JSON-aware (string vs not, escape-aware) and close
|
|
917
|
+
* unclosed `{` / `[` in LIFO order at the tail. If the input was already
|
|
918
|
+
* balanced returns it unchanged. If a string was open at end-of-input we
|
|
919
|
+
* also close it with `"` first, since a truncated string-mid-value is the
|
|
920
|
+
* most common LLM cap-hit failure mode and JSON.parse cannot proceed
|
|
921
|
+
* without one.
|
|
922
|
+
*
|
|
923
|
+
* Returns null when the structure is unrecoverable (e.g. depth would go
|
|
924
|
+
* negative — that's an *over*-closed prefix, not a truncation).
|
|
925
|
+
*/
|
|
926
|
+
function autoCloseTruncatedJson(raw) {
|
|
927
|
+
const stack = [];
|
|
928
|
+
let inString = false;
|
|
929
|
+
let escaped = false;
|
|
930
|
+
for (const c of raw) {
|
|
931
|
+
if (escaped) {
|
|
932
|
+
escaped = false;
|
|
933
|
+
continue;
|
|
934
|
+
}
|
|
935
|
+
if (inString) {
|
|
936
|
+
if (c === "\\") {
|
|
937
|
+
escaped = true;
|
|
938
|
+
continue;
|
|
939
|
+
}
|
|
940
|
+
if (c === "\"") {
|
|
941
|
+
inString = false;
|
|
942
|
+
continue;
|
|
943
|
+
}
|
|
944
|
+
continue;
|
|
945
|
+
}
|
|
946
|
+
if (c === "\"") {
|
|
947
|
+
inString = true;
|
|
948
|
+
continue;
|
|
949
|
+
}
|
|
950
|
+
if (c === "{" || c === "[") stack.push(c);
|
|
951
|
+
else if (c === "}") {
|
|
952
|
+
if (stack.pop() !== "{") return null;
|
|
953
|
+
} else if (c === "]") {
|
|
954
|
+
if (stack.pop() !== "[") return null;
|
|
955
|
+
}
|
|
956
|
+
}
|
|
957
|
+
if (stack.length === 0 && !inString) return raw;
|
|
958
|
+
let suffix = "";
|
|
959
|
+
if (escaped) suffix += "\\";
|
|
960
|
+
if (inString) suffix += "\"";
|
|
961
|
+
while (stack.length > 0) {
|
|
962
|
+
const opener = stack.pop();
|
|
963
|
+
suffix += opener === "{" ? "}" : "]";
|
|
964
|
+
}
|
|
965
|
+
return raw + suffix;
|
|
966
|
+
}
|
|
967
|
+
const UNPARSEABLE = Symbol("unparseable");
|
|
968
|
+
function tryParse(candidate) {
|
|
969
|
+
try {
|
|
970
|
+
return JSON.parse(candidate);
|
|
971
|
+
} catch {
|
|
972
|
+
return UNPARSEABLE;
|
|
973
|
+
}
|
|
974
|
+
}
|
|
975
|
+
/** Index of the last `,` that sits outside any string literal, or -1. Cutting
|
|
976
|
+
* there discards a dangling key / half-emitted value at the tail while
|
|
977
|
+
* keeping every complete member before it. */
|
|
978
|
+
function lastCommaOutsideString(s) {
|
|
979
|
+
let inString = false;
|
|
980
|
+
let escaped = false;
|
|
981
|
+
let last = -1;
|
|
982
|
+
for (let i = 0; i < s.length; i++) {
|
|
983
|
+
const c = s[i];
|
|
984
|
+
if (escaped) {
|
|
985
|
+
escaped = false;
|
|
986
|
+
continue;
|
|
987
|
+
}
|
|
988
|
+
if (inString) {
|
|
989
|
+
if (c === "\\") escaped = true;
|
|
990
|
+
else if (c === "\"") inString = false;
|
|
991
|
+
continue;
|
|
992
|
+
}
|
|
993
|
+
if (c === "\"") inString = true;
|
|
994
|
+
else if (c === ",") last = i;
|
|
995
|
+
}
|
|
996
|
+
return last;
|
|
997
|
+
}
|
|
998
|
+
/**
|
|
999
|
+
* Best-effort parse of a possibly-truncated JSON payload embedded in model
|
|
1000
|
+
* output (prose and markdown fences tolerated). Tries each opener position
|
|
1001
|
+
* (earliest `{`/`[` first, then the other when the first yields nothing —
|
|
1002
|
+
* prose like `[note] {"score":3}` must not poison the slice), and from each
|
|
1003
|
+
* start, in order:
|
|
1004
|
+
*
|
|
1005
|
+
* 1. plain `JSON.parse` of the opener → last-closer slice,
|
|
1006
|
+
* 2. auto-closing unclosed structures at the tail
|
|
1007
|
+
* (`autoCloseTruncatedJson`),
|
|
1008
|
+
* 3. trimming the tail back to the previous complete member boundary (the
|
|
1009
|
+
* last comma outside a string) and auto-closing again, repeatedly.
|
|
1010
|
+
*
|
|
1011
|
+
* Recovers e.g. `{"correct": false, "` → `{ correct: false }` — the exact
|
|
1012
|
+
* cap-hit shape that has zeroed real eval rows. Returns the parsed value
|
|
1013
|
+
* (always an object or array, given the slice starts at an opener), or
|
|
1014
|
+
* `null` when nothing parseable can be recovered. Never throws.
|
|
1015
|
+
*/
|
|
1016
|
+
function recoverTruncatedJson(text) {
|
|
1017
|
+
const starts = [text.indexOf("{"), text.indexOf("[")].filter((i) => i >= 0).sort((a, b) => a - b);
|
|
1018
|
+
for (const start of starts) {
|
|
1019
|
+
const recovered = recoverFrom(text.slice(start));
|
|
1020
|
+
if (recovered !== UNPARSEABLE) return recovered;
|
|
1021
|
+
}
|
|
1022
|
+
return null;
|
|
1023
|
+
}
|
|
1024
|
+
function recoverFrom(slice) {
|
|
1025
|
+
let candidate = slice;
|
|
1026
|
+
const lastClose = Math.max(candidate.lastIndexOf("}"), candidate.lastIndexOf("]"));
|
|
1027
|
+
if (lastClose > 0) {
|
|
1028
|
+
const balanced = tryParse(candidate.slice(0, lastClose + 1));
|
|
1029
|
+
if (balanced !== UNPARSEABLE) return balanced;
|
|
1030
|
+
}
|
|
1031
|
+
for (let i = 0; i < 64; i++) {
|
|
1032
|
+
const closed = autoCloseTruncatedJson(candidate);
|
|
1033
|
+
if (closed !== null) {
|
|
1034
|
+
const parsed = tryParse(closed);
|
|
1035
|
+
if (parsed !== UNPARSEABLE) return parsed;
|
|
1036
|
+
}
|
|
1037
|
+
const cut = lastCommaOutsideString(candidate);
|
|
1038
|
+
if (cut <= 0) return UNPARSEABLE;
|
|
1039
|
+
candidate = candidate.slice(0, cut);
|
|
1040
|
+
}
|
|
1041
|
+
return UNPARSEABLE;
|
|
1042
|
+
}
|
|
1043
|
+
//#endregion
|
|
1044
|
+
//#region src/reflective-mutation.ts
|
|
1045
|
+
/**
|
|
1046
|
+
* Reflective mutation — primitives for trace-conditioned prompt rewriting.
|
|
1047
|
+
*
|
|
1048
|
+
* Used by `prompt-evolution.ts` (and any consumer running iterative
|
|
1049
|
+
* improvement). Given a parent prompt + concrete trace evidence (top trials,
|
|
1050
|
+
* bottom trials, missed expectations), produce an LLM-ready prompt that
|
|
1051
|
+
* proposes targeted mutations — not blind rephrasings.
|
|
1052
|
+
*
|
|
1053
|
+
* Why this lives outside `prompt-evolution.ts`: any consumer that wants to
|
|
1054
|
+
* run reflective rewriting WITHOUT the population/Pareto machinery can
|
|
1055
|
+
* import these primitives directly.
|
|
1056
|
+
*
|
|
1057
|
+
* Quality bar (vs. naive "mutate this prompt"):
|
|
1058
|
+
* - Show parent ↔ children diff, not just one variant
|
|
1059
|
+
* - Quote specific missed goldens with their match phrases
|
|
1060
|
+
* - Surface the model's actual emitted output side-by-side with what was expected
|
|
1061
|
+
* - Quote concrete mutation primitives so the model has a vocabulary
|
|
1062
|
+
*/
|
|
1063
|
+
/** Bound on rendered/carried `emitted` evidence. ONE constant shared by the
|
|
1064
|
+
* producer (campaignBreakdown's per-scenario excerpt) and this renderer — if
|
|
1065
|
+
* the two drifted, the tighter side would silently re-clip carried evidence. */
|
|
1066
|
+
const EMITTED_EVIDENCE_MAX_CHARS = 2e3;
|
|
1067
|
+
const DEFAULT_MUTATION_PRIMITIVES = [
|
|
1068
|
+
"Strengthen an imperative (\"should\" → \"must\")",
|
|
1069
|
+
"Add a concrete example pulled from a missed-golden phrase",
|
|
1070
|
+
"Remove a redundant rule that did not improve recall",
|
|
1071
|
+
"Add a counterfactual (\"if X is missing, the score is capped at Y\")",
|
|
1072
|
+
"Reorder sections so the highest-impact rule is first",
|
|
1073
|
+
"Replace abstract language with a domain-specific noun the trial misses"
|
|
1074
|
+
];
|
|
1075
|
+
/**
|
|
1076
|
+
* Build the LLM-ready reflection prompt. Output is plain text — pass it as
|
|
1077
|
+
* the user message. The system message should be small and stable (e.g.
|
|
1078
|
+
* "Output ONLY a JSON object matching the schema below.").
|
|
1079
|
+
*/
|
|
1080
|
+
function buildReflectionPrompt(ctx) {
|
|
1081
|
+
const primitives = ctx.mutationPrimitives ?? DEFAULT_MUTATION_PRIMITIVES;
|
|
1082
|
+
const sections = [];
|
|
1083
|
+
sections.push(`# Mutation target: ${ctx.target}`);
|
|
1084
|
+
sections.push("");
|
|
1085
|
+
sections.push(`You are tuning the prompt component named \`${ctx.target}\`. The current variant is shown below; you have ${ctx.topTrials.length} top trials and ${ctx.bottomTrials.length} bottom trials as evidence. Propose ${ctx.childCount} mutation${ctx.childCount === 1 ? "" : "s"} that fix specific weaknesses visible in the bottom trials. Avoid blank rephrasings.`);
|
|
1086
|
+
sections.push("");
|
|
1087
|
+
sections.push("## Current variant");
|
|
1088
|
+
sections.push("```json");
|
|
1089
|
+
sections.push(JSON.stringify(ctx.parentPayload, null, 2));
|
|
1090
|
+
sections.push("```");
|
|
1091
|
+
sections.push("");
|
|
1092
|
+
if (ctx.bottomTrials.length > 0) {
|
|
1093
|
+
sections.push("## Failures (bottom trials) — what went wrong");
|
|
1094
|
+
sections.push("");
|
|
1095
|
+
for (const trial of ctx.bottomTrials) {
|
|
1096
|
+
sections.push(`### Trial \`${trial.id}\` — score ${trial.score.toFixed(2)}${trial.inputName ? ` (${trial.inputName})` : ""}`);
|
|
1097
|
+
if (trial.failureNote) {
|
|
1098
|
+
sections.push("");
|
|
1099
|
+
sections.push(`**Why it scored low:** ${truncate(trial.failureNote, 1500)}`);
|
|
1100
|
+
}
|
|
1101
|
+
const missed = (trial.expectations ?? []).filter((e) => !e.matched);
|
|
1102
|
+
if (missed.length > 0) {
|
|
1103
|
+
sections.push("");
|
|
1104
|
+
sections.push("**Missed expectations:**");
|
|
1105
|
+
for (const m of missed) sections.push(`- \`${m.id}\`: should match phrase \`${quote(m.phrase)}\``);
|
|
1106
|
+
}
|
|
1107
|
+
if (trial.emitted) {
|
|
1108
|
+
sections.push("");
|
|
1109
|
+
sections.push("**What the agent emitted:**");
|
|
1110
|
+
sections.push("```");
|
|
1111
|
+
sections.push(truncate(trial.emitted, EMITTED_EVIDENCE_MAX_CHARS));
|
|
1112
|
+
sections.push("```");
|
|
1113
|
+
}
|
|
1114
|
+
sections.push("");
|
|
1115
|
+
}
|
|
1116
|
+
}
|
|
1117
|
+
if (ctx.topTrials.length > 0) {
|
|
1118
|
+
sections.push("## Successes (top trials) — what to preserve");
|
|
1119
|
+
sections.push("");
|
|
1120
|
+
for (const trial of ctx.topTrials) sections.push(`- \`${trial.id}\`: score ${trial.score.toFixed(2)}${trial.inputName ? ` (${trial.inputName})` : ""}`);
|
|
1121
|
+
sections.push("");
|
|
1122
|
+
}
|
|
1123
|
+
sections.push("## Allowed mutation primitives");
|
|
1124
|
+
sections.push("");
|
|
1125
|
+
for (const p of primitives) sections.push(`- ${p}`);
|
|
1126
|
+
sections.push("");
|
|
1127
|
+
sections.push("## Output schema");
|
|
1128
|
+
sections.push("");
|
|
1129
|
+
sections.push("Respond with a JSON object — no prose, no markdown fences:");
|
|
1130
|
+
sections.push("```json");
|
|
1131
|
+
sections.push(JSON.stringify({ proposals: [{
|
|
1132
|
+
label: "<short label, ≤ 40 chars>",
|
|
1133
|
+
rationale: "<which failure this targets and which primitive you used>",
|
|
1134
|
+
payload: "<full payload of the new variant — same shape as the current variant>"
|
|
1135
|
+
}] }, null, 2));
|
|
1136
|
+
sections.push("```");
|
|
1137
|
+
return sections.join("\n");
|
|
1138
|
+
}
|
|
1139
|
+
function truncate(s, max) {
|
|
1140
|
+
if (s.length <= max) return s;
|
|
1141
|
+
return `${s.slice(0, max)}… [truncated]`;
|
|
1142
|
+
}
|
|
1143
|
+
function quote(s) {
|
|
1144
|
+
return s.replace(/`/g, "\\`");
|
|
1145
|
+
}
|
|
1146
|
+
/**
|
|
1147
|
+
* Parse the model's JSON response back into proposals. Tolerates markdown
|
|
1148
|
+
* fences and surrounding prose. Returns at most `maxProposals`.
|
|
1149
|
+
*/
|
|
1150
|
+
function parseReflectionResponse(raw, maxProposals) {
|
|
1151
|
+
let text = raw.trim();
|
|
1152
|
+
if (text.startsWith("```")) text = text.replace(/^```(?:json)?\n?/, "").replace(/\n?```$/, "");
|
|
1153
|
+
let parsed = null;
|
|
1154
|
+
const objectStart = text.indexOf("{");
|
|
1155
|
+
const objectEnd = text.lastIndexOf("}");
|
|
1156
|
+
const arrayStart = text.indexOf("[");
|
|
1157
|
+
const arrayEnd = text.lastIndexOf("]");
|
|
1158
|
+
const tryObjectFirst = objectStart >= 0 && (arrayStart < 0 || objectStart < arrayStart);
|
|
1159
|
+
const candidates = [];
|
|
1160
|
+
if (tryObjectFirst) {
|
|
1161
|
+
if (objectStart >= 0 && objectEnd > objectStart) candidates.push(text.slice(objectStart, objectEnd + 1));
|
|
1162
|
+
if (arrayStart >= 0 && arrayEnd > arrayStart) candidates.push(text.slice(arrayStart, arrayEnd + 1));
|
|
1163
|
+
} else {
|
|
1164
|
+
if (arrayStart >= 0 && arrayEnd > arrayStart) candidates.push(text.slice(arrayStart, arrayEnd + 1));
|
|
1165
|
+
if (objectStart >= 0 && objectEnd > objectStart) candidates.push(text.slice(objectStart, objectEnd + 1));
|
|
1166
|
+
}
|
|
1167
|
+
for (const slice of candidates) try {
|
|
1168
|
+
parsed = JSON.parse(slice);
|
|
1169
|
+
break;
|
|
1170
|
+
} catch {}
|
|
1171
|
+
if (parsed == null) for (const slice of candidates) {
|
|
1172
|
+
const closed = autoCloseTruncatedJson(slice);
|
|
1173
|
+
if (closed != null && closed !== slice) try {
|
|
1174
|
+
parsed = JSON.parse(closed);
|
|
1175
|
+
break;
|
|
1176
|
+
} catch {}
|
|
1177
|
+
}
|
|
1178
|
+
if (parsed == null) return [];
|
|
1179
|
+
let proposalsRaw;
|
|
1180
|
+
if (Array.isArray(parsed)) proposalsRaw = parsed;
|
|
1181
|
+
else if (parsed && typeof parsed === "object") proposalsRaw = parsed.proposals;
|
|
1182
|
+
if (!Array.isArray(proposalsRaw)) return [];
|
|
1183
|
+
const out = [];
|
|
1184
|
+
for (const p of proposalsRaw) {
|
|
1185
|
+
if (!p || typeof p !== "object") continue;
|
|
1186
|
+
const obj = p;
|
|
1187
|
+
if (!("payload" in obj)) continue;
|
|
1188
|
+
out.push({
|
|
1189
|
+
label: typeof obj.label === "string" ? obj.label : "mutation",
|
|
1190
|
+
rationale: typeof obj.rationale === "string" ? obj.rationale : "",
|
|
1191
|
+
payload: obj.payload
|
|
1192
|
+
});
|
|
1193
|
+
if (maxProposals !== void 0 && out.length >= maxProposals) break;
|
|
1194
|
+
}
|
|
1195
|
+
return out;
|
|
1196
|
+
}
|
|
1197
|
+
//#endregion
|
|
1198
|
+
//#region src/campaign/score-utils.ts
|
|
1199
|
+
/**
|
|
1200
|
+
* Shared campaign-score reductions used by every optimizer preset
|
|
1201
|
+
* (`runOptimization`, external optimization methods, `compareOptimizationMethods`).
|
|
1202
|
+
* "composite of a campaign" and "per-scenario / per-dimension breakdown" so
|
|
1203
|
+
* the optimizers cannot drift on how a surface's score is computed.
|
|
1204
|
+
*/
|
|
1205
|
+
/** Mean composite across cells with complete task-quality evidence.
|
|
1206
|
+
* Partial judge results remain on their cells but never enter this value.
|
|
1207
|
+
* A campaign with no complete score has no numeric mean and fails loudly. */
|
|
1208
|
+
function campaignMeanComposite(campaign) {
|
|
1209
|
+
const mean = campaignMeanCompositeOrNull(campaign);
|
|
1210
|
+
if (mean === null) throw new Error("campaignMeanComposite: campaign has no complete cell-quality scores");
|
|
1211
|
+
return mean;
|
|
1212
|
+
}
|
|
1213
|
+
/** Nullable campaign mean for wire and report fields that represent missing quality. */
|
|
1214
|
+
function campaignMeanCompositeOrNull(campaign) {
|
|
1215
|
+
const scores = campaign.cells.flatMap((cell) => {
|
|
1216
|
+
const score = projectCampaignCellQuality(cell).score;
|
|
1217
|
+
return score === void 0 ? [] : [score];
|
|
1218
|
+
});
|
|
1219
|
+
return scores.length === 0 ? null : scores.reduce((sum, score) => sum + score, 0) / scores.length;
|
|
1220
|
+
}
|
|
1221
|
+
/** Reject rank keys that cannot produce deterministic lexicographic ordering. */
|
|
1222
|
+
function assertFiniteRankKey(key, label, expectedLength) {
|
|
1223
|
+
if (!Array.isArray(key) || key.length === 0) throw new Error(`${label} must return a non-empty array`);
|
|
1224
|
+
if (expectedLength !== void 0 && key.length !== expectedLength) throw new Error(`${label} returned ${key.length} elements; expected ${expectedLength}`);
|
|
1225
|
+
for (let index = 0; index < key.length; index++) if (!Number.isFinite(key[index])) throw new Error(`${label}[${index}] must be finite`);
|
|
1226
|
+
}
|
|
1227
|
+
/** Compare fixed-length lexicographic rank keys where each element is higher-is-better.
|
|
1228
|
+
* Returns a positive number when `a` ranks above `b`, negative when below, and
|
|
1229
|
+
* zero when equal. */
|
|
1230
|
+
function compareRankKeys(a, b) {
|
|
1231
|
+
assertFiniteRankKey(a, "rank key a");
|
|
1232
|
+
assertFiniteRankKey(b, "rank key b", a.length);
|
|
1233
|
+
for (let i = 0; i < a.length; i++) {
|
|
1234
|
+
const av = a[i];
|
|
1235
|
+
const bv = b[i];
|
|
1236
|
+
if (av !== bv) return av - bv;
|
|
1237
|
+
}
|
|
1238
|
+
return 0;
|
|
1239
|
+
}
|
|
1240
|
+
/** Per-candidate evidence a reflective/patch proposer grounds its next proposal
|
|
1241
|
+
* on: mean score per judge dimension + per-scenario composite. */
|
|
1242
|
+
function campaignBreakdown(campaign) {
|
|
1243
|
+
const dimSums = {};
|
|
1244
|
+
const dimCounts = {};
|
|
1245
|
+
const byScenario = /* @__PURE__ */ new Map();
|
|
1246
|
+
const notesByScenario = /* @__PURE__ */ new Map();
|
|
1247
|
+
const emittedByScenario = /* @__PURE__ */ new Map();
|
|
1248
|
+
for (const cell of campaign.cells) {
|
|
1249
|
+
const quality = projectCampaignCellQuality(cell);
|
|
1250
|
+
if (quality.score === void 0) continue;
|
|
1251
|
+
const judgeScores = Object.values(quality.successfulJudgeScores);
|
|
1252
|
+
const cellComposite = quality.score;
|
|
1253
|
+
const arr = byScenario.get(cell.scenarioId) ?? [];
|
|
1254
|
+
arr.push(cellComposite);
|
|
1255
|
+
byScenario.set(cell.scenarioId, arr);
|
|
1256
|
+
if (typeof cell.artifact === "string" && cell.artifact.trim().length > 0) {
|
|
1257
|
+
const prev = emittedByScenario.get(cell.scenarioId);
|
|
1258
|
+
if (!prev || cellComposite < prev.composite) emittedByScenario.set(cell.scenarioId, {
|
|
1259
|
+
composite: cellComposite,
|
|
1260
|
+
text: cell.artifact.slice(0, EMITTED_EVIDENCE_MAX_CHARS)
|
|
1261
|
+
});
|
|
1262
|
+
}
|
|
1263
|
+
for (const s of judgeScores) if (s.notes?.trim()) {
|
|
1264
|
+
const set = notesByScenario.get(cell.scenarioId) ?? /* @__PURE__ */ new Set();
|
|
1265
|
+
set.add(s.notes.trim());
|
|
1266
|
+
notesByScenario.set(cell.scenarioId, set);
|
|
1267
|
+
}
|
|
1268
|
+
for (const score of judgeScores) for (const [key, value] of Object.entries(score.dimensions)) {
|
|
1269
|
+
if (!Number.isFinite(value)) continue;
|
|
1270
|
+
dimSums[key] = (dimSums[key] ?? 0) + value;
|
|
1271
|
+
dimCounts[key] = (dimCounts[key] ?? 0) + 1;
|
|
1272
|
+
}
|
|
1273
|
+
}
|
|
1274
|
+
const dimensions = {};
|
|
1275
|
+
for (const key of Object.keys(dimSums)) {
|
|
1276
|
+
const count = dimCounts[key] ?? 0;
|
|
1277
|
+
dimensions[key] = count > 0 ? (dimSums[key] ?? 0) / count : 0;
|
|
1278
|
+
}
|
|
1279
|
+
return {
|
|
1280
|
+
dimensions,
|
|
1281
|
+
scenarios: [...byScenario.entries()].map(([scenarioId, comps]) => {
|
|
1282
|
+
const notesSet = notesByScenario.get(scenarioId);
|
|
1283
|
+
const notes = notesSet && notesSet.size > 0 ? [...notesSet].join(" | ") : void 0;
|
|
1284
|
+
const emitted = emittedByScenario.get(scenarioId)?.text;
|
|
1285
|
+
return {
|
|
1286
|
+
scenarioId,
|
|
1287
|
+
composite: comps.reduce((a, b) => a + b, 0) / comps.length,
|
|
1288
|
+
...notes ? { notes } : {},
|
|
1289
|
+
...emitted ? { emitted } : {}
|
|
1290
|
+
};
|
|
1291
|
+
})
|
|
1292
|
+
};
|
|
1293
|
+
}
|
|
1294
|
+
//#endregion
|
|
1295
|
+
//#region src/campaign/provenance.ts
|
|
1296
|
+
/**
|
|
1297
|
+
* Loop provenance — the durable, queryable record of WHAT a self-improvement
|
|
1298
|
+
* loop did and WHY, plus the OTel spans that let an OTLP collector pivot from
|
|
1299
|
+
* an eval-run to the underlying candidate→cell→gate→promote chain.
|
|
1300
|
+
*
|
|
1301
|
+
* Two artifacts, one source of truth:
|
|
1302
|
+
*
|
|
1303
|
+
* 1. `LoopProvenanceRecord` — a structured JSON record capturing every
|
|
1304
|
+
* candidate (surfaceHash + label + rationale + structured cause), its measured composite,
|
|
1305
|
+
* the gate decision + reasons + delta, the held-out lift, the explicit
|
|
1306
|
+
* baseline→candidate diff, and BACKEND PROVENANCE (the
|
|
1307
|
+
* `assertRealBackend` verdict + worker call count + model). This is the
|
|
1308
|
+
* ingestable audit artifact: the +lift recomputes from it, the "because
|
|
1309
|
+
* Z" rationale survives in it, and a stub backend is detectable from it.
|
|
1310
|
+
*
|
|
1311
|
+
* 2. `loopProvenanceSpans()` — the same chain emitted as OTLP-ingestable
|
|
1312
|
+
* `TraceSpanEvent`s, pivoted on the substrate's standard
|
|
1313
|
+
* `tangle.runId` / `tangle.scenarioId` / `tangle.cellId` /
|
|
1314
|
+
* `tangle.generation` attributes (the same pivots `/adapters/otel`
|
|
1315
|
+
* reads). The hosted `/v1/ingest/traces` endpoint receives the FULL loop,
|
|
1316
|
+
* not just the `cost.*` spans `runCampaign` already emits per cell.
|
|
1317
|
+
*
|
|
1318
|
+
* The record is built from the loop result and its settled cost receipts — no
|
|
1319
|
+
* second usage collector can contradict what the measured cells recorded.
|
|
1320
|
+
*/
|
|
1321
|
+
/** One translation from a completed improvement loop into durable evidence. */
|
|
1322
|
+
function loopProvenanceArgsFromResult(input) {
|
|
1323
|
+
const { result } = input;
|
|
1324
|
+
return {
|
|
1325
|
+
runId: input.runId,
|
|
1326
|
+
runDir: input.runDir,
|
|
1327
|
+
timestamp: input.timestamp,
|
|
1328
|
+
baselineSurface: input.baselineSurface,
|
|
1329
|
+
winnerSurface: result.winnerSurface,
|
|
1330
|
+
...result.winnerLabel ? { winnerLabel: result.winnerLabel } : {},
|
|
1331
|
+
...result.winnerRationale ? { winnerRationale: result.winnerRationale } : {},
|
|
1332
|
+
baselineSearchCampaign: result.baselineCampaign,
|
|
1333
|
+
generations: result.generations.map(({ record, surfaces }) => ({
|
|
1334
|
+
generationIndex: record.generationIndex,
|
|
1335
|
+
candidates: record.candidates,
|
|
1336
|
+
promoted: record.promoted,
|
|
1337
|
+
surfaces: surfaces.map(({ surfaceHash, surface, campaign }) => ({
|
|
1338
|
+
surfaceHash,
|
|
1339
|
+
surface,
|
|
1340
|
+
campaign
|
|
1341
|
+
}))
|
|
1342
|
+
})),
|
|
1343
|
+
gate: result.gateResult,
|
|
1344
|
+
...result.holdout === "deferred" ? { holdout: "deferred" } : {},
|
|
1345
|
+
baselineOnHoldout: result.baselineOnHoldout,
|
|
1346
|
+
winnerOnHoldout: result.winnerOnHoldout,
|
|
1347
|
+
...result.neutralizedSurface && result.neutralizedOnHoldout ? {
|
|
1348
|
+
neutralizedSurface: result.neutralizedSurface,
|
|
1349
|
+
neutralizedOnHoldout: result.neutralizedOnHoldout
|
|
1350
|
+
} : {},
|
|
1351
|
+
costReceipts: input.costReceipts,
|
|
1352
|
+
totalCostUsd: input.totalCostUsd,
|
|
1353
|
+
totalDurationMs: input.totalDurationMs
|
|
1354
|
+
};
|
|
1355
|
+
}
|
|
1356
|
+
function meanHoldoutComposite(campaign) {
|
|
1357
|
+
return campaignMeanComposite(campaign);
|
|
1358
|
+
}
|
|
1359
|
+
/** Build the durable provenance record from a completed loop result. */
|
|
1360
|
+
function buildLoopProvenanceRecord(args) {
|
|
1361
|
+
if (!args.runId.trim() || !args.runDir.trim()) throw new Error("buildLoopProvenanceRecord: runId and runDir must be non-empty");
|
|
1362
|
+
const timestampMs = Date.parse(args.timestamp);
|
|
1363
|
+
if (!Number.isFinite(timestampMs) || new Date(timestampMs).toISOString() !== args.timestamp) throw new Error("buildLoopProvenanceRecord: timestamp must be a canonical ISO instant");
|
|
1364
|
+
assertGateContributions(args.gate.contributingGates, "buildLoopProvenanceRecord");
|
|
1365
|
+
const agentReceipts = args.costReceipts.filter((receipt) => receipt.channel === "agent");
|
|
1366
|
+
const integrity = summarizeAgentReceiptIntegrity(agentReceipts);
|
|
1367
|
+
const models = [...new Set(agentReceipts.map((receipt) => receipt.model))].sort();
|
|
1368
|
+
const baselineSearchComposite = campaignMeanComposite(args.baselineSearchCampaign);
|
|
1369
|
+
if (!Number.isFinite(baselineSearchComposite)) throw new Error("buildLoopProvenanceRecord: baselineSearchComposite must be finite");
|
|
1370
|
+
const candidates = [];
|
|
1371
|
+
let incumbentSurfaceHash = surfaceHash(args.baselineSurface);
|
|
1372
|
+
let incumbentComposite = baselineSearchComposite;
|
|
1373
|
+
let previousGeneration = -1;
|
|
1374
|
+
for (const gen of args.generations) {
|
|
1375
|
+
if (!Number.isSafeInteger(gen.generationIndex) || gen.generationIndex !== previousGeneration + 1) throw new Error("buildLoopProvenanceRecord: generation indices must be contiguous integers starting at zero");
|
|
1376
|
+
previousGeneration = gen.generationIndex;
|
|
1377
|
+
if (gen.candidates.length === 0) throw new Error("buildLoopProvenanceRecord: a recorded generation must contain a candidate");
|
|
1378
|
+
if (new Set(gen.promoted).size !== gen.promoted.length || gen.promoted.length > 1) throw new Error("buildLoopProvenanceRecord: each generation may promote at most one candidate");
|
|
1379
|
+
const promotedSet = new Set(gen.promoted);
|
|
1380
|
+
const surfaceByHash = new Map(gen.surfaces.map((measured) => [measured.surfaceHash, measured]));
|
|
1381
|
+
const candidateByHash = new Map(gen.candidates.map((candidate) => [candidate.surfaceHash, candidate]));
|
|
1382
|
+
if (candidateByHash.size !== gen.candidates.length) throw new Error("buildLoopProvenanceRecord: duplicate candidate surface hash");
|
|
1383
|
+
if (surfaceByHash.size !== gen.surfaces.length) throw new Error("buildLoopProvenanceRecord: duplicate candidate surface entry");
|
|
1384
|
+
if (surfaceByHash.size !== candidateByHash.size) throw new Error("buildLoopProvenanceRecord: every measured candidate requires exactly one surface");
|
|
1385
|
+
for (const promotedHash of promotedSet) if (!candidateByHash.has(promotedHash)) throw new Error("buildLoopProvenanceRecord: promoted hash has no measured candidate");
|
|
1386
|
+
for (const c of gen.candidates) {
|
|
1387
|
+
validateCandidateMeasurement(c, incumbentSurfaceHash, incumbentComposite, promotedSet.has(c.surfaceHash));
|
|
1388
|
+
const measured = surfaceByHash.get(c.surfaceHash);
|
|
1389
|
+
if (measured === void 0) throw new Error("buildLoopProvenanceRecord: measured candidate is missing its surface");
|
|
1390
|
+
const { surface, campaign } = measured;
|
|
1391
|
+
if (!surfaceHashMatches(surface, c.surfaceHash)) throw new Error("buildLoopProvenanceRecord: candidate surface hash does not match its surface bytes");
|
|
1392
|
+
if (campaign.splitDigest !== args.baselineSearchCampaign.splitDigest) throw new Error("buildLoopProvenanceRecord: candidate campaign does not match the search split");
|
|
1393
|
+
const entry = {
|
|
1394
|
+
generation: gen.generationIndex,
|
|
1395
|
+
surfaceHash: c.surfaceHash,
|
|
1396
|
+
contentHash: surfaceContentHash(surface),
|
|
1397
|
+
campaignDigest: campaignMeasurementDigest(campaign),
|
|
1398
|
+
parentSurfaceHash: c.parentSurfaceHash,
|
|
1399
|
+
parentComposite: c.parentComposite,
|
|
1400
|
+
eligibleForPromotion: c.eligibleForPromotion,
|
|
1401
|
+
coverage: {
|
|
1402
|
+
expectedCells: c.coverage.expectedCells,
|
|
1403
|
+
scorableCells: c.coverage.scorableCells,
|
|
1404
|
+
unscorableCells: c.coverage.unscorableCells.map((cell) => ({ ...cell }))
|
|
1405
|
+
},
|
|
1406
|
+
composite: c.composite,
|
|
1407
|
+
promoted: promotedSet.has(c.surfaceHash)
|
|
1408
|
+
};
|
|
1409
|
+
if (c.label) entry.label = c.label;
|
|
1410
|
+
if (c.rationale) entry.rationale = c.rationale;
|
|
1411
|
+
if (c.attribution) entry.attribution = c.attribution;
|
|
1412
|
+
if (c.observedDeltaFromParent !== void 0) entry.observedDeltaFromParent = c.observedDeltaFromParent;
|
|
1413
|
+
candidates.push(entry);
|
|
1414
|
+
}
|
|
1415
|
+
const promotedHash = gen.promoted[0];
|
|
1416
|
+
if (promotedHash) {
|
|
1417
|
+
const promoted = candidateByHash.get(promotedHash);
|
|
1418
|
+
incumbentSurfaceHash = promoted.surfaceHash;
|
|
1419
|
+
if (promoted.composite === null) throw new Error("buildLoopProvenanceRecord: promoted candidate is missing a composite");
|
|
1420
|
+
incumbentComposite = promoted.composite;
|
|
1421
|
+
}
|
|
1422
|
+
}
|
|
1423
|
+
if (surfaceHash(args.winnerSurface) !== incumbentSurfaceHash) throw new Error("buildLoopProvenanceRecord: winner surface does not match the final promoted incumbent");
|
|
1424
|
+
const holdoutDeferred = args.holdout === "deferred";
|
|
1425
|
+
if (args.baselineOnHoldout.splitDigest !== args.winnerOnHoldout.splitDigest) throw new Error("buildLoopProvenanceRecord: baseline and winner use different holdout splits");
|
|
1426
|
+
if (args.neutralizedSurface === void 0 !== (args.neutralizedOnHoldout === void 0)) throw new Error("buildLoopProvenanceRecord: neutralized surface and campaign must be supplied together");
|
|
1427
|
+
if (args.neutralizedOnHoldout && args.neutralizedOnHoldout.splitDigest !== args.baselineOnHoldout.splitDigest) throw new Error("buildLoopProvenanceRecord: neutralized campaign uses a different holdout split");
|
|
1428
|
+
if (holdoutDeferred && args.neutralizedOnHoldout) throw new Error("buildLoopProvenanceRecord: a deferred holdout cannot include a neutralized measurement");
|
|
1429
|
+
const holdoutMeasurement = holdoutDeferred ? { kind: "deferred" } : {
|
|
1430
|
+
kind: "measured",
|
|
1431
|
+
baseline: meanHoldoutComposite(args.baselineOnHoldout),
|
|
1432
|
+
winner: meanHoldoutComposite(args.winnerOnHoldout),
|
|
1433
|
+
...args.neutralizedOnHoldout ? { neutralized: meanHoldoutComposite(args.neutralizedOnHoldout) } : {}
|
|
1434
|
+
};
|
|
1435
|
+
const diff = surfaceContentHash(args.baselineSurface) === surfaceContentHash(args.winnerSurface) ? "" : renderSurfaceDiff(args.winnerSurface, args.baselineSurface);
|
|
1436
|
+
const recordWithoutDigest = {
|
|
1437
|
+
schema: "tangle.loop-provenance",
|
|
1438
|
+
runId: args.runId,
|
|
1439
|
+
runDir: args.runDir,
|
|
1440
|
+
timestamp: args.timestamp,
|
|
1441
|
+
baselineContentHash: surfaceContentHash(args.baselineSurface),
|
|
1442
|
+
winnerContentHash: surfaceContentHash(args.winnerSurface),
|
|
1443
|
+
diff,
|
|
1444
|
+
candidates,
|
|
1445
|
+
evidence: {
|
|
1446
|
+
search: {
|
|
1447
|
+
splitDigest: args.baselineSearchCampaign.splitDigest,
|
|
1448
|
+
baselineCampaignDigest: campaignMeasurementDigest(args.baselineSearchCampaign)
|
|
1449
|
+
},
|
|
1450
|
+
holdout: {
|
|
1451
|
+
splitDigest: args.baselineOnHoldout.splitDigest,
|
|
1452
|
+
baselineCampaignDigest: campaignMeasurementDigest(args.baselineOnHoldout),
|
|
1453
|
+
winnerCampaignDigest: campaignMeasurementDigest(args.winnerOnHoldout),
|
|
1454
|
+
...args.neutralizedSurface && args.neutralizedOnHoldout && holdoutMeasurement.kind === "measured" && holdoutMeasurement.neutralized !== void 0 ? { neutralized: {
|
|
1455
|
+
contentHash: surfaceContentHash(args.neutralizedSurface),
|
|
1456
|
+
campaignDigest: campaignMeasurementDigest(args.neutralizedOnHoldout),
|
|
1457
|
+
composite: holdoutMeasurement.neutralized,
|
|
1458
|
+
lift: holdoutMeasurement.neutralized - holdoutMeasurement.baseline
|
|
1459
|
+
} } : {}
|
|
1460
|
+
},
|
|
1461
|
+
costReceiptsDigest: canonicalDigest([...args.costReceipts].sort((left, right) => compareCodeUnits(left.callId, right.callId)))
|
|
1462
|
+
},
|
|
1463
|
+
baselineSearchComposite,
|
|
1464
|
+
gate: {
|
|
1465
|
+
decision: args.gate.decision,
|
|
1466
|
+
reasons: args.gate.reasons,
|
|
1467
|
+
...args.gate.delta === void 0 ? {} : { delta: args.gate.delta },
|
|
1468
|
+
contributingGates: args.gate.contributingGates.map((g) => ({
|
|
1469
|
+
name: g.name,
|
|
1470
|
+
status: g.status,
|
|
1471
|
+
detail: durableGateDetail(g.detail)
|
|
1472
|
+
}))
|
|
1473
|
+
},
|
|
1474
|
+
...holdoutMeasurement.kind === "deferred" ? { holdout: "deferred" } : {
|
|
1475
|
+
baselineHoldoutComposite: holdoutMeasurement.baseline,
|
|
1476
|
+
winnerHoldoutComposite: holdoutMeasurement.winner,
|
|
1477
|
+
heldOutLift: holdoutMeasurement.winner - holdoutMeasurement.baseline
|
|
1478
|
+
},
|
|
1479
|
+
backend: {
|
|
1480
|
+
verdict: integrity.verdict,
|
|
1481
|
+
workerCallCount: integrity.totalRecords,
|
|
1482
|
+
models,
|
|
1483
|
+
totalInputTokens: integrity.totalInputTokens,
|
|
1484
|
+
totalOutputTokens: integrity.totalOutputTokens,
|
|
1485
|
+
totalCostUsd: integrity.totalCostUsd
|
|
1486
|
+
},
|
|
1487
|
+
totalCostUsd: args.totalCostUsd,
|
|
1488
|
+
totalDurationMs: args.totalDurationMs
|
|
1489
|
+
};
|
|
1490
|
+
if (args.optimizationMethod) recordWithoutDigest.optimizationMethod = durableOptimizationMethod(args.optimizationMethod);
|
|
1491
|
+
if (args.winnerLabel) recordWithoutDigest.winnerLabel = args.winnerLabel;
|
|
1492
|
+
if (args.winnerRationale) recordWithoutDigest.winnerRationale = args.winnerRationale;
|
|
1493
|
+
return {
|
|
1494
|
+
...recordWithoutDigest,
|
|
1495
|
+
recordDigest: canonicalDigest(recordWithoutDigest)
|
|
1496
|
+
};
|
|
1497
|
+
}
|
|
1498
|
+
function durableOptimizationMethod(value) {
|
|
1499
|
+
if (!value || typeof value !== "object" || typeof value.name !== "string" || !value.name.trim() || value.name.trim() !== value.name) throw new Error("buildLoopProvenanceRecord: optimization method name is invalid");
|
|
1500
|
+
if (!value.cost || !Number.isFinite(value.cost.totalCostUsd) || value.cost.totalCostUsd < 0 || typeof value.cost.accountingComplete !== "boolean" || !Array.isArray(value.cost.incompleteReasons) || value.cost.incompleteReasons.some((reason) => typeof reason !== "string" || !reason.trim()) || value.cost.accountingComplete !== (value.cost.incompleteReasons.length === 0)) throw new Error("buildLoopProvenanceRecord: optimization method cost is invalid");
|
|
1501
|
+
if (value.durationMs !== void 0 && (!Number.isFinite(value.durationMs) || value.durationMs < 0)) throw new Error("buildLoopProvenanceRecord: optimization method duration is invalid");
|
|
1502
|
+
try {
|
|
1503
|
+
return JSON.parse(canonicalString(value));
|
|
1504
|
+
} catch (cause) {
|
|
1505
|
+
throw new Error("buildLoopProvenanceRecord: optimization method data must be canonical JSON", { cause });
|
|
1506
|
+
}
|
|
1507
|
+
}
|
|
1508
|
+
/** Digest the exact campaign fields that can affect a measured comparison. */
|
|
1509
|
+
function campaignMeasurementDigest(campaign) {
|
|
1510
|
+
assertCampaignSplitIdentity(campaign.scenarios, campaign.reps, campaign.splitDigest);
|
|
1511
|
+
return canonicalDigest({
|
|
1512
|
+
schema: "tangle.campaign-measurement",
|
|
1513
|
+
manifestHash: campaign.manifestHash,
|
|
1514
|
+
splitDigest: campaign.splitDigest,
|
|
1515
|
+
seed: campaign.seed,
|
|
1516
|
+
reps: campaign.reps,
|
|
1517
|
+
runDir: campaign.runDir,
|
|
1518
|
+
scenarios: campaign.scenarios,
|
|
1519
|
+
cells: [...campaign.cells].sort((left, right) => compareCodeUnits(left.cellId, right.cellId)).map((cell) => ({
|
|
1520
|
+
manifestHash: cell.manifestHash ?? null,
|
|
1521
|
+
cellId: cell.cellId,
|
|
1522
|
+
scenarioId: cell.scenarioId,
|
|
1523
|
+
rep: cell.rep,
|
|
1524
|
+
generation: cell.generation ?? null,
|
|
1525
|
+
judgeScores: cell.judgeScores,
|
|
1526
|
+
costUsd: cell.costUsd,
|
|
1527
|
+
costProvenance: cell.costProvenance,
|
|
1528
|
+
costCallIds: [...cell.costCallIds ?? []].sort(),
|
|
1529
|
+
tokenUsage: cell.tokenUsage,
|
|
1530
|
+
resolvedModels: [...cell.resolvedModels ?? []].sort(),
|
|
1531
|
+
resolvedModel: cell.resolvedModel ?? null,
|
|
1532
|
+
durationMs: cell.durationMs,
|
|
1533
|
+
seed: cell.seed,
|
|
1534
|
+
cached: cell.cached,
|
|
1535
|
+
errorStage: cell.errorStage ?? null,
|
|
1536
|
+
errorJudge: cell.errorJudge ?? null,
|
|
1537
|
+
error: cell.error ?? null
|
|
1538
|
+
}))
|
|
1539
|
+
});
|
|
1540
|
+
}
|
|
1541
|
+
/** Recompute and validate the self-addressed durable record. */
|
|
1542
|
+
function verifyLoopProvenanceRecord(record) {
|
|
1543
|
+
if (record.schema !== "tangle.loop-provenance") throw new Error("loop provenance has an unsupported schema");
|
|
1544
|
+
const { recordDigest, ...recordWithoutDigest } = record;
|
|
1545
|
+
if (recordDigest !== canonicalDigest(recordWithoutDigest)) throw new Error("loop provenance record digest does not match its contents");
|
|
1546
|
+
assertGateContributions(record.gate?.contributingGates, "loop provenance");
|
|
1547
|
+
return record;
|
|
1548
|
+
}
|
|
1549
|
+
/** SHA-256 over the RFC 8785 canonical JSON of `value`. Throws
|
|
1550
|
+
* `LedgerCanonicalizationError` for a value with no canonical form. */
|
|
1551
|
+
function canonicalDigest(value) {
|
|
1552
|
+
return hashCanonical(value);
|
|
1553
|
+
}
|
|
1554
|
+
function durableGateDetail(detail) {
|
|
1555
|
+
if (detail === void 0) return null;
|
|
1556
|
+
try {
|
|
1557
|
+
return JSON.parse(canonicalString(detail));
|
|
1558
|
+
} catch (cause) {
|
|
1559
|
+
throw new Error("buildLoopProvenanceRecord: gate detail must be canonical JSON", { cause });
|
|
1560
|
+
}
|
|
1561
|
+
}
|
|
1562
|
+
function assertGateContributions(value, source) {
|
|
1563
|
+
if (!Array.isArray(value)) throw new Error(`${source}: gate contributingGates must be an array`);
|
|
1564
|
+
const statuses = /* @__PURE__ */ new Set([
|
|
1565
|
+
"pass",
|
|
1566
|
+
"fail",
|
|
1567
|
+
"not_evaluated"
|
|
1568
|
+
]);
|
|
1569
|
+
for (const [index, contribution] of value.entries()) {
|
|
1570
|
+
if (!contribution || typeof contribution !== "object") throw new Error(`${source}: gate contribution ${index} must be an object`);
|
|
1571
|
+
const item = contribution;
|
|
1572
|
+
if (typeof item.name !== "string" || item.name.length === 0) throw new Error(`${source}: gate contribution ${index} must have a non-empty name`);
|
|
1573
|
+
if (!statuses.has(String(item.status))) throw new Error(`${source}: gate contribution '${item.name}' must have status pass, fail, or not_evaluated`);
|
|
1574
|
+
if ("passed" in item) throw new Error(`${source}: gate contribution '${item.name}' uses obsolete passed; use status instead`);
|
|
1575
|
+
}
|
|
1576
|
+
}
|
|
1577
|
+
function validateCandidateMeasurement(candidate, expectedParentHash, expectedParentComposite, promoted) {
|
|
1578
|
+
if (!candidate.parentSurfaceHash || !/^[a-f0-9]{16}$/.test(candidate.parentSurfaceHash)) throw new Error("buildLoopProvenanceRecord: parentSurfaceHash must be 16 lowercase hex characters");
|
|
1579
|
+
if (candidate.parentSurfaceHash !== expectedParentHash) throw new Error("buildLoopProvenanceRecord: candidate parent does not match the incumbent");
|
|
1580
|
+
if (candidate.parentComposite === void 0 || !Number.isFinite(candidate.parentComposite) || Math.abs(candidate.parentComposite - expectedParentComposite) > 1e-12) throw new Error("buildLoopProvenanceRecord: candidate parentComposite does not match the incumbent");
|
|
1581
|
+
if (candidate.observedDeltaFromParent !== void 0) {
|
|
1582
|
+
if (!Number.isFinite(candidate.observedDeltaFromParent)) throw new Error("buildLoopProvenanceRecord: observedDeltaFromParent must be finite");
|
|
1583
|
+
if (candidate.eligibleForPromotion !== true) throw new Error("buildLoopProvenanceRecord: observedDeltaFromParent requires a complete eligible candidate and parentSurfaceHash");
|
|
1584
|
+
}
|
|
1585
|
+
const coverage = candidate.coverage;
|
|
1586
|
+
if (!Number.isSafeInteger(coverage.expectedCells) || coverage.expectedCells <= 0 || !Number.isSafeInteger(coverage.scorableCells) || coverage.scorableCells < 0 || coverage.scorableCells > coverage.expectedCells) throw new Error("buildLoopProvenanceRecord: invalid candidate coverage denominator");
|
|
1587
|
+
const unscorableIds = /* @__PURE__ */ new Set();
|
|
1588
|
+
for (const failure of coverage.unscorableCells) {
|
|
1589
|
+
if (typeof failure.cellId !== "string" || failure.cellId.length === 0 || typeof failure.reason !== "string" || failure.reason.length === 0 || unscorableIds.has(failure.cellId)) throw new Error("buildLoopProvenanceRecord: invalid candidate coverage failures");
|
|
1590
|
+
unscorableIds.add(failure.cellId);
|
|
1591
|
+
}
|
|
1592
|
+
if (coverage.expectedCells - coverage.scorableCells !== coverage.unscorableCells.length) throw new Error("buildLoopProvenanceRecord: candidate coverage counts do not match its failures");
|
|
1593
|
+
const complete = coverage.scorableCells === coverage.expectedCells && coverage.unscorableCells.length === 0;
|
|
1594
|
+
if (candidate.eligibleForPromotion !== complete) throw new Error("buildLoopProvenanceRecord: candidate eligibility contradicts its coverage receipt");
|
|
1595
|
+
if (complete) {
|
|
1596
|
+
if (candidate.composite === null || !Number.isFinite(candidate.composite)) throw new Error("buildLoopProvenanceRecord: complete candidate composite must be finite");
|
|
1597
|
+
if (candidate.observedDeltaFromParent === void 0) throw new Error("buildLoopProvenanceRecord: complete candidate is missing observedDeltaFromParent");
|
|
1598
|
+
const recomputed = candidate.composite - candidate.parentComposite;
|
|
1599
|
+
if (Math.abs(candidate.observedDeltaFromParent - recomputed) > 1e-12) throw new Error("buildLoopProvenanceRecord: observed delta does not match measured scores");
|
|
1600
|
+
} else {
|
|
1601
|
+
if (candidate.composite !== null && !Number.isFinite(candidate.composite)) throw new Error("buildLoopProvenanceRecord: candidate composite must be finite or null");
|
|
1602
|
+
if (candidate.observedDeltaFromParent !== void 0) throw new Error("buildLoopProvenanceRecord: incomplete candidate cannot carry observed delta");
|
|
1603
|
+
}
|
|
1604
|
+
if (promoted && (!complete || (candidate.observedDeltaFromParent ?? 0) <= 0)) throw new Error("buildLoopProvenanceRecord: promoted candidate must improve the incumbent");
|
|
1605
|
+
}
|
|
1606
|
+
function hashId(parts) {
|
|
1607
|
+
return createHash("sha256").update(parts.join(":")).digest("hex");
|
|
1608
|
+
}
|
|
1609
|
+
/**
|
|
1610
|
+
* Build the loop's OTLP-ingestable spans from a provenance record. One root
|
|
1611
|
+
* span per loop (`tangle.runId`), one span per generation, one span per
|
|
1612
|
+
* candidate (carrying its surfaceHash + label), and one span for the gate
|
|
1613
|
+
* decision (carrying reasons + delta + lift). Candidate + gate spans pivot on
|
|
1614
|
+
* the same `tangle.runId` / `tangle.generation` attributes `/adapters/otel`
|
|
1615
|
+
* reads, so the hosted collector reconstructs the full tree.
|
|
1616
|
+
*
|
|
1617
|
+
* Times are synthesized monotonically off a single base so the span tree is
|
|
1618
|
+
* orderable; the substrate does not retain per-candidate wall-clock starts.
|
|
1619
|
+
*/
|
|
1620
|
+
function loopProvenanceSpans(record, opts = {}) {
|
|
1621
|
+
const traceId = hashId(["trace", record.runId]).slice(0, 32);
|
|
1622
|
+
const baseTimeMs = opts.baseTimeMs ?? (Date.parse(record.timestamp) || Date.now());
|
|
1623
|
+
const durationMs = Math.max(1, record.totalDurationMs);
|
|
1624
|
+
if (!Number.isSafeInteger(baseTimeMs) || baseTimeMs < 0) throw new RangeError("loop provenance baseTimeMs must be a non-negative safe integer");
|
|
1625
|
+
if (!Number.isSafeInteger(durationMs)) throw new RangeError("loop provenance duration must be a safe integer number of milliseconds");
|
|
1626
|
+
const baseTime = BigInt(baseTimeMs);
|
|
1627
|
+
const baseNano = (baseTime * 1000000n).toString();
|
|
1628
|
+
const endNano = ((baseTime + BigInt(durationMs)) * 1000000n).toString();
|
|
1629
|
+
const spans = [];
|
|
1630
|
+
const rootSpanId = hashId(["root", record.runId]).slice(0, 16);
|
|
1631
|
+
const rootAttributes = {
|
|
1632
|
+
"tangle.runId": record.runId,
|
|
1633
|
+
"tangle.runDir": record.runDir,
|
|
1634
|
+
"tangle.baselineContentHash": record.baselineContentHash,
|
|
1635
|
+
"tangle.winnerContentHash": record.winnerContentHash,
|
|
1636
|
+
"tangle.baselineSearchComposite": record.baselineSearchComposite,
|
|
1637
|
+
"tangle.gateDecision": record.gate.decision,
|
|
1638
|
+
"tangle.backendVerdict": record.backend.verdict,
|
|
1639
|
+
"tangle.workerCallCount": record.backend.workerCallCount,
|
|
1640
|
+
"tangle.totalCostUsd": record.totalCostUsd
|
|
1641
|
+
};
|
|
1642
|
+
if (record.heldOutLift !== void 0) rootAttributes["tangle.heldOutLift"] = record.heldOutLift;
|
|
1643
|
+
if (record.holdout) rootAttributes["tangle.holdout"] = record.holdout;
|
|
1644
|
+
spans.push({
|
|
1645
|
+
traceId,
|
|
1646
|
+
spanId: rootSpanId,
|
|
1647
|
+
name: "improvement-loop",
|
|
1648
|
+
startTimeUnixNano: baseNano,
|
|
1649
|
+
endTimeUnixNano: endNano,
|
|
1650
|
+
attributes: rootAttributes,
|
|
1651
|
+
status: { code: "OK" },
|
|
1652
|
+
"tangle.runId": record.runId
|
|
1653
|
+
});
|
|
1654
|
+
const byGen = /* @__PURE__ */ new Map();
|
|
1655
|
+
for (const c of record.candidates) {
|
|
1656
|
+
const arr = byGen.get(c.generation) ?? [];
|
|
1657
|
+
arr.push(c);
|
|
1658
|
+
byGen.set(c.generation, arr);
|
|
1659
|
+
}
|
|
1660
|
+
for (const [generation, cands] of [...byGen.entries()].sort((a, b) => a[0] - b[0])) {
|
|
1661
|
+
const genSpanId = hashId([
|
|
1662
|
+
"gen",
|
|
1663
|
+
record.runId,
|
|
1664
|
+
String(generation)
|
|
1665
|
+
]).slice(0, 16);
|
|
1666
|
+
const measuredComposites = cands.flatMap((candidate) => candidate.composite === null ? [] : [candidate.composite]);
|
|
1667
|
+
spans.push({
|
|
1668
|
+
traceId,
|
|
1669
|
+
spanId: genSpanId,
|
|
1670
|
+
parentSpanId: rootSpanId,
|
|
1671
|
+
name: `generation-${generation}`,
|
|
1672
|
+
startTimeUnixNano: baseNano,
|
|
1673
|
+
endTimeUnixNano: endNano,
|
|
1674
|
+
attributes: {
|
|
1675
|
+
"tangle.runId": record.runId,
|
|
1676
|
+
"tangle.generation": generation,
|
|
1677
|
+
"tangle.populationSize": cands.length,
|
|
1678
|
+
...measuredComposites.length > 0 ? { "tangle.bestComposite": Math.max(...measuredComposites) } : {}
|
|
1679
|
+
},
|
|
1680
|
+
"tangle.runId": record.runId,
|
|
1681
|
+
"tangle.generation": generation
|
|
1682
|
+
});
|
|
1683
|
+
for (let i = 0; i < cands.length; i++) {
|
|
1684
|
+
const c = cands[i];
|
|
1685
|
+
const candSpanId = hashId([
|
|
1686
|
+
"cand",
|
|
1687
|
+
record.runId,
|
|
1688
|
+
String(generation),
|
|
1689
|
+
c.surfaceHash
|
|
1690
|
+
]).slice(0, 16);
|
|
1691
|
+
const attributes = {
|
|
1692
|
+
"tangle.runId": record.runId,
|
|
1693
|
+
"tangle.generation": generation,
|
|
1694
|
+
"tangle.surfaceHash": c.surfaceHash,
|
|
1695
|
+
"tangle.contentHash": c.contentHash,
|
|
1696
|
+
"tangle.parentSurfaceHash": c.parentSurfaceHash,
|
|
1697
|
+
"tangle.parentComposite": c.parentComposite,
|
|
1698
|
+
"tangle.eligibleForPromotion": c.eligibleForPromotion,
|
|
1699
|
+
"tangle.expectedCells": c.coverage.expectedCells,
|
|
1700
|
+
"tangle.scorableCells": c.coverage.scorableCells,
|
|
1701
|
+
"tangle.unscorableCells": c.coverage.unscorableCells.length,
|
|
1702
|
+
"tangle.promoted": c.promoted
|
|
1703
|
+
};
|
|
1704
|
+
if (c.composite !== null) attributes["tangle.composite"] = c.composite;
|
|
1705
|
+
if (c.observedDeltaFromParent !== void 0) attributes["tangle.observedDeltaFromParent"] = c.observedDeltaFromParent;
|
|
1706
|
+
if (c.label) attributes["tangle.candidateLabel"] = c.label;
|
|
1707
|
+
if (c.rationale) attributes["tangle.candidateRationale"] = c.rationale;
|
|
1708
|
+
spans.push({
|
|
1709
|
+
traceId,
|
|
1710
|
+
spanId: candSpanId,
|
|
1711
|
+
parentSpanId: genSpanId,
|
|
1712
|
+
name: `candidate-${c.surfaceHash}`,
|
|
1713
|
+
startTimeUnixNano: baseNano,
|
|
1714
|
+
endTimeUnixNano: endNano,
|
|
1715
|
+
attributes,
|
|
1716
|
+
"tangle.runId": record.runId,
|
|
1717
|
+
"tangle.generation": generation
|
|
1718
|
+
});
|
|
1719
|
+
}
|
|
1720
|
+
}
|
|
1721
|
+
const gateSpanId = hashId(["gate", record.runId]).slice(0, 16);
|
|
1722
|
+
const gateAttributes = {
|
|
1723
|
+
"tangle.runId": record.runId,
|
|
1724
|
+
"tangle.gateDecision": record.gate.decision,
|
|
1725
|
+
"tangle.gateReasons": JSON.stringify(record.gate.reasons)
|
|
1726
|
+
};
|
|
1727
|
+
const gateDelta = record.gate.delta ?? record.heldOutLift;
|
|
1728
|
+
if (gateDelta !== void 0) gateAttributes["tangle.gateDelta"] = gateDelta;
|
|
1729
|
+
if (record.heldOutLift !== void 0) gateAttributes["tangle.heldOutLift"] = record.heldOutLift;
|
|
1730
|
+
if (record.baselineHoldoutComposite !== void 0) gateAttributes["tangle.baselineHoldoutComposite"] = record.baselineHoldoutComposite;
|
|
1731
|
+
if (record.winnerHoldoutComposite !== void 0) gateAttributes["tangle.winnerHoldoutComposite"] = record.winnerHoldoutComposite;
|
|
1732
|
+
if (record.holdout) gateAttributes["tangle.holdout"] = record.holdout;
|
|
1733
|
+
spans.push({
|
|
1734
|
+
traceId,
|
|
1735
|
+
spanId: gateSpanId,
|
|
1736
|
+
parentSpanId: rootSpanId,
|
|
1737
|
+
name: "gate-decision",
|
|
1738
|
+
startTimeUnixNano: endNano,
|
|
1739
|
+
endTimeUnixNano: endNano,
|
|
1740
|
+
attributes: gateAttributes,
|
|
1741
|
+
status: { code: "OK" },
|
|
1742
|
+
"tangle.runId": record.runId
|
|
1743
|
+
});
|
|
1744
|
+
return spans;
|
|
1745
|
+
}
|
|
1746
|
+
/** Canonical durable paths under the run dir. */
|
|
1747
|
+
function provenanceRecordPath(runDir) {
|
|
1748
|
+
return join(runDir, "loop-provenance.json");
|
|
1749
|
+
}
|
|
1750
|
+
/**
|
|
1751
|
+
* Canonical path for the durable OTLP spans JSONL file under a loop run directory.
|
|
1752
|
+
*/
|
|
1753
|
+
function provenanceSpansPath(runDir) {
|
|
1754
|
+
return join(runDir, "loop-provenance-spans.jsonl");
|
|
1755
|
+
}
|
|
1756
|
+
/** Snapshot a held-out campaign into the hosted `EvalRunGenerationSnapshot`
|
|
1757
|
+
* shape — per-cell composite + per-judge dimensions, aggregate mean, cost,
|
|
1758
|
+
* duration. The dashboard renders these as the baseline → winner comparison. */
|
|
1759
|
+
function snapshotFromHoldout(index, surfaceHash, surface, campaign) {
|
|
1760
|
+
return {
|
|
1761
|
+
index,
|
|
1762
|
+
surfaceHash,
|
|
1763
|
+
surface,
|
|
1764
|
+
cells: campaign.cells.map((cell) => {
|
|
1765
|
+
const execution = campaignCellExecutionEvidence(cell);
|
|
1766
|
+
const quality = projectCampaignCellQuality(cell);
|
|
1767
|
+
const score = {
|
|
1768
|
+
scenarioId: cell.scenarioId,
|
|
1769
|
+
rep: cell.rep,
|
|
1770
|
+
compositeMean: quality.score ?? null,
|
|
1771
|
+
dimensions: quality.judgeScores?.perJudge ?? {},
|
|
1772
|
+
terminalOutcome: execution.terminalOutcome,
|
|
1773
|
+
executionErrorCount: execution.executionErrorCount ?? null
|
|
1774
|
+
};
|
|
1775
|
+
if (cell.error) score.errorMessage = cell.error;
|
|
1776
|
+
return score;
|
|
1777
|
+
}),
|
|
1778
|
+
compositeMean: campaignMeanCompositeOrNull(campaign),
|
|
1779
|
+
costUsd: campaign.aggregates.cost.totalCostUsd,
|
|
1780
|
+
durationMs: campaign.durationMs
|
|
1781
|
+
};
|
|
1782
|
+
}
|
|
1783
|
+
/** Build the hosted `EvalRunEvent` from the loop args + record — baseline +
|
|
1784
|
+
* winner snapshots, gate decision, held-out lift, cost, duration. Shipped to
|
|
1785
|
+
* `/v1/ingest/eval-runs` so the run appears in the dashboard's run list (the
|
|
1786
|
+
* trace spans, shipped separately, back the per-candidate drill-down). */
|
|
1787
|
+
function buildEvalRunEvent(args, record) {
|
|
1788
|
+
return {
|
|
1789
|
+
runId: args.runId,
|
|
1790
|
+
runDir: args.runDir,
|
|
1791
|
+
timestamp: args.timestamp,
|
|
1792
|
+
status: "finished",
|
|
1793
|
+
labels: {},
|
|
1794
|
+
baseline: snapshotFromHoldout(0, record.baselineContentHash, args.baselineSurface, args.baselineOnHoldout),
|
|
1795
|
+
generations: [snapshotFromHoldout(1, record.winnerContentHash, args.winnerSurface, args.winnerOnHoldout)],
|
|
1796
|
+
gateDecision: args.gate.decision,
|
|
1797
|
+
...record.heldOutLift !== void 0 ? { holdoutLift: record.heldOutLift } : {},
|
|
1798
|
+
totalCostUsd: args.totalCostUsd,
|
|
1799
|
+
totalDurationMs: args.totalDurationMs
|
|
1800
|
+
};
|
|
1801
|
+
}
|
|
1802
|
+
/**
|
|
1803
|
+
* Build the provenance record + OTel spans and persist them durably under the
|
|
1804
|
+
* run dir (and ship spans to a hosted collector when one is wired). Returns
|
|
1805
|
+
* both artifacts so the caller can assert on / re-derive from them.
|
|
1806
|
+
*
|
|
1807
|
+
* Fail-loud: the durable write throws on storage failure (a swallowed write is
|
|
1808
|
+
* exactly the "emitted but lost" failure this closes). The hosted span ship is
|
|
1809
|
+
* the one best-effort leg — its failure is logged, not thrown, so an offline
|
|
1810
|
+
* collector never fails the loop (the durable artifact is the source of truth).
|
|
1811
|
+
*/
|
|
1812
|
+
async function emitLoopProvenance(args) {
|
|
1813
|
+
const record = buildLoopProvenanceRecord(args);
|
|
1814
|
+
const spans = loopProvenanceSpans(record);
|
|
1815
|
+
args.storage.ensureDir(args.runDir);
|
|
1816
|
+
const recordPath = provenanceRecordPath(args.runDir);
|
|
1817
|
+
const spansPath = provenanceSpansPath(args.runDir);
|
|
1818
|
+
args.storage.write(recordPath, JSON.stringify(record, null, 2));
|
|
1819
|
+
args.storage.write(spansPath, spans.map((s) => JSON.stringify(s)).join("\n"));
|
|
1820
|
+
if (args.hostedClient) {
|
|
1821
|
+
try {
|
|
1822
|
+
await args.hostedClient.ingestEvalRun(buildEvalRunEvent(args, record));
|
|
1823
|
+
} catch (err) {
|
|
1824
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
1825
|
+
console.warn(`[agent-eval] hosted eval-run ingest failed (continuing): ${msg}`);
|
|
1826
|
+
}
|
|
1827
|
+
try {
|
|
1828
|
+
await args.hostedClient.ingestTraces(spans);
|
|
1829
|
+
} catch (err) {
|
|
1830
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
1831
|
+
console.warn(`[agent-eval] provenance span ingest failed (continuing): ${msg}`);
|
|
1832
|
+
}
|
|
1833
|
+
}
|
|
1834
|
+
return {
|
|
1835
|
+
record,
|
|
1836
|
+
spans,
|
|
1837
|
+
recordPath,
|
|
1838
|
+
spansPath
|
|
1839
|
+
};
|
|
1840
|
+
}
|
|
1841
|
+
//#endregion
|
|
1842
|
+
//#region src/attestation.ts
|
|
1843
|
+
/**
|
|
1844
|
+
* Reproducibility attestation for any serializable report object.
|
|
1845
|
+
*
|
|
1846
|
+
* `attest()` binds a report to its content address (sha-256 over canonical
|
|
1847
|
+
* JSON) AND binds that address to the provenance needed to reproduce it:
|
|
1848
|
+
* model versions, seeds, price-table hash, code SHA, inputs hash. The outer
|
|
1849
|
+
* `envelopeHash` prevents provenance from being rewritten while leaving the
|
|
1850
|
+
* report hash valid.
|
|
1851
|
+
*
|
|
1852
|
+
* Layering: content-addressing is the substrate's job; cryptographic SIGNING
|
|
1853
|
+
* (who vouches for the attestation, key management, transparency logs) is the
|
|
1854
|
+
* consumer's layer on top. An `AttestedReport` is a stable byte-identical
|
|
1855
|
+
* payload a consumer can sign — the substrate never holds keys.
|
|
1856
|
+
*
|
|
1857
|
+
* Generic by design: the report parameter is ANY value `canonicalJson`
|
|
1858
|
+
* accepts (campaign results, fuzz capsules, scorecards, cost ledgers). Do not
|
|
1859
|
+
* couple this module to a specific report schema.
|
|
1860
|
+
*/
|
|
1861
|
+
/** Hash scheme identifier carried by every attestation. A verifier rejects
|
|
1862
|
+
* unknown algorithms instead of guessing. */
|
|
1863
|
+
const ATTESTATION_ALGORITHM = "sha256/canonical-json";
|
|
1864
|
+
function envelopeMaterial(reportHash, provenance, algorithm) {
|
|
1865
|
+
return {
|
|
1866
|
+
reportHash,
|
|
1867
|
+
provenance,
|
|
1868
|
+
algorithm
|
|
1869
|
+
};
|
|
1870
|
+
}
|
|
1871
|
+
/**
|
|
1872
|
+
* Content-address a report and bind it to its provenance. Throws (via
|
|
1873
|
+
* `canonicalJson`) if the report or provenance contains undefined / function /
|
|
1874
|
+
* symbol / non-finite numbers — an attestation that cannot be unambiguously
|
|
1875
|
+
* serialized cannot be trusted.
|
|
1876
|
+
*/
|
|
1877
|
+
function attest(report, provenance) {
|
|
1878
|
+
const reportHash = contentHash(report);
|
|
1879
|
+
const algorithm = ATTESTATION_ALGORITHM;
|
|
1880
|
+
return {
|
|
1881
|
+
reportHash,
|
|
1882
|
+
provenance,
|
|
1883
|
+
algorithm,
|
|
1884
|
+
envelopeHash: contentHash(envelopeMaterial(reportHash, provenance, algorithm))
|
|
1885
|
+
};
|
|
1886
|
+
}
|
|
1887
|
+
/**
|
|
1888
|
+
* Verify a report against its attestation. Returns a typed outcome rather
|
|
1889
|
+
* than throwing: an unverifiable report (e.g. one that no longer
|
|
1890
|
+
* canonicalizes) is a verification failure with the cause in `reason`, not a
|
|
1891
|
+
* crash — verifiers run in pipelines that must record WHY, not die.
|
|
1892
|
+
*
|
|
1893
|
+
* Legacy attestations without `envelopeHash` remain readable, but verification
|
|
1894
|
+
* explicitly marks their provenance as unbound so a promotion path can refuse
|
|
1895
|
+
* them instead of accidentally treating old metadata as cryptographic proof.
|
|
1896
|
+
*/
|
|
1897
|
+
function verifyAttestation(report, attested) {
|
|
1898
|
+
if (attested.algorithm !== "sha256/canonical-json") return {
|
|
1899
|
+
valid: false,
|
|
1900
|
+
reason: `unknown algorithm '${attested.algorithm}' — this verifier only checks '${ATTESTATION_ALGORITHM}'`
|
|
1901
|
+
};
|
|
1902
|
+
let recomputed;
|
|
1903
|
+
try {
|
|
1904
|
+
recomputed = contentHash(report);
|
|
1905
|
+
} catch (err) {
|
|
1906
|
+
return {
|
|
1907
|
+
valid: false,
|
|
1908
|
+
reason: `report is not canonicalizable: ${err instanceof Error ? err.message : String(err)}`
|
|
1909
|
+
};
|
|
1910
|
+
}
|
|
1911
|
+
if (recomputed !== attested.reportHash) return {
|
|
1912
|
+
valid: false,
|
|
1913
|
+
reason: `report hash mismatch: attested ${attested.reportHash}, recomputed ${recomputed}`
|
|
1914
|
+
};
|
|
1915
|
+
if (attested.envelopeHash === void 0) return {
|
|
1916
|
+
valid: true,
|
|
1917
|
+
legacyUnboundProvenance: true
|
|
1918
|
+
};
|
|
1919
|
+
let envelopeHash;
|
|
1920
|
+
try {
|
|
1921
|
+
envelopeHash = contentHash(envelopeMaterial(attested.reportHash, attested.provenance, attested.algorithm));
|
|
1922
|
+
} catch (err) {
|
|
1923
|
+
return {
|
|
1924
|
+
valid: false,
|
|
1925
|
+
reason: `attestation provenance is not canonicalizable: ${err instanceof Error ? err.message : String(err)}`
|
|
1926
|
+
};
|
|
1927
|
+
}
|
|
1928
|
+
if (envelopeHash !== attested.envelopeHash) return {
|
|
1929
|
+
valid: false,
|
|
1930
|
+
reason: `attestation envelope hash mismatch: attested ${attested.envelopeHash}, recomputed ${envelopeHash}`
|
|
1931
|
+
};
|
|
1932
|
+
return { valid: true };
|
|
1933
|
+
}
|
|
1934
|
+
//#endregion
|
|
1935
|
+
//#region src/experiment/evidence-receipt.ts
|
|
1936
|
+
/**
|
|
1937
|
+
* Evidence receipts are the join between Runtime execution and Eval promotion.
|
|
1938
|
+
*
|
|
1939
|
+
* Runtime says what actually ran. Eval says what independently measured it.
|
|
1940
|
+
* This receipt binds those worlds without making either package import the other:
|
|
1941
|
+
* stable pursuit/run identity, exact candidate/evaluator/environment/input/output
|
|
1942
|
+
* content identities, and the authority class that produced the observation.
|
|
1943
|
+
*
|
|
1944
|
+
* The payload is attested with agent-eval's existing canonical report attestation.
|
|
1945
|
+
* Signing/key management deliberately remains outside this substrate; consumers may
|
|
1946
|
+
* sign the byte-stable receipt or anchor its attestation in a transparency log.
|
|
1947
|
+
*/
|
|
1948
|
+
const EVIDENCE_RECEIPT_VERSION = "1.0.0";
|
|
1949
|
+
/** Closed promotion vocabulary. A typo or unknown future kind is never independent by default. */
|
|
1950
|
+
const EVIDENCE_AUTHORITY_KINDS = [
|
|
1951
|
+
"candidate-self-report",
|
|
1952
|
+
"independent-evaluator",
|
|
1953
|
+
"independent-replication",
|
|
1954
|
+
"human-review",
|
|
1955
|
+
"production-canary"
|
|
1956
|
+
];
|
|
1957
|
+
const INDEPENDENT_EVIDENCE_AUTHORITY_KINDS = [
|
|
1958
|
+
"independent-evaluator",
|
|
1959
|
+
"independent-replication",
|
|
1960
|
+
"human-review",
|
|
1961
|
+
"production-canary"
|
|
1962
|
+
];
|
|
1963
|
+
/**
|
|
1964
|
+
* Mint a content-attested evidence receipt. Required identity fields are deliberately
|
|
1965
|
+
* non-optional: unknown evidence stays unknown and cannot accidentally look certified.
|
|
1966
|
+
* The attestation's input provenance must equal the receipt commitment so two competing
|
|
1967
|
+
* descriptions of the evaluated population cannot coexist inside one valid receipt.
|
|
1968
|
+
*/
|
|
1969
|
+
function createEvidenceReceipt(input, provenance) {
|
|
1970
|
+
const inputSetCommitment = requiredIdentity(input.inputSetCommitment, "inputSetCommitment");
|
|
1971
|
+
assertInputCommitment(provenance, inputSetCommitment);
|
|
1972
|
+
const binding = Object.freeze({
|
|
1973
|
+
schemaVersion: EVIDENCE_RECEIPT_VERSION,
|
|
1974
|
+
pursuitId: requiredIdentity(input.pursuitId, "pursuitId"),
|
|
1975
|
+
runId: requiredIdentity(input.runId, "runId"),
|
|
1976
|
+
candidateDigest: requiredIdentity(input.candidateDigest, "candidateDigest"),
|
|
1977
|
+
evaluatorDigest: requiredIdentity(input.evaluatorDigest, "evaluatorDigest"),
|
|
1978
|
+
environmentDigest: requiredIdentity(input.environmentDigest, "environmentDigest"),
|
|
1979
|
+
inputSetCommitment,
|
|
1980
|
+
outputDigest: requiredIdentity(input.outputDigest, "outputDigest"),
|
|
1981
|
+
resultDigest: requiredIdentity(input.resultDigest, "resultDigest"),
|
|
1982
|
+
authority: requiredAuthority(input.authority),
|
|
1983
|
+
...input.experimentDigest === void 0 ? {} : { experimentDigest: requiredIdentity(input.experimentDigest, "experimentDigest") },
|
|
1984
|
+
...input.observerDigest === void 0 ? {} : { observerDigest: requiredIdentity(input.observerDigest, "observerDigest") }
|
|
1985
|
+
});
|
|
1986
|
+
return Object.freeze({
|
|
1987
|
+
binding,
|
|
1988
|
+
attestation: attest(binding, provenance)
|
|
1989
|
+
});
|
|
1990
|
+
}
|
|
1991
|
+
/**
|
|
1992
|
+
* Verify promotion-grade evidence. Generic report attestation keeps a legacy read path,
|
|
1993
|
+
* but an EvidenceReceipt never accepts unbound provenance: changing the evaluator code,
|
|
1994
|
+
* model versions, input commitment provenance, or creation record must invalidate the
|
|
1995
|
+
* evidence rather than merely annotating it as legacy.
|
|
1996
|
+
*/
|
|
1997
|
+
function verifyEvidenceReceipt(receipt) {
|
|
1998
|
+
if (receipt.binding.schemaVersion !== "1.0.0") return {
|
|
1999
|
+
valid: false,
|
|
2000
|
+
reason: `unsupported evidence receipt version '${receipt.binding.schemaVersion}'`
|
|
2001
|
+
};
|
|
2002
|
+
try {
|
|
2003
|
+
for (const [field, value] of Object.entries({
|
|
2004
|
+
pursuitId: receipt.binding.pursuitId,
|
|
2005
|
+
runId: receipt.binding.runId,
|
|
2006
|
+
candidateDigest: receipt.binding.candidateDigest,
|
|
2007
|
+
evaluatorDigest: receipt.binding.evaluatorDigest,
|
|
2008
|
+
environmentDigest: receipt.binding.environmentDigest,
|
|
2009
|
+
inputSetCommitment: receipt.binding.inputSetCommitment,
|
|
2010
|
+
outputDigest: receipt.binding.outputDigest,
|
|
2011
|
+
resultDigest: receipt.binding.resultDigest
|
|
2012
|
+
})) requiredIdentity(value, field);
|
|
2013
|
+
requiredAuthority(receipt.binding.authority);
|
|
2014
|
+
assertInputCommitment(receipt.attestation.provenance, receipt.binding.inputSetCommitment);
|
|
2015
|
+
} catch (error) {
|
|
2016
|
+
return {
|
|
2017
|
+
valid: false,
|
|
2018
|
+
reason: error instanceof Error ? error.message : String(error)
|
|
2019
|
+
};
|
|
2020
|
+
}
|
|
2021
|
+
const verification = verifyAttestation(receipt.binding, receipt.attestation);
|
|
2022
|
+
if (!verification.valid) return verification;
|
|
2023
|
+
if (verification.legacyUnboundProvenance === true || receipt.attestation.envelopeHash === void 0) return {
|
|
2024
|
+
valid: false,
|
|
2025
|
+
reason: "evidence receipt provenance is not bound by an attestation envelope"
|
|
2026
|
+
};
|
|
2027
|
+
return { valid: true };
|
|
2028
|
+
}
|
|
2029
|
+
/**
|
|
2030
|
+
* Promotion may choose a stricter policy, but this primitive makes the basic separation
|
|
2031
|
+
* explicit: only a recognized independent authority is independent. Unknown/forged kinds
|
|
2032
|
+
* and candidate self-reports both return false.
|
|
2033
|
+
*/
|
|
2034
|
+
function isIndependentEvidence(receipt) {
|
|
2035
|
+
return INDEPENDENT_EVIDENCE_AUTHORITY_KINDS.includes(receipt.binding.authority?.kind);
|
|
2036
|
+
}
|
|
2037
|
+
function requiredAuthority(value) {
|
|
2038
|
+
if (typeof value !== "object" || value === null || Array.isArray(value)) throw new TypeError("evidence receipt: authority must be an object");
|
|
2039
|
+
const authority = value;
|
|
2040
|
+
if (typeof authority.kind !== "string" || !EVIDENCE_AUTHORITY_KINDS.includes(authority.kind)) throw new TypeError(`evidence receipt: unknown authority kind '${String(authority.kind)}'`);
|
|
2041
|
+
if (typeof authority.id !== "string") throw new TypeError("evidence receipt: authority.id must be a string");
|
|
2042
|
+
return Object.freeze({
|
|
2043
|
+
kind: authority.kind,
|
|
2044
|
+
id: requiredIdentity(authority.id, "authority.id")
|
|
2045
|
+
});
|
|
2046
|
+
}
|
|
2047
|
+
function assertInputCommitment(provenance, inputSetCommitment) {
|
|
2048
|
+
if (typeof provenance?.inputsHash !== "string" || provenance.inputsHash.trim().length === 0) throw new TypeError("evidence receipt: provenance.inputsHash is required");
|
|
2049
|
+
if (provenance.inputsHash.trim() !== inputSetCommitment) throw new TypeError("evidence receipt: provenance.inputsHash must equal binding.inputSetCommitment");
|
|
2050
|
+
}
|
|
2051
|
+
function requiredIdentity(value, field) {
|
|
2052
|
+
const normalized = value.trim();
|
|
2053
|
+
if (normalized.length === 0) throw new TypeError(`evidence receipt: ${field} must be non-empty`);
|
|
2054
|
+
return normalized;
|
|
2055
|
+
}
|
|
2056
|
+
//#endregion
|
|
2057
|
+
//#region src/experiment/campaign-evidence.ts
|
|
2058
|
+
/** Bind a complete measured campaign to its executed surface and actual outputs. */
|
|
2059
|
+
function createCampaignEvidenceReceipt(input) {
|
|
2060
|
+
const { campaign, surface, context } = input;
|
|
2061
|
+
assertCampaignSplitIdentity(campaign.scenarios, campaign.reps, campaign.splitDigest);
|
|
2062
|
+
const coverage = campaignCoverage(campaign.cells, campaign.scenarios, campaign.reps, true);
|
|
2063
|
+
if (campaign.scenarios.length === 0 || !coverage.complete) throw new Error("campaign evidence requires a complete, nonempty measurement");
|
|
2064
|
+
const { provenance, ...binding } = context;
|
|
2065
|
+
return createEvidenceReceipt({
|
|
2066
|
+
...binding,
|
|
2067
|
+
runId: campaign.runDir,
|
|
2068
|
+
candidateDigest: surfaceContentHash(surface),
|
|
2069
|
+
inputSetCommitment: campaign.splitDigest,
|
|
2070
|
+
outputDigest: hashCanonical([...campaign.cells].sort((a, b) => a.cellId < b.cellId ? -1 : a.cellId > b.cellId ? 1 : 0).map((cell) => ({
|
|
2071
|
+
cellId: cell.cellId,
|
|
2072
|
+
artifact: cell.artifact
|
|
2073
|
+
}))),
|
|
2074
|
+
resultDigest: campaignMeasurementDigest(campaign)
|
|
2075
|
+
}, {
|
|
2076
|
+
...provenance,
|
|
2077
|
+
inputsHash: campaign.splitDigest
|
|
2078
|
+
});
|
|
2079
|
+
}
|
|
2080
|
+
//#endregion
|
|
2081
|
+
export { campaignScenarioIdentity as $, detectScale as A, componentSurfaceIdentityMaterial as B, campaignMeanCompositeOrNull as C, recoverTruncatedJson as D, parseReflectionResponse as E, pairedDecisionShape as F, surfaceHashMatches as G, surfaceContentHash as H, minimumPairsForPairedDeltaTest as I, summarizeBackendIntegrity as J, BackendIntegrityError as K, pairedDeltaTest as L, heldoutSignificance as M, pairHoldout as N, powerPreflight as O, decidePairedPromotion as P, campaignCoverage as Q, assertCodeSurfaceIdentity as R, campaignMeanComposite as S, buildReflectionPrompt as T, surfaceDispatchRef as U, renderSurfaceDiff as V, surfaceHash as W, assertCampaignSplitIdentity as X, assertCampaignDesign as Y, assertCompleteCampaign as Z, provenanceRecordPath as _, createEvidenceReceipt as a, assertFiniteRankKey as b, ATTESTATION_ALGORITHM as c, buildLoopProvenanceRecord as d, campaignSplitDigest as et, campaignMeasurementDigest as f, loopProvenanceSpans as g, loopProvenanceArgsFromResult as h, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS as i, dimensionRegressions as j, TIE_WARN_FRACTION as k, attest as l, emitLoopProvenance as m, EVIDENCE_AUTHORITY_KINDS as n, formatCoverageFailures as nt, isIndependentEvidence as o, canonicalDigest as p, assertRealBackend as q, EVIDENCE_RECEIPT_VERSION as r, verifyEvidenceReceipt as s, createCampaignEvidenceReceipt as t, campaignSplitDigestFromIdentities as tt, verifyAttestation as u, provenanceSpansPath as v, compareRankKeys as w, campaignBreakdown as x, verifyLoopProvenanceRecord as y, codeSurfaceIdentityMaterial as z };
|
|
2082
|
+
|
|
2083
|
+
//# sourceMappingURL=campaign-evidence-D8DBLqLI.js.map
|