@tangle-network/agent-eval 0.175.0 → 0.177.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/dist/adapters/http.d.ts +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js +374 -0
- package/dist/agent-profile-cell-0gSi5ffD.js.map +1 -0
- package/dist/analyst/index.d.ts +2 -2
- package/dist/analyst/index.js +3 -3
- package/dist/{benchmark-command-D_5xG9LG.js → benchmark-command-BrsZhMSk.js} +7 -7
- package/dist/{benchmark-command-D_5xG9LG.js.map → benchmark-command-BrsZhMSk.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +3 -3
- package/dist/campaign/index.js +9 -9
- package/dist/{campaign-BzMSCejE.js → campaign-CIn-ErlJ.js} +111 -13
- package/dist/campaign-CIn-ErlJ.js.map +1 -0
- package/dist/campaign-evidence-D8DBLqLI.js +2083 -0
- package/dist/campaign-evidence-D8DBLqLI.js.map +1 -0
- package/dist/cli.js +1 -1
- package/dist/contract/index.d.ts +4 -4
- package/dist/contract/index.js +8 -8
- package/dist/{define-agent-eval-ox5McL6e.js → define-agent-eval-CJG7LD9M.js} +54 -36
- package/dist/define-agent-eval-CJG7LD9M.js.map +1 -0
- package/dist/{define-agent-eval-V1jQyCDR.d.ts → define-agent-eval-CqmlXUfQ.d.ts} +11 -4
- package/dist/define-agent-eval-CqmlXUfQ.d.ts.map +1 -0
- package/dist/{dspy-rlm-engine-Caz2pl4L.js → dspy-rlm-engine-DqjER2sV.js} +2 -2
- package/dist/{dspy-rlm-engine-Caz2pl4L.js.map → dspy-rlm-engine-DqjER2sV.js.map} +1 -1
- package/dist/{eval-campaign-BeAjdhzC.js → eval-campaign-Cs-7MiCs.js} +4 -5
- package/dist/{eval-campaign-BeAjdhzC.js.map → eval-campaign-Cs-7MiCs.js.map} +1 -1
- package/dist/experiment/index.d.ts +3 -68
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +6 -128
- package/dist/experiment/index.js.map +1 -1
- package/dist/{attestation-XSUpbc4o.js → experiment-tracker-BKEumQug.js} +2 -96
- package/dist/experiment-tracker-BKEumQug.js.map +1 -0
- package/dist/{attestation-c1QvaBdX.d.ts → experiment-tracker-CNwqCZFD.d.ts} +2 -78
- package/dist/experiment-tracker-CNwqCZFD.d.ts.map +1 -0
- package/dist/{external-optimizer-process-CxnFL1hd.js → external-optimizer-process-Dlz8YxrT.js} +3 -3
- package/dist/{external-optimizer-process-CxnFL1hd.js.map → external-optimizer-process-Dlz8YxrT.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-CQi27uEI.js → external-optimizer-subprocess-q3VzlGAO.js} +2 -2
- package/dist/{external-optimizer-subprocess-CQi27uEI.js.map → external-optimizer-subprocess-q3VzlGAO.js.map} +1 -1
- package/dist/{index-DKXuBPXf.d.ts → index-CtGf6e6X.d.ts} +50 -10
- package/dist/index-CtGf6e6X.d.ts.map +1 -0
- package/dist/{index-BTrx5s8m.d.ts → index-DuaNwvse.d.ts} +4 -4
- package/dist/{index-BTrx5s8m.d.ts.map → index-DuaNwvse.d.ts.map} +1 -1
- package/dist/{index-D-UdhAmg.d.ts → index-u0d1Jp4F.d.ts} +4 -2
- package/dist/{index-D-UdhAmg.d.ts.map → index-u0d1Jp4F.d.ts.map} +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.js +14 -15
- package/dist/index.js.map +1 -1
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-PIfjCbKn.js → ledger-core-Cs9f7385.js} +60 -47
- package/dist/{ledger-core-PIfjCbKn.js.map → ledger-core-Cs9f7385.js.map} +1 -1
- package/dist/{llm-judge-DmNaBrXB.js → llm-judge-CV80fkYA.js} +1039 -1517
- package/dist/llm-judge-CV80fkYA.js.map +1 -0
- package/dist/{mint-vWOdD8Ae.js → mint-Cc1_zwRQ.js} +2 -2
- package/dist/{mint-vWOdD8Ae.js.map → mint-Cc1_zwRQ.js.map} +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-B8mw6zj9.js → produced-state-CQFi465p.js} +3 -2
- package/dist/{produced-state-B8mw6zj9.js.map → produced-state-CQFi465p.js.map} +1 -1
- package/dist/profile-cell.js +1 -268
- package/dist/{promotion-policy-LY9mVQ7W.js → promotion-policy-DWOm70gx.js} +2 -2
- package/dist/{promotion-policy-LY9mVQ7W.js.map → promotion-policy-DWOm70gx.js.map} +1 -1
- package/dist/{release-confidence-BsGEg_xg.js → release-confidence-BcGCclTB.js} +2 -2
- package/dist/{release-confidence-BsGEg_xg.js.map → release-confidence-BcGCclTB.js.map} +1 -1
- package/dist/reporting.js +2 -2
- package/dist/{reward-hacking-CKW4teig.js → reward-hacking-D0XwhVWE.js} +2 -215
- package/dist/reward-hacking-D0XwhVWE.js.map +1 -0
- package/dist/rl.js +5 -4
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-CGlDq1GI.js → rollout-DmoJVqrF.js} +2 -2
- package/dist/{rollout-CGlDq1GI.js.map → rollout-DmoJVqrF.js.map} +1 -1
- package/dist/run-record-CR63CpHK.js +216 -0
- package/dist/run-record-CR63CpHK.js.map +1 -0
- package/dist/{run-record-ZIsR9Fif.js → run-record-DQpSf7t-.js} +2 -2
- package/dist/{run-record-ZIsR9Fif.js.map → run-record-DQpSf7t-.js.map} +1 -1
- package/dist/{semantic-concept-judge-E3s_fEjB.js → semantic-concept-judge-Dw-f7TEs.js} +3 -3
- package/dist/{semantic-concept-judge-E3s_fEjB.js.map → semantic-concept-judge-Dw-f7TEs.js.map} +1 -1
- package/dist/{sequential-B51qAYE4.js → sequential-B5gXgcyp.js} +3 -3
- package/dist/{sequential-B51qAYE4.js.map → sequential-B5gXgcyp.js.map} +1 -1
- package/dist/{skillopt-optimization-method-f7399oGb.js → skillopt-optimization-method-D0o2c2yM.js} +6 -749
- package/dist/skillopt-optimization-method-D0o2c2yM.js.map +1 -0
- package/dist/{statistical-heldout-Cqb73yE9.d.ts → statistical-heldout-Z9NROFFS.d.ts} +156 -3
- package/dist/statistical-heldout-Z9NROFFS.d.ts.map +1 -0
- package/dist/{summary-report-Bgh8CpNK.js → summary-report-B16xy9Kd.js} +2 -2
- package/dist/{summary-report-Bgh8CpNK.js.map → summary-report-B16xy9Kd.js.map} +1 -1
- package/dist/traces.js +1 -1
- package/docs/campaign-proposers.md +44 -0
- package/docs/public-api.md +62 -39
- package/docs/search-history-receipts.md +48 -1
- package/package.json +1 -1
- package/dist/attestation-XSUpbc4o.js.map +0 -1
- package/dist/attestation-c1QvaBdX.d.ts.map +0 -1
- package/dist/campaign-BzMSCejE.js.map +0 -1
- package/dist/define-agent-eval-V1jQyCDR.d.ts.map +0 -1
- package/dist/define-agent-eval-ox5McL6e.js.map +0 -1
- package/dist/index-DKXuBPXf.d.ts.map +0 -1
- package/dist/llm-judge-DmNaBrXB.js.map +0 -1
- package/dist/power-preflight-CFXm0Vjo.js +0 -502
- package/dist/power-preflight-CFXm0Vjo.js.map +0 -1
- package/dist/pre-registration-D94b7Of5.js +0 -110
- package/dist/pre-registration-D94b7Of5.js.map +0 -1
- package/dist/profile-cell.js.map +0 -1
- package/dist/reward-hacking-CKW4teig.js.map +0 -1
- package/dist/skillopt-optimization-method-f7399oGb.js.map +0 -1
- package/dist/statistical-heldout-Cqb73yE9.d.ts.map +0 -1
package/dist/profile-cell.js
CHANGED
|
@@ -1,269 +1,2 @@
|
|
|
1
|
-
import { s as
|
|
2
|
-
import { n as hashJson } from "./pre-registration-D94b7Of5.js";
|
|
3
|
-
import { createHash } from "node:crypto";
|
|
4
|
-
//#region src/agent-profile-cell.ts
|
|
5
|
-
var AgentProfileCellValidationError = class extends ValidationError {
|
|
6
|
-
path;
|
|
7
|
-
constructor(message, path = "") {
|
|
8
|
-
super(path ? `${message} (at ${path})` : message);
|
|
9
|
-
this.path = path;
|
|
10
|
-
}
|
|
11
|
-
};
|
|
12
|
-
const SHA256_HEX = /^[0-9a-f]{64}$/;
|
|
13
|
-
/**
|
|
14
|
-
* A cell id names the digest scheme that produced it. `sha256-rfc8785` is what
|
|
15
|
-
* {@link buildAgentProfileCell} mints; the bare `sha256` form is read-only,
|
|
16
|
-
* carried by cells built under an earlier release, and still verifies.
|
|
17
|
-
*/
|
|
18
|
-
const CELL_ID = /^agent-profile-cell:sha256(?:-rfc8785)?:[0-9a-f]{64}$/;
|
|
19
|
-
const CELL_ID_PREFIX = "agent-profile-cell:sha256-rfc8785:";
|
|
20
|
-
const LEGACY_CELL_ID_PREFIX = "agent-profile-cell:sha256:";
|
|
21
|
-
async function buildAgentProfileCell(input) {
|
|
22
|
-
const material = await normalizeAgentProfileCellInput(input);
|
|
23
|
-
const cellId = `${CELL_ID_PREFIX}${await hashJson(material)}`;
|
|
24
|
-
return {
|
|
25
|
-
...material,
|
|
26
|
-
cellId
|
|
27
|
-
};
|
|
28
|
-
}
|
|
29
|
-
function agentProfileCellHashMaterial(cell) {
|
|
30
|
-
const { cellId: _cellId, ...material } = cell;
|
|
31
|
-
return normalizeAgentProfileCell(material);
|
|
32
|
-
}
|
|
33
|
-
/**
|
|
34
|
-
* Verify an `AgentProfileCell`'s `cellId` matches the sha256 of its hash-material
|
|
35
|
-
* fields, confirming the record has not been tampered with. The id names its own
|
|
36
|
-
* digest scheme, so a cell minted by an earlier release verifies under that scheme.
|
|
37
|
-
*/
|
|
38
|
-
async function verifyAgentProfileCell(cell) {
|
|
39
|
-
validateAgentProfileCell(cell);
|
|
40
|
-
const material = agentProfileCellHashMaterial(cell);
|
|
41
|
-
if (cell.cellId.startsWith(CELL_ID_PREFIX)) return cell.cellId === `${CELL_ID_PREFIX}${await hashJson(material)}`;
|
|
42
|
-
return cell.cellId === `${LEGACY_CELL_ID_PREFIX}${legacyCellDigest(material)}`;
|
|
43
|
-
}
|
|
44
|
-
/**
|
|
45
|
-
* Key-sorted `JSON.stringify` digest. Private and read-only: it verifies a cell
|
|
46
|
-
* id minted before the RFC 8785 scheme, and no path that MINTS an id calls it.
|
|
47
|
-
*/
|
|
48
|
-
function legacyCellDigest(value) {
|
|
49
|
-
return createHash("sha256").update(JSON.stringify(sortKeysDeep(value)), "utf8").digest("hex");
|
|
50
|
-
}
|
|
51
|
-
function sortKeysDeep(value) {
|
|
52
|
-
if (value === null || typeof value !== "object") return value;
|
|
53
|
-
if (Array.isArray(value)) return value.map(sortKeysDeep);
|
|
54
|
-
const out = {};
|
|
55
|
-
for (const key of Object.keys(value).sort()) out[key] = sortKeysDeep(value[key]);
|
|
56
|
-
return out;
|
|
57
|
-
}
|
|
58
|
-
function validateAgentProfileCell(input) {
|
|
59
|
-
if (input === null || typeof input !== "object") throw new AgentProfileCellValidationError("expected object");
|
|
60
|
-
const obj = input;
|
|
61
|
-
expectLiteral(obj.schemaVersion, "agent-profile-cell/v1", "schemaVersion");
|
|
62
|
-
if (typeof obj.cellId !== "string" || !CELL_ID.test(obj.cellId)) throw new AgentProfileCellValidationError("cellId must match agent-profile-cell:sha256:<64 lowercase hex chars>", "cellId");
|
|
63
|
-
expectString(obj.profileId, "profileId");
|
|
64
|
-
validateSource(obj.sourceProfile, "sourceProfile");
|
|
65
|
-
if (obj.harness !== void 0) validateHarness(obj.harness, "harness");
|
|
66
|
-
if (obj.model !== void 0) expectString(obj.model, "model");
|
|
67
|
-
if (obj.promptHash !== void 0) expectString(obj.promptHash, "promptHash");
|
|
68
|
-
if (obj.dimensions !== void 0) validateDimensions(obj.dimensions, "dimensions");
|
|
69
|
-
return input;
|
|
70
|
-
}
|
|
71
|
-
function requireAgentProfileCell(record) {
|
|
72
|
-
if (!record.agentProfile) throw new AgentProfileCellValidationError(`run "${record.runId}" is missing agentProfile; profile-cell grouping requires explicit profile identity`, "agentProfile");
|
|
73
|
-
return validateAgentProfileCell(record.agentProfile);
|
|
74
|
-
}
|
|
75
|
-
function agentProfileCellKey(record) {
|
|
76
|
-
return requireAgentProfileCell(record).cellId;
|
|
77
|
-
}
|
|
78
|
-
async function assertRunAgentProfileCell(record) {
|
|
79
|
-
const profile = requireAgentProfileCell(record);
|
|
80
|
-
if (!await verifyAgentProfileCell(profile)) throw new AgentProfileCellValidationError(`run "${record.runId}" has an agentProfile.cellId that does not match its content`, "agentProfile.cellId");
|
|
81
|
-
if (profile.model !== void 0 && profile.model !== record.model) throw new AgentProfileCellValidationError(`run "${record.runId}" agentProfile.model "${profile.model}" does not match model "${record.model}"`, "agentProfile.model");
|
|
82
|
-
if (profile.promptHash !== void 0 && profile.promptHash !== record.promptHash) throw new AgentProfileCellValidationError(`run "${record.runId}" agentProfile.promptHash "${profile.promptHash}" does not match promptHash "${record.promptHash}"`, "agentProfile.promptHash");
|
|
83
|
-
return profile;
|
|
84
|
-
}
|
|
85
|
-
function groupRunsByAgentProfileCell(records) {
|
|
86
|
-
const groups = /* @__PURE__ */ new Map();
|
|
87
|
-
for (const record of records) {
|
|
88
|
-
const key = agentProfileCellKey(record);
|
|
89
|
-
const bucket = groups.get(key);
|
|
90
|
-
if (bucket) bucket.push(record);
|
|
91
|
-
else groups.set(key, [record]);
|
|
92
|
-
}
|
|
93
|
-
return groups;
|
|
94
|
-
}
|
|
95
|
-
async function normalizeAgentProfileCellInput(input) {
|
|
96
|
-
return normalizeAgentProfileCell({
|
|
97
|
-
schemaVersion: "agent-profile-cell/v1",
|
|
98
|
-
profileId: input.profileId,
|
|
99
|
-
sourceProfile: await normalizeSourceInput(input.sourceProfile),
|
|
100
|
-
harness: input.harness,
|
|
101
|
-
model: input.model,
|
|
102
|
-
promptHash: input.promptHash,
|
|
103
|
-
dimensions: input.dimensions
|
|
104
|
-
});
|
|
105
|
-
}
|
|
106
|
-
function normalizeAgentProfileCell(input) {
|
|
107
|
-
return compactObject({
|
|
108
|
-
schemaVersion: "agent-profile-cell/v1",
|
|
109
|
-
profileId: requireNonEmpty(input.profileId, "profileId"),
|
|
110
|
-
sourceProfile: normalizeSource(input.sourceProfile),
|
|
111
|
-
harness: input.harness ? normalizeHarness(input.harness, "harness") : void 0,
|
|
112
|
-
model: optionalNonEmpty(input.model, "model"),
|
|
113
|
-
promptHash: optionalNonEmpty(input.promptHash, "promptHash"),
|
|
114
|
-
dimensions: input.dimensions ? nonEmptyRecord(normalizeDimensions(input.dimensions)) : void 0
|
|
115
|
-
});
|
|
116
|
-
}
|
|
117
|
-
async function normalizeSourceInput(input) {
|
|
118
|
-
const kind = requireNonEmpty(input.kind, "sourceProfile.kind");
|
|
119
|
-
if (input.hash !== void 0 && input.profile !== void 0) throw new AgentProfileCellValidationError("sourceProfile must provide either hash or profile, not both", "sourceProfile");
|
|
120
|
-
if (input.hash !== void 0) return {
|
|
121
|
-
kind,
|
|
122
|
-
hash: requireSha256Hex(input.hash, "sourceProfile.hash")
|
|
123
|
-
};
|
|
124
|
-
if (input.profile === void 0) throw new AgentProfileCellValidationError("sourceProfile must provide hash or profile", "sourceProfile");
|
|
125
|
-
assertJson(input.profile, "sourceProfile.profile");
|
|
126
|
-
return {
|
|
127
|
-
kind,
|
|
128
|
-
hash: await hashJson(input.profile)
|
|
129
|
-
};
|
|
130
|
-
}
|
|
131
|
-
function normalizeSource(input) {
|
|
132
|
-
return {
|
|
133
|
-
kind: requireNonEmpty(input.kind, "sourceProfile.kind"),
|
|
134
|
-
hash: requireSha256Hex(input.hash, "sourceProfile.hash")
|
|
135
|
-
};
|
|
136
|
-
}
|
|
137
|
-
function normalizeHarness(input, path) {
|
|
138
|
-
return compactObject({
|
|
139
|
-
id: requireNonEmpty(input.id, `${path}.id`),
|
|
140
|
-
version: optionalNonEmpty(input.version, `${path}.version`),
|
|
141
|
-
hash: optionalNonEmpty(input.hash, `${path}.hash`)
|
|
142
|
-
});
|
|
143
|
-
}
|
|
144
|
-
function normalizeDimensions(input) {
|
|
145
|
-
const out = {};
|
|
146
|
-
for (const key of Object.keys(input).sort()) {
|
|
147
|
-
const value = input[key];
|
|
148
|
-
requireNonEmpty(key, "dimensions.<key>");
|
|
149
|
-
if (value !== null && typeof value !== "string" && typeof value !== "number" && typeof value !== "boolean") throw new AgentProfileCellValidationError("expected primitive dimension value", `dimensions.${key}`);
|
|
150
|
-
if (typeof value === "number" && !Number.isFinite(value)) throw new AgentProfileCellValidationError("expected finite number", `dimensions.${key}`);
|
|
151
|
-
out[key] = value;
|
|
152
|
-
}
|
|
153
|
-
return out;
|
|
154
|
-
}
|
|
155
|
-
function compactObject(input) {
|
|
156
|
-
const out = {};
|
|
157
|
-
for (const [key, value] of Object.entries(input)) if (value !== void 0) out[key] = value;
|
|
158
|
-
return out;
|
|
159
|
-
}
|
|
160
|
-
function nonEmptyRecord(input) {
|
|
161
|
-
return Object.keys(input).length > 0 ? input : void 0;
|
|
162
|
-
}
|
|
163
|
-
function validateSource(value, path) {
|
|
164
|
-
if (value === null || typeof value !== "object" || Array.isArray(value)) throw new AgentProfileCellValidationError("expected object", path);
|
|
165
|
-
const rec = value;
|
|
166
|
-
expectString(rec.kind, `${path}.kind`);
|
|
167
|
-
requireSha256Hex(rec.hash, `${path}.hash`);
|
|
168
|
-
}
|
|
169
|
-
function validateHarness(value, path) {
|
|
170
|
-
if (value === null || typeof value !== "object" || Array.isArray(value)) throw new AgentProfileCellValidationError("expected object", path);
|
|
171
|
-
const rec = value;
|
|
172
|
-
expectString(rec.id, `${path}.id`);
|
|
173
|
-
if (rec.version !== void 0) expectString(rec.version, `${path}.version`);
|
|
174
|
-
if (rec.hash !== void 0) expectString(rec.hash, `${path}.hash`);
|
|
175
|
-
}
|
|
176
|
-
function validateDimensions(value, path) {
|
|
177
|
-
if (value === null || typeof value !== "object" || Array.isArray(value)) throw new AgentProfileCellValidationError("expected object", path);
|
|
178
|
-
normalizeDimensions(value);
|
|
179
|
-
}
|
|
180
|
-
function assertJson(value, path) {
|
|
181
|
-
if (value === null) return;
|
|
182
|
-
const type = typeof value;
|
|
183
|
-
if (type === "string" || type === "boolean") return;
|
|
184
|
-
if (type === "number") {
|
|
185
|
-
if (!Number.isFinite(value)) throw new AgentProfileCellValidationError("expected finite number", path);
|
|
186
|
-
return;
|
|
187
|
-
}
|
|
188
|
-
if (Array.isArray(value)) {
|
|
189
|
-
value.forEach((item, index) => {
|
|
190
|
-
assertJson(item, `${path}[${index}]`);
|
|
191
|
-
});
|
|
192
|
-
return;
|
|
193
|
-
}
|
|
194
|
-
if (type === "object") {
|
|
195
|
-
for (const [key, nested] of Object.entries(value)) {
|
|
196
|
-
requireNonEmpty(key, `${path}.<key>`);
|
|
197
|
-
assertJson(nested, `${path}.${key}`);
|
|
198
|
-
}
|
|
199
|
-
return;
|
|
200
|
-
}
|
|
201
|
-
throw new AgentProfileCellValidationError("expected JSON-compatible value", path);
|
|
202
|
-
}
|
|
203
|
-
function expectLiteral(value, expected, path) {
|
|
204
|
-
if (value !== expected) throw new AgentProfileCellValidationError(`expected ${expected}`, path);
|
|
205
|
-
}
|
|
206
|
-
function expectString(value, path) {
|
|
207
|
-
if (typeof value !== "string" || value.length === 0) throw new AgentProfileCellValidationError("expected non-empty string", path);
|
|
208
|
-
}
|
|
209
|
-
function requireNonEmpty(value, path) {
|
|
210
|
-
if (typeof value !== "string" || value.length === 0) throw new AgentProfileCellValidationError("expected non-empty string", path);
|
|
211
|
-
return value;
|
|
212
|
-
}
|
|
213
|
-
function optionalNonEmpty(value, path) {
|
|
214
|
-
if (value === void 0) return void 0;
|
|
215
|
-
return requireNonEmpty(value, path);
|
|
216
|
-
}
|
|
217
|
-
function requireSha256Hex(value, path) {
|
|
218
|
-
if (typeof value !== "string" || !SHA256_HEX.test(value)) throw new AgentProfileCellValidationError("expected 64 lowercase sha256 hex chars", path);
|
|
219
|
-
return value;
|
|
220
|
-
}
|
|
221
|
-
/** Canonical `sourceProfile.kind` values. Two products fingerprinting the
|
|
222
|
-
* same canonical profile MUST use the same kind for their cells to share
|
|
223
|
-
* `sourceProfile.hash`. Extend rather than create new strings — adding a
|
|
224
|
-
* new kind is a deliberate cross-product schema change. */
|
|
225
|
-
const AGENT_PROFILE_KINDS = {
|
|
226
|
-
/** A profile declared via `defineAgentProfile(...)` from
|
|
227
|
-
* `@tangle-network/agent-interface`. The default kind for router-backed
|
|
228
|
-
* and sandbox-backed products. */
|
|
229
|
-
AGENT_INTERFACE_PROFILE: "agent-interface-profile" };
|
|
230
|
-
/** Canonicalize an arbitrary value into `AgentProfileJson` by JSON
|
|
231
|
-
* round-trip. Throws when the value contains anything not representable
|
|
232
|
-
* as JSON (functions, BigInt, cycles) — non-portable profiles fail loud
|
|
233
|
-
* rather than silently dropping fields. */
|
|
234
|
-
function toAgentProfileJson(value) {
|
|
235
|
-
let serialized;
|
|
236
|
-
try {
|
|
237
|
-
serialized = JSON.stringify(value);
|
|
238
|
-
} catch (err) {
|
|
239
|
-
throw new AgentProfileCellValidationError(`agent profile must be JSON-serializable: ${err instanceof Error ? err.message : String(err)}`, "sourceProfile.profile");
|
|
240
|
-
}
|
|
241
|
-
if (serialized === void 0) throw new AgentProfileCellValidationError("agent profile must be JSON-serializable (got undefined after JSON.stringify)", "sourceProfile.profile");
|
|
242
|
-
return JSON.parse(serialized);
|
|
243
|
-
}
|
|
244
|
-
/** Higher-level helper that hard-codes the canonical
|
|
245
|
-
* `agent-interface-profile` kind plus the JSON canonicalization. Equivalent
|
|
246
|
-
* to calling `buildAgentProfileCell` with `profileId = \`${name}@${version}\``
|
|
247
|
-
* and `sourceProfile = { kind: AGENT_INTERFACE_PROFILE, profile: <round-tripped> }`.
|
|
248
|
-
*
|
|
249
|
-
* Use this from any product consuming an agent-interface `AgentProfile`; the
|
|
250
|
-
* manual `buildAgentProfileCell` call is reserved for advanced cases
|
|
251
|
-
* (custom kinds, pre-computed source hashes, alternate profileId
|
|
252
|
-
* conventions). */
|
|
253
|
-
async function buildAgentInterfaceProfileCell(profile, input) {
|
|
254
|
-
if (!profile || typeof profile !== "object") throw new AgentProfileCellValidationError("AgentProfile must be an object", "profile");
|
|
255
|
-
if (typeof profile.name !== "string" || profile.name.length === 0) throw new AgentProfileCellValidationError("AgentProfile must have a non-empty `name`", "profile.name");
|
|
256
|
-
if (typeof profile.version !== "string" || profile.version.length === 0) throw new AgentProfileCellValidationError("AgentProfile must have a non-empty `version`", "profile.version");
|
|
257
|
-
return buildAgentProfileCell({
|
|
258
|
-
...input,
|
|
259
|
-
profileId: `${profile.name}@${profile.version}`,
|
|
260
|
-
sourceProfile: {
|
|
261
|
-
kind: AGENT_PROFILE_KINDS.AGENT_INTERFACE_PROFILE,
|
|
262
|
-
profile: toAgentProfileJson(profile)
|
|
263
|
-
}
|
|
264
|
-
});
|
|
265
|
-
}
|
|
266
|
-
//#endregion
|
|
1
|
+
import { a as assertRunAgentProfileCell, c as groupRunsByAgentProfileCell, d as validateAgentProfileCell, f as verifyAgentProfileCell, i as agentProfileCellKey, l as requireAgentProfileCell, n as AgentProfileCellValidationError, o as buildAgentInterfaceProfileCell, r as agentProfileCellHashMaterial, s as buildAgentProfileCell, t as AGENT_PROFILE_KINDS, u as toAgentProfileJson } from "./agent-profile-cell-0gSi5ffD.js";
|
|
267
2
|
export { AGENT_PROFILE_KINDS, AgentProfileCellValidationError, agentProfileCellHashMaterial, agentProfileCellKey, assertRunAgentProfileCell, buildAgentInterfaceProfileCell, buildAgentProfileCell, groupRunsByAgentProfileCell, requireAgentProfileCell, toAgentProfileJson, validateAgentProfileCell, verifyAgentProfileCell };
|
|
268
|
-
|
|
269
|
-
//# sourceMappingURL=profile-cell.js.map
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
|
|
2
|
-
import {
|
|
2
|
+
import { A as detectScale, N as pairHoldout, P as decidePairedPromotion } from "./campaign-evidence-D8DBLqLI.js";
|
|
3
3
|
//#region src/campaign/gates/promotion-policy.ts
|
|
4
4
|
/**
|
|
5
5
|
* Promotion policy over the evidence VECTOR — the substrate's answer to "never
|
|
@@ -183,4 +183,4 @@ function paretoSignificanceGate(options) {
|
|
|
183
183
|
//#endregion
|
|
184
184
|
export { paretoPolicy as n, paretoSignificanceGate as r, buildEvidenceVector as t };
|
|
185
185
|
|
|
186
|
-
//# sourceMappingURL=promotion-policy-
|
|
186
|
+
//# sourceMappingURL=promotion-policy-DWOm70gx.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"promotion-policy-LY9mVQ7W.js","names":[],"sources":["../src/campaign/gates/promotion-policy.ts"],"sourcesContent":["/**\n * Promotion policy over the evidence VECTOR — the substrate's answer to \"never\n * collapse the multi-objective promotion decision into one scalar.\" A\n * `defaultProductionGate` is one opinionated composition; this module factors\n * the decision into two reusable pieces so MANY policies can compete over the\n * SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):\n *\n * buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus\n * PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy\n * paretoPolicy(ev) // the default strategy\n * paretoSignificanceGate(options): Gate // bus + policy as a Gate\n *\n * The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a\n * potential gain source AND a safety floor (unlike `defaultProductionGate`,\n * where only `composite` can win and `criticalDimensions` are pure floors). A\n * candidate ships iff it weakly DOMINATES the baseline at the confidence level —\n * no objective credibly worse (CI floor breach) AND at least one objective\n * credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work\n * (NOT folded into hold: \"gather more reps\" and \"reject\" are different actions).\n *\n * Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate\n * per-side cost, no per-cell observation vector to bootstrap. Treat them as hard\n * constraints (compose with a budget gate via `composeGate`), not faked CIs.\n */\n\nimport {\n decidePairedPromotion,\n type PairedDecisionMethod,\n type PairedDecisionStatistic,\n type PairedMcNemarEvidence,\n} from '../../paired-promotion-decision'\nimport type { Direction } from '../../pareto'\nimport {\n DECISION_PAIRED_DELTA_STATISTIC,\n type PairedBootstrapResult,\n pairedBootstrap,\n} from '../../statistics'\nimport type { Gate, GateContext, GateDecision, GateResult, JudgeScore, Scenario } from '../types'\nimport { detectScale, pairHoldout } from './statistical-heldout'\n\n/** Where an objective's per-cell scalar comes from. `composite` reads the\n * judge's composite; `dimension` reads a named per-dimension score. */\nexport type ObjectiveSource = { kind: 'composite' } | { kind: 'dimension'; dimension: string }\n\nexport interface PromotionObjective {\n /** Stable label used in reports + `contributingGates`. */\n name: string\n source: ObjectiveSource\n /** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients\n * the paired delta so a positive bootstrap always means \"candidate better\". */\n direction: Direction\n /** The good-direction paired-delta CI lower bound must EXCEED this to count\n * as a significant gain on this axis. Interpreted in the judge's native\n * scale. Default 0 (⇒ \"confidently better\"). */\n gainThreshold?: number\n /** A floor breach (regression) is declared when the good-direction CI lower\n * bound is below −floorTolerance, or when the exact small-sample test proves\n * a drop past it. When omitted it auto-scales off observed magnitudes\n * (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */\n floorTolerance?: number\n}\n\n/** Per-axis verdict from the good-direction paired bootstrap. */\nexport type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs'\n\nexport interface AxisEvidence {\n name: string\n source: ObjectiveSource\n direction: Direction\n /** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):\n * a positive value means the candidate is better on this axis.\n *\n * DIAGNOSTIC on a pass/fail axis: there the verdict is decided on Tango's\n * score interval instead, because a percentile bootstrap over a three-atom\n * delta lattice is not a valid interval at the nonzero margin `floorTolerance`\n * and `gainThreshold` create. `ci` carries the interval that decided. */\n bootstrap: PairedBootstrapResult\n /** Which paired statistic `bootstrap.low`/`.high` bracket. `'mean'` unless the\n * caller asked for the median — on a pass/fail axis the median and its whole\n * CI are pinned at 0 by tie domination and can see neither a gain nor a\n * regression. `bootstrap.median` still carries the median point estimate. */\n bootstrapStatistic: 'median' | 'mean'\n /** The interval the axis verdict was actually decided on, good-direction and\n * in the axis's native units. */\n ci: { low: number; high: number }\n /** Which estimator produced `ci`. */\n decisionStatistic: PairedDecisionStatistic\n /** McNemar's exact evidence on a pass/fail axis; null otherwise. */\n mcnemar: PairedMcNemarEvidence | null\n /** `ci` has zero width — no evidence in either direction, so the axis is\n * neither improved nor regressed however the point estimate sits. */\n indeterminate: boolean\n /** Paired observations contributing to this axis. */\n n: number\n minimumRequired: number\n decisionMethod: PairedDecisionMethod\n gainThreshold: number\n floorTolerance: number\n verdict: AxisVerdict\n}\n\nexport interface EvidenceVector {\n /** One entry per objective — NOTHING averaged across axes. */\n axes: AxisEvidence[]\n /** Smallest paired n across axes that produced observations — the binding\n * evidence-sufficiency constraint. 0 when no axis produced observations. */\n minN: number\n /** Aggregate per-side cost from the gate context (a constraint input, not a\n * CI axis — see the module header). */\n cost: { candidate: number; baseline: number }\n}\n\n/** A promotion strategy: a pure function from the evidence vector to a verdict.\n * Many policies can run over the same `EvidenceVector` and disagree — that's\n * the point (competing strategies, shared evidence). */\nexport type PromotionPolicy = (ev: EvidenceVector) => GateResult\n\nexport interface BuildEvidenceVectorOptions {\n /** Minimum paired observations before an axis can claim significance; below\n * it the axis is `few_runs`. The exact small-sample test may require more\n * observations at the selected confidence. */\n minProductiveRuns?: number\n /** Confidence level for every axis bootstrap. Default 0.95. */\n confidence?: number\n /** Bootstrap resamples. Default 2000. */\n resamples?: number\n /** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */\n seed?: number\n /** Paired statistic every axis CI is computed on. Default `'mean'` — see\n * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */\n statistic?: 'mean' | 'median'\n}\n\n/**\n * The Evidence Bus. For each objective, pair candidate vs baseline by full\n * cellId and bootstrap a CI on the good-direction paired delta. Reuses the\n * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so\n * a single source of truth governs pairing granularity + scale handling.\n */\nexport function buildEvidenceVector<TArtifact, TScenario extends Scenario>(\n ctx: GateContext<TArtifact, TScenario>,\n objectives: PromotionObjective[],\n opts: BuildEvidenceVectorOptions = {},\n): EvidenceVector {\n if (objectives.length === 0) {\n throw new Error('buildEvidenceVector: at least 1 objective required')\n }\n const confidence = opts.confidence ?? 0.95\n const resamples = opts.resamples ?? 2000\n const seed = opts.seed ?? 1337\n const baseline = ctx.baselineJudgeScores ?? ctx.judgeScores\n const scenarioIds = new Set(ctx.scenarios.map((s) => s.id))\n\n const axes: AxisEvidence[] = []\n for (const obj of objectives) {\n let select: (s: JudgeScore) => number | undefined\n if (obj.source.kind === 'composite') {\n select = (s) => s.composite\n } else {\n const dim = obj.source.dimension\n select = (s) => s.dimensions[dim]\n }\n const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select)\n // Orient to the good direction: maximize ⇒ bootstrap (candidate − baseline);\n // minimize ⇒ bootstrap (baseline − candidate) by swapping args, so a\n // positive bootstrap always reads as \"candidate better on this axis\".\n const before = obj.direction === 'maximize' ? paired.before : paired.after\n const after = obj.direction === 'maximize' ? paired.after : paired.before\n const n = paired.before.length\n const floorTolerance =\n obj.floorTolerance ?? 0.05 * detectScale([...paired.before, ...paired.after])\n const gainThreshold = obj.gainThreshold ?? 0\n // Axes are decided on the MEAN paired delta — which for a pass/fail axis is\n // exactly the change in success rate. The median is structurally blind on\n // the shapes eval data lands in: with most pairs tied its bootstrap CI\n // collapses to [0,0] and the axis reads 'flat', hiding real gains AND real\n // regressions — pass/fail axes on ANY encoding ({0,1} and the 0-100 one\n // `detectScale` exists for), and low-cardinality axes even below half ties.\n // Orthogonal to `pairedDeltaTest`'s own small-sample switch: that picks the\n // TEST (bootstrap CI at n ≥ 20, exact sign test below), this picks the\n // ESTIMATOR the test is applied to. Both are needed — an exact sign test on\n // a tie-pinned median is still blind.\n const bootstrapStatistic = opts.statistic ?? DECISION_PAIRED_DELTA_STATISTIC\n // Both burdens of proof route through the ONE shared rule\n // (`decidePairedPromotion`), so a pass/fail axis is judged on Tango's score\n // interval — the only paired-binary construction valid at the nonzero\n // margins `gainThreshold` / `floorTolerance` create — and a zero-width\n // interval cannot buy a verdict in either direction.\n const improvement = decidePairedPromotion(before, after, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n threshold: gainThreshold,\n minPairs: opts.minProductiveRuns,\n })\n const regression = decidePairedPromotion(after, before, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n threshold: floorTolerance,\n minPairs: opts.minProductiveRuns,\n })\n const bootstrap =\n improvement.bootstrap ??\n pairedBootstrap(before, after, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n })\n // A floor breach fires on EITHER burden of proof, because they cover\n // different failures and the floor is the anti-Goodhart guard:\n // - `improvement.low < -floorTolerance` — the CREDIBLE WORST CASE exceeds\n // the tolerance. This is the contract `AxisEvidence.floorTolerance` and\n // `paretoPolicy`'s own reason string state, and it is the conservative\n // posture a safety axis needs: block unless the data can rule the\n // breach out, rather than waiting for the breach to be proven. Read off\n // the DECIDING interval, so a pass/fail axis is not screened by a\n // bootstrap that is pinned wherever ties dominate.\n // - `regression.promote` — a PROVEN drop past the tolerance. Adds the\n // small-sample path, where the decision is an exact sign test because\n // the bootstrap interval is descriptive only.\n // The credible-worst-case arm stays on the BOOTSTRAP, deliberately. Reading\n // it off the score interval instead would change what the floor MEANS on a\n // pass/fail axis: with every pair concordant the score interval is\n // ±z²/(n+z²) — ±0.39 at n=6, ±0.16 at n=20 — so a completely unchanged\n // safety axis would breach a 0.05 floor at any realistic n, and the gate\n // would refuse everything. That the bootstrap arm is instead fail-OPEN on a\n // tied pass/fail axis is a real and separate weakness: the honest fix is a\n // minimum-power requirement on the floor, not a wider interval, because the\n // data genuinely cannot rule a 5pp drop out at n=20 and a gate that says so\n // by blocking every candidate is not usable. `regression.promote` — the\n // PROVEN-drop arm — does route through the shared rule, so a real pass/fail\n // regression is now caught on an interval valid at the nonzero tolerance.\n const floorBreached = bootstrap.low < -floorTolerance || regression.promote\n // Floor check precedes the gain check: a credible regression must never be\n // masked as \"improved\". With the defaults (gainThreshold 0, positive floor)\n // the regions are disjoint and order is moot, but a consumer who sets a\n // negative gainThreshold (\"accept small dips\") could otherwise have a real\n // floor breach classified as a gain — anti-Goodhart wins the tie.\n const verdict: AxisVerdict = !improvement.sufficient\n ? 'few_runs'\n : floorBreached\n ? 'regressed'\n : improvement.promote\n ? 'improved'\n : 'flat'\n axes.push({\n name: obj.name,\n source: obj.source,\n direction: obj.direction,\n bootstrap,\n bootstrapStatistic,\n ci: { low: improvement.low, high: improvement.high },\n decisionStatistic: improvement.statistic,\n mcnemar: improvement.mcnemar,\n indeterminate: improvement.indeterminate,\n n,\n minimumRequired: improvement.minimumPairs,\n decisionMethod: improvement.method,\n gainThreshold,\n floorTolerance,\n verdict,\n })\n }\n const ns = axes.map((a) => a.n).filter((n) => n > 0)\n const minN = ns.length > 0 ? Math.min(...ns) : 0\n return { axes, minN, cost: { candidate: ctx.cost.candidate, baseline: ctx.cost.baseline } }\n}\n\n/**\n * The default strategy: symmetric multi-objective Pareto significance. Ship iff\n * the candidate weakly dominates the baseline at the confidence level — no axis\n * credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold\n * (anti-Goodhart, dominates everything). Insufficient evidence on any axis →\n * need_more_work. Statistically equivalent → hold (never ship noise).\n */\nexport const paretoPolicy: PromotionPolicy = (ev) => {\n const contributingGates = ev.axes.map((ax) => ({\n name: `objective:${ax.name}`,\n status:\n ax.verdict === 'regressed'\n ? ('fail' as const)\n : ax.verdict === 'few_runs'\n ? ('not_evaluated' as const)\n : ('pass' as const),\n detail: {\n direction: ax.direction,\n source: ax.source,\n verdict: ax.verdict,\n n: ax.n,\n deltaMedian: ax.bootstrap.median,\n ciLow: ax.ci.low,\n ciHigh: ax.ci.high,\n decisionStatistic: ax.decisionStatistic,\n decisionMethod: ax.decisionMethod,\n mcnemar: ax.mcnemar,\n indeterminate: ax.indeterminate,\n bootstrapCiLow: ax.bootstrap.low,\n bootstrapCiHigh: ax.bootstrap.high,\n confidence: ax.bootstrap.confidence,\n gainThreshold: ax.gainThreshold,\n floorTolerance: ax.floorTolerance,\n },\n }))\n\n const regressed = ev.axes.filter((a) => a.verdict === 'regressed')\n const fewRuns = ev.axes.filter((a) => a.verdict === 'few_runs')\n const improved = ev.axes.filter((a) => a.verdict === 'improved')\n\n let decision: GateDecision\n const reasons: string[] = []\n if (regressed.length > 0) {\n // Floor breach dominates: a credible regression on ANY axis blocks ship even\n // if another axis improved. This makes the +gain/−safety false positive\n // structurally impossible whenever the safety dim is an objective.\n decision = 'hold'\n for (const a of regressed) {\n reasons.push(\n `objective '${a.name}' regressed: good-direction CI.low ${a.ci.low.toFixed(3)} < -${a.floorTolerance} (n=${a.n})`,\n )\n }\n } else if (fewRuns.length > 0) {\n // No credible regression on the scored axes, but ≥1 axis lacks the evidence\n // to claim a gain ⇒ gather more reps, do NOT reject.\n decision = 'need_more_work'\n for (const a of fewRuns) {\n reasons.push(\n `objective '${a.name}' has only n=${a.n} paired runs — insufficient evidence to claim significance`,\n )\n }\n } else if (improved.length > 0) {\n // Weakly dominates (no axis worse) AND strictly better on ≥1 axis ⇒ a Pareto\n // improvement at the confidence level.\n decision = 'ship'\n reasons.push(\n `Pareto improvement at the confidence level: ${improved\n .map(\n (a) =>\n `'${a.name}' +${a.ci.low > 0 ? a.ci.low.toFixed(3) : a.bootstrap.mean.toFixed(3)} (CI.low ${a.ci.low.toFixed(3)})`,\n )\n .join(', ')}; no objective regressed`,\n )\n } else {\n // Enough evidence, nothing credibly better or worse ⇒ statistically\n // equivalent. Do NOT ship a no-op.\n decision = 'hold'\n reasons.push(\n 'no Pareto improvement: candidate statistically equivalent to baseline on every objective',\n )\n }\n\n // `delta` surfaces the composite axis if present, else the first axis — a\n // single convenience scalar; the vector lives in `contributingGates`.\n const composite = ev.axes.find((a) => a.source.kind === 'composite') ?? ev.axes[0]\n return { decision, reasons, contributingGates, delta: composite?.bootstrap.median }\n}\n\nexport interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {\n /** The objective vector. Every axis is both a gain source and a safety floor. */\n objectives: PromotionObjective[]\n /** Strategy applied to the evidence vector. Default `paretoPolicy`. Override\n * to run a stricter/looser strategy over the SAME bus (competing policies). */\n policy?: PromotionPolicy\n /** Override the gate name in reports. */\n name?: string\n}\n\n/**\n * Wrap the bus + a policy as a `Gate`. Plugs into the existing\n * `runImprovementLoop({ gate })` slot and composes via `composeGate`; default\n * loop behavior is unchanged because consumers opt in by passing this gate.\n */\nexport function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(\n options: ParetoSignificanceGateOptions,\n): Gate<TArtifact, TScenario> {\n if (options.objectives.length === 0) {\n throw new Error('paretoSignificanceGate: at least 1 objective required')\n }\n const policy = options.policy ?? paretoPolicy\n return {\n name: options.name ?? 'paretoSignificanceGate',\n async decide(ctx: GateContext<TArtifact, TScenario>): Promise<GateResult> {\n const ev = buildEvidenceVector(ctx, options.objectives, options)\n return policy(ev)\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA2IA,SAAgB,oBACd,KACA,YACA,OAAmC,CAAC,GACpB;CAChB,IAAI,WAAW,WAAW,GACxB,MAAM,IAAI,MAAM,oDAAoD;CAEtE,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,WAAW,IAAI,uBAAuB,IAAI;CAChD,MAAM,cAAc,IAAI,IAAI,IAAI,UAAU,KAAK,MAAM,EAAE,EAAE,CAAC;CAE1D,MAAM,OAAuB,CAAC;CAC9B,KAAK,MAAM,OAAO,YAAY;EAC5B,IAAI;EACJ,IAAI,IAAI,OAAO,SAAS,aACtB,UAAU,MAAM,EAAE;OACb;GACL,MAAM,MAAM,IAAI,OAAO;GACvB,UAAU,MAAM,EAAE,WAAW;EAC/B;EACA,MAAM,SAAS,YAAY,IAAI,aAAa,UAAU,aAAa,MAAM;EAIzE,MAAM,SAAS,IAAI,cAAc,aAAa,OAAO,SAAS,OAAO;EACrE,MAAM,QAAQ,IAAI,cAAc,aAAa,OAAO,QAAQ,OAAO;EACnE,MAAM,IAAI,OAAO,OAAO;EACxB,MAAM,iBACJ,IAAI,kBAAkB,MAAO,YAAY,CAAC,GAAG,OAAO,QAAQ,GAAG,OAAO,KAAK,CAAC;EAC9E,MAAM,gBAAgB,IAAI,iBAAiB;EAW3C,MAAM,qBAAqB,KAAK,aAAA;EAMhC,MAAM,cAAc,sBAAsB,QAAQ,OAAO;GACvD;GACA;GACA,WAAW;GACX;GACA,WAAW;GACX,UAAU,KAAK;EACjB,CAAC;EACD,MAAM,aAAa,sBAAsB,OAAO,QAAQ;GACtD;GACA;GACA,WAAW;GACX;GACA,WAAW;GACX,UAAU,KAAK;EACjB,CAAC;EACD,MAAM,YACJ,YAAY,aACZ,gBAAgB,QAAQ,OAAO;GAC7B;GACA;GACA,WAAW;GACX;EACF,CAAC;EAyBH,MAAM,gBAAgB,UAAU,MAAM,CAAC,kBAAkB,WAAW;EAMpE,MAAM,UAAuB,CAAC,YAAY,aACtC,aACA,gBACE,cACA,YAAY,UACV,aACA;EACR,KAAK,KAAK;GACR,MAAM,IAAI;GACV,QAAQ,IAAI;GACZ,WAAW,IAAI;GACf;GACA;GACA,IAAI;IAAE,KAAK,YAAY;IAAK,MAAM,YAAY;GAAK;GACnD,mBAAmB,YAAY;GAC/B,SAAS,YAAY;GACrB,eAAe,YAAY;GAC3B;GACA,iBAAiB,YAAY;GAC7B,gBAAgB,YAAY;GAC5B;GACA;GACA;EACF,CAAC;CACH;CACA,MAAM,KAAK,KAAK,KAAK,MAAM,EAAE,CAAC,CAAC,CAAC,QAAQ,MAAM,IAAI,CAAC;CAEnD,OAAO;EAAE;EAAM,MADF,GAAG,SAAS,IAAI,KAAK,IAAI,GAAG,EAAE,IAAI;EAC1B,MAAM;GAAE,WAAW,IAAI,KAAK;GAAW,UAAU,IAAI,KAAK;EAAS;CAAE;AAC5F;;;;;;;;AASA,MAAa,gBAAiC,OAAO;CACnD,MAAM,oBAAoB,GAAG,KAAK,KAAK,QAAQ;EAC7C,MAAM,aAAa,GAAG;EACtB,QACE,GAAG,YAAY,cACV,SACD,GAAG,YAAY,aACZ,kBACA;EACT,QAAQ;GACN,WAAW,GAAG;GACd,QAAQ,GAAG;GACX,SAAS,GAAG;GACZ,GAAG,GAAG;GACN,aAAa,GAAG,UAAU;GAC1B,OAAO,GAAG,GAAG;GACb,QAAQ,GAAG,GAAG;GACd,mBAAmB,GAAG;GACtB,gBAAgB,GAAG;GACnB,SAAS,GAAG;GACZ,eAAe,GAAG;GAClB,gBAAgB,GAAG,UAAU;GAC7B,iBAAiB,GAAG,UAAU;GAC9B,YAAY,GAAG,UAAU;GACzB,eAAe,GAAG;GAClB,gBAAgB,GAAG;EACrB;CACF,EAAE;CAEF,MAAM,YAAY,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,WAAW;CACjE,MAAM,UAAU,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,UAAU;CAC9D,MAAM,WAAW,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,UAAU;CAE/D,IAAI;CACJ,MAAM,UAAoB,CAAC;CAC3B,IAAI,UAAU,SAAS,GAAG;EAIxB,WAAW;EACX,KAAK,MAAM,KAAK,WACd,QAAQ,KACN,cAAc,EAAE,KAAK,qCAAqC,EAAE,GAAG,IAAI,QAAQ,CAAC,EAAE,MAAM,EAAE,eAAe,MAAM,EAAE,EAAE,EACjH;CAEJ,OAAO,IAAI,QAAQ,SAAS,GAAG;EAG7B,WAAW;EACX,KAAK,MAAM,KAAK,SACd,QAAQ,KACN,cAAc,EAAE,KAAK,eAAe,EAAE,EAAE,2DAC1C;CAEJ,OAAO,IAAI,SAAS,SAAS,GAAG;EAG9B,WAAW;EACX,QAAQ,KACN,+CAA+C,SAC5C,KACE,MACC,IAAI,EAAE,KAAK,KAAK,EAAE,GAAG,MAAM,IAAI,EAAE,GAAG,IAAI,QAAQ,CAAC,IAAI,EAAE,UAAU,KAAK,QAAQ,CAAC,EAAE,WAAW,EAAE,GAAG,IAAI,QAAQ,CAAC,EAAE,EACpH,CAAC,CACA,KAAK,IAAI,EAAE,yBAChB;CACF,OAAO;EAGL,WAAW;EACX,QAAQ,KACN,0FACF;CACF;CAIA,MAAM,YAAY,GAAG,KAAK,MAAM,MAAM,EAAE,OAAO,SAAS,WAAW,KAAK,GAAG,KAAK;CAChF,OAAO;EAAE;EAAU;EAAS;EAAmB,OAAO,WAAW,UAAU;CAAO;AACpF;;;;;;AAiBA,SAAgB,uBACd,SAC4B;CAC5B,IAAI,QAAQ,WAAW,WAAW,GAChC,MAAM,IAAI,MAAM,uDAAuD;CAEzE,MAAM,SAAS,QAAQ,UAAU;CACjC,OAAO;EACL,MAAM,QAAQ,QAAQ;EACtB,MAAM,OAAO,KAA6D;GACxE,MAAM,KAAK,oBAAoB,KAAK,QAAQ,YAAY,OAAO;GAC/D,OAAO,OAAO,EAAE;EAClB;CACF;AACF"}
|
|
1
|
+
{"version":3,"file":"promotion-policy-DWOm70gx.js","names":[],"sources":["../src/campaign/gates/promotion-policy.ts"],"sourcesContent":["/**\n * Promotion policy over the evidence VECTOR — the substrate's answer to \"never\n * collapse the multi-objective promotion decision into one scalar.\" A\n * `defaultProductionGate` is one opinionated composition; this module factors\n * the decision into two reusable pieces so MANY policies can compete over the\n * SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):\n *\n * buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus\n * PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy\n * paretoPolicy(ev) // the default strategy\n * paretoSignificanceGate(options): Gate // bus + policy as a Gate\n *\n * The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a\n * potential gain source AND a safety floor (unlike `defaultProductionGate`,\n * where only `composite` can win and `criticalDimensions` are pure floors). A\n * candidate ships iff it weakly DOMINATES the baseline at the confidence level —\n * no objective credibly worse (CI floor breach) AND at least one objective\n * credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work\n * (NOT folded into hold: \"gather more reps\" and \"reject\" are different actions).\n *\n * Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate\n * per-side cost, no per-cell observation vector to bootstrap. Treat them as hard\n * constraints (compose with a budget gate via `composeGate`), not faked CIs.\n */\n\nimport {\n decidePairedPromotion,\n type PairedDecisionMethod,\n type PairedDecisionStatistic,\n type PairedMcNemarEvidence,\n} from '../../paired-promotion-decision'\nimport type { Direction } from '../../pareto'\nimport {\n DECISION_PAIRED_DELTA_STATISTIC,\n type PairedBootstrapResult,\n pairedBootstrap,\n} from '../../statistics'\nimport type { Gate, GateContext, GateDecision, GateResult, JudgeScore, Scenario } from '../types'\nimport { detectScale, pairHoldout } from './statistical-heldout'\n\n/** Where an objective's per-cell scalar comes from. `composite` reads the\n * judge's composite; `dimension` reads a named per-dimension score. */\nexport type ObjectiveSource = { kind: 'composite' } | { kind: 'dimension'; dimension: string }\n\nexport interface PromotionObjective {\n /** Stable label used in reports + `contributingGates`. */\n name: string\n source: ObjectiveSource\n /** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients\n * the paired delta so a positive bootstrap always means \"candidate better\". */\n direction: Direction\n /** The good-direction paired-delta CI lower bound must EXCEED this to count\n * as a significant gain on this axis. Interpreted in the judge's native\n * scale. Default 0 (⇒ \"confidently better\"). */\n gainThreshold?: number\n /** A floor breach (regression) is declared when the good-direction CI lower\n * bound is below −floorTolerance, or when the exact small-sample test proves\n * a drop past it. When omitted it auto-scales off observed magnitudes\n * (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */\n floorTolerance?: number\n}\n\n/** Per-axis verdict from the good-direction paired bootstrap. */\nexport type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs'\n\nexport interface AxisEvidence {\n name: string\n source: ObjectiveSource\n direction: Direction\n /** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):\n * a positive value means the candidate is better on this axis.\n *\n * DIAGNOSTIC on a pass/fail axis: there the verdict is decided on Tango's\n * score interval instead, because a percentile bootstrap over a three-atom\n * delta lattice is not a valid interval at the nonzero margin `floorTolerance`\n * and `gainThreshold` create. `ci` carries the interval that decided. */\n bootstrap: PairedBootstrapResult\n /** Which paired statistic `bootstrap.low`/`.high` bracket. `'mean'` unless the\n * caller asked for the median — on a pass/fail axis the median and its whole\n * CI are pinned at 0 by tie domination and can see neither a gain nor a\n * regression. `bootstrap.median` still carries the median point estimate. */\n bootstrapStatistic: 'median' | 'mean'\n /** The interval the axis verdict was actually decided on, good-direction and\n * in the axis's native units. */\n ci: { low: number; high: number }\n /** Which estimator produced `ci`. */\n decisionStatistic: PairedDecisionStatistic\n /** McNemar's exact evidence on a pass/fail axis; null otherwise. */\n mcnemar: PairedMcNemarEvidence | null\n /** `ci` has zero width — no evidence in either direction, so the axis is\n * neither improved nor regressed however the point estimate sits. */\n indeterminate: boolean\n /** Paired observations contributing to this axis. */\n n: number\n minimumRequired: number\n decisionMethod: PairedDecisionMethod\n gainThreshold: number\n floorTolerance: number\n verdict: AxisVerdict\n}\n\nexport interface EvidenceVector {\n /** One entry per objective — NOTHING averaged across axes. */\n axes: AxisEvidence[]\n /** Smallest paired n across axes that produced observations — the binding\n * evidence-sufficiency constraint. 0 when no axis produced observations. */\n minN: number\n /** Aggregate per-side cost from the gate context (a constraint input, not a\n * CI axis — see the module header). */\n cost: { candidate: number; baseline: number }\n}\n\n/** A promotion strategy: a pure function from the evidence vector to a verdict.\n * Many policies can run over the same `EvidenceVector` and disagree — that's\n * the point (competing strategies, shared evidence). */\nexport type PromotionPolicy = (ev: EvidenceVector) => GateResult\n\nexport interface BuildEvidenceVectorOptions {\n /** Minimum paired observations before an axis can claim significance; below\n * it the axis is `few_runs`. The exact small-sample test may require more\n * observations at the selected confidence. */\n minProductiveRuns?: number\n /** Confidence level for every axis bootstrap. Default 0.95. */\n confidence?: number\n /** Bootstrap resamples. Default 2000. */\n resamples?: number\n /** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */\n seed?: number\n /** Paired statistic every axis CI is computed on. Default `'mean'` — see\n * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */\n statistic?: 'mean' | 'median'\n}\n\n/**\n * The Evidence Bus. For each objective, pair candidate vs baseline by full\n * cellId and bootstrap a CI on the good-direction paired delta. Reuses the\n * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so\n * a single source of truth governs pairing granularity + scale handling.\n */\nexport function buildEvidenceVector<TArtifact, TScenario extends Scenario>(\n ctx: GateContext<TArtifact, TScenario>,\n objectives: PromotionObjective[],\n opts: BuildEvidenceVectorOptions = {},\n): EvidenceVector {\n if (objectives.length === 0) {\n throw new Error('buildEvidenceVector: at least 1 objective required')\n }\n const confidence = opts.confidence ?? 0.95\n const resamples = opts.resamples ?? 2000\n const seed = opts.seed ?? 1337\n const baseline = ctx.baselineJudgeScores ?? ctx.judgeScores\n const scenarioIds = new Set(ctx.scenarios.map((s) => s.id))\n\n const axes: AxisEvidence[] = []\n for (const obj of objectives) {\n let select: (s: JudgeScore) => number | undefined\n if (obj.source.kind === 'composite') {\n select = (s) => s.composite\n } else {\n const dim = obj.source.dimension\n select = (s) => s.dimensions[dim]\n }\n const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select)\n // Orient to the good direction: maximize ⇒ bootstrap (candidate − baseline);\n // minimize ⇒ bootstrap (baseline − candidate) by swapping args, so a\n // positive bootstrap always reads as \"candidate better on this axis\".\n const before = obj.direction === 'maximize' ? paired.before : paired.after\n const after = obj.direction === 'maximize' ? paired.after : paired.before\n const n = paired.before.length\n const floorTolerance =\n obj.floorTolerance ?? 0.05 * detectScale([...paired.before, ...paired.after])\n const gainThreshold = obj.gainThreshold ?? 0\n // Axes are decided on the MEAN paired delta — which for a pass/fail axis is\n // exactly the change in success rate. The median is structurally blind on\n // the shapes eval data lands in: with most pairs tied its bootstrap CI\n // collapses to [0,0] and the axis reads 'flat', hiding real gains AND real\n // regressions — pass/fail axes on ANY encoding ({0,1} and the 0-100 one\n // `detectScale` exists for), and low-cardinality axes even below half ties.\n // Orthogonal to `pairedDeltaTest`'s own small-sample switch: that picks the\n // TEST (bootstrap CI at n ≥ 20, exact sign test below), this picks the\n // ESTIMATOR the test is applied to. Both are needed — an exact sign test on\n // a tie-pinned median is still blind.\n const bootstrapStatistic = opts.statistic ?? DECISION_PAIRED_DELTA_STATISTIC\n // Both burdens of proof route through the ONE shared rule\n // (`decidePairedPromotion`), so a pass/fail axis is judged on Tango's score\n // interval — the only paired-binary construction valid at the nonzero\n // margins `gainThreshold` / `floorTolerance` create — and a zero-width\n // interval cannot buy a verdict in either direction.\n const improvement = decidePairedPromotion(before, after, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n threshold: gainThreshold,\n minPairs: opts.minProductiveRuns,\n })\n const regression = decidePairedPromotion(after, before, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n threshold: floorTolerance,\n minPairs: opts.minProductiveRuns,\n })\n const bootstrap =\n improvement.bootstrap ??\n pairedBootstrap(before, after, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n })\n // A floor breach fires on EITHER burden of proof, because they cover\n // different failures and the floor is the anti-Goodhart guard:\n // - `improvement.low < -floorTolerance` — the CREDIBLE WORST CASE exceeds\n // the tolerance. This is the contract `AxisEvidence.floorTolerance` and\n // `paretoPolicy`'s own reason string state, and it is the conservative\n // posture a safety axis needs: block unless the data can rule the\n // breach out, rather than waiting for the breach to be proven. Read off\n // the DECIDING interval, so a pass/fail axis is not screened by a\n // bootstrap that is pinned wherever ties dominate.\n // - `regression.promote` — a PROVEN drop past the tolerance. Adds the\n // small-sample path, where the decision is an exact sign test because\n // the bootstrap interval is descriptive only.\n // The credible-worst-case arm stays on the BOOTSTRAP, deliberately. Reading\n // it off the score interval instead would change what the floor MEANS on a\n // pass/fail axis: with every pair concordant the score interval is\n // ±z²/(n+z²) — ±0.39 at n=6, ±0.16 at n=20 — so a completely unchanged\n // safety axis would breach a 0.05 floor at any realistic n, and the gate\n // would refuse everything. That the bootstrap arm is instead fail-OPEN on a\n // tied pass/fail axis is a real and separate weakness: the honest fix is a\n // minimum-power requirement on the floor, not a wider interval, because the\n // data genuinely cannot rule a 5pp drop out at n=20 and a gate that says so\n // by blocking every candidate is not usable. `regression.promote` — the\n // PROVEN-drop arm — does route through the shared rule, so a real pass/fail\n // regression is now caught on an interval valid at the nonzero tolerance.\n const floorBreached = bootstrap.low < -floorTolerance || regression.promote\n // Floor check precedes the gain check: a credible regression must never be\n // masked as \"improved\". With the defaults (gainThreshold 0, positive floor)\n // the regions are disjoint and order is moot, but a consumer who sets a\n // negative gainThreshold (\"accept small dips\") could otherwise have a real\n // floor breach classified as a gain — anti-Goodhart wins the tie.\n const verdict: AxisVerdict = !improvement.sufficient\n ? 'few_runs'\n : floorBreached\n ? 'regressed'\n : improvement.promote\n ? 'improved'\n : 'flat'\n axes.push({\n name: obj.name,\n source: obj.source,\n direction: obj.direction,\n bootstrap,\n bootstrapStatistic,\n ci: { low: improvement.low, high: improvement.high },\n decisionStatistic: improvement.statistic,\n mcnemar: improvement.mcnemar,\n indeterminate: improvement.indeterminate,\n n,\n minimumRequired: improvement.minimumPairs,\n decisionMethod: improvement.method,\n gainThreshold,\n floorTolerance,\n verdict,\n })\n }\n const ns = axes.map((a) => a.n).filter((n) => n > 0)\n const minN = ns.length > 0 ? Math.min(...ns) : 0\n return { axes, minN, cost: { candidate: ctx.cost.candidate, baseline: ctx.cost.baseline } }\n}\n\n/**\n * The default strategy: symmetric multi-objective Pareto significance. Ship iff\n * the candidate weakly dominates the baseline at the confidence level — no axis\n * credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold\n * (anti-Goodhart, dominates everything). Insufficient evidence on any axis →\n * need_more_work. Statistically equivalent → hold (never ship noise).\n */\nexport const paretoPolicy: PromotionPolicy = (ev) => {\n const contributingGates = ev.axes.map((ax) => ({\n name: `objective:${ax.name}`,\n status:\n ax.verdict === 'regressed'\n ? ('fail' as const)\n : ax.verdict === 'few_runs'\n ? ('not_evaluated' as const)\n : ('pass' as const),\n detail: {\n direction: ax.direction,\n source: ax.source,\n verdict: ax.verdict,\n n: ax.n,\n deltaMedian: ax.bootstrap.median,\n ciLow: ax.ci.low,\n ciHigh: ax.ci.high,\n decisionStatistic: ax.decisionStatistic,\n decisionMethod: ax.decisionMethod,\n mcnemar: ax.mcnemar,\n indeterminate: ax.indeterminate,\n bootstrapCiLow: ax.bootstrap.low,\n bootstrapCiHigh: ax.bootstrap.high,\n confidence: ax.bootstrap.confidence,\n gainThreshold: ax.gainThreshold,\n floorTolerance: ax.floorTolerance,\n },\n }))\n\n const regressed = ev.axes.filter((a) => a.verdict === 'regressed')\n const fewRuns = ev.axes.filter((a) => a.verdict === 'few_runs')\n const improved = ev.axes.filter((a) => a.verdict === 'improved')\n\n let decision: GateDecision\n const reasons: string[] = []\n if (regressed.length > 0) {\n // Floor breach dominates: a credible regression on ANY axis blocks ship even\n // if another axis improved. This makes the +gain/−safety false positive\n // structurally impossible whenever the safety dim is an objective.\n decision = 'hold'\n for (const a of regressed) {\n reasons.push(\n `objective '${a.name}' regressed: good-direction CI.low ${a.ci.low.toFixed(3)} < -${a.floorTolerance} (n=${a.n})`,\n )\n }\n } else if (fewRuns.length > 0) {\n // No credible regression on the scored axes, but ≥1 axis lacks the evidence\n // to claim a gain ⇒ gather more reps, do NOT reject.\n decision = 'need_more_work'\n for (const a of fewRuns) {\n reasons.push(\n `objective '${a.name}' has only n=${a.n} paired runs — insufficient evidence to claim significance`,\n )\n }\n } else if (improved.length > 0) {\n // Weakly dominates (no axis worse) AND strictly better on ≥1 axis ⇒ a Pareto\n // improvement at the confidence level.\n decision = 'ship'\n reasons.push(\n `Pareto improvement at the confidence level: ${improved\n .map(\n (a) =>\n `'${a.name}' +${a.ci.low > 0 ? a.ci.low.toFixed(3) : a.bootstrap.mean.toFixed(3)} (CI.low ${a.ci.low.toFixed(3)})`,\n )\n .join(', ')}; no objective regressed`,\n )\n } else {\n // Enough evidence, nothing credibly better or worse ⇒ statistically\n // equivalent. Do NOT ship a no-op.\n decision = 'hold'\n reasons.push(\n 'no Pareto improvement: candidate statistically equivalent to baseline on every objective',\n )\n }\n\n // `delta` surfaces the composite axis if present, else the first axis — a\n // single convenience scalar; the vector lives in `contributingGates`.\n const composite = ev.axes.find((a) => a.source.kind === 'composite') ?? ev.axes[0]\n return { decision, reasons, contributingGates, delta: composite?.bootstrap.median }\n}\n\nexport interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {\n /** The objective vector. Every axis is both a gain source and a safety floor. */\n objectives: PromotionObjective[]\n /** Strategy applied to the evidence vector. Default `paretoPolicy`. Override\n * to run a stricter/looser strategy over the SAME bus (competing policies). */\n policy?: PromotionPolicy\n /** Override the gate name in reports. */\n name?: string\n}\n\n/**\n * Wrap the bus + a policy as a `Gate`. Plugs into the existing\n * `runImprovementLoop({ gate })` slot and composes via `composeGate`; default\n * loop behavior is unchanged because consumers opt in by passing this gate.\n */\nexport function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(\n options: ParetoSignificanceGateOptions,\n): Gate<TArtifact, TScenario> {\n if (options.objectives.length === 0) {\n throw new Error('paretoSignificanceGate: at least 1 objective required')\n }\n const policy = options.policy ?? paretoPolicy\n return {\n name: options.name ?? 'paretoSignificanceGate',\n async decide(ctx: GateContext<TArtifact, TScenario>): Promise<GateResult> {\n const ev = buildEvidenceVector(ctx, options.objectives, options)\n return policy(ev)\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA2IA,SAAgB,oBACd,KACA,YACA,OAAmC,CAAC,GACpB;CAChB,IAAI,WAAW,WAAW,GACxB,MAAM,IAAI,MAAM,oDAAoD;CAEtE,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,WAAW,IAAI,uBAAuB,IAAI;CAChD,MAAM,cAAc,IAAI,IAAI,IAAI,UAAU,KAAK,MAAM,EAAE,EAAE,CAAC;CAE1D,MAAM,OAAuB,CAAC;CAC9B,KAAK,MAAM,OAAO,YAAY;EAC5B,IAAI;EACJ,IAAI,IAAI,OAAO,SAAS,aACtB,UAAU,MAAM,EAAE;OACb;GACL,MAAM,MAAM,IAAI,OAAO;GACvB,UAAU,MAAM,EAAE,WAAW;EAC/B;EACA,MAAM,SAAS,YAAY,IAAI,aAAa,UAAU,aAAa,MAAM;EAIzE,MAAM,SAAS,IAAI,cAAc,aAAa,OAAO,SAAS,OAAO;EACrE,MAAM,QAAQ,IAAI,cAAc,aAAa,OAAO,QAAQ,OAAO;EACnE,MAAM,IAAI,OAAO,OAAO;EACxB,MAAM,iBACJ,IAAI,kBAAkB,MAAO,YAAY,CAAC,GAAG,OAAO,QAAQ,GAAG,OAAO,KAAK,CAAC;EAC9E,MAAM,gBAAgB,IAAI,iBAAiB;EAW3C,MAAM,qBAAqB,KAAK,aAAA;EAMhC,MAAM,cAAc,sBAAsB,QAAQ,OAAO;GACvD;GACA;GACA,WAAW;GACX;GACA,WAAW;GACX,UAAU,KAAK;EACjB,CAAC;EACD,MAAM,aAAa,sBAAsB,OAAO,QAAQ;GACtD;GACA;GACA,WAAW;GACX;GACA,WAAW;GACX,UAAU,KAAK;EACjB,CAAC;EACD,MAAM,YACJ,YAAY,aACZ,gBAAgB,QAAQ,OAAO;GAC7B;GACA;GACA,WAAW;GACX;EACF,CAAC;EAyBH,MAAM,gBAAgB,UAAU,MAAM,CAAC,kBAAkB,WAAW;EAMpE,MAAM,UAAuB,CAAC,YAAY,aACtC,aACA,gBACE,cACA,YAAY,UACV,aACA;EACR,KAAK,KAAK;GACR,MAAM,IAAI;GACV,QAAQ,IAAI;GACZ,WAAW,IAAI;GACf;GACA;GACA,IAAI;IAAE,KAAK,YAAY;IAAK,MAAM,YAAY;GAAK;GACnD,mBAAmB,YAAY;GAC/B,SAAS,YAAY;GACrB,eAAe,YAAY;GAC3B;GACA,iBAAiB,YAAY;GAC7B,gBAAgB,YAAY;GAC5B;GACA;GACA;EACF,CAAC;CACH;CACA,MAAM,KAAK,KAAK,KAAK,MAAM,EAAE,CAAC,CAAC,CAAC,QAAQ,MAAM,IAAI,CAAC;CAEnD,OAAO;EAAE;EAAM,MADF,GAAG,SAAS,IAAI,KAAK,IAAI,GAAG,EAAE,IAAI;EAC1B,MAAM;GAAE,WAAW,IAAI,KAAK;GAAW,UAAU,IAAI,KAAK;EAAS;CAAE;AAC5F;;;;;;;;AASA,MAAa,gBAAiC,OAAO;CACnD,MAAM,oBAAoB,GAAG,KAAK,KAAK,QAAQ;EAC7C,MAAM,aAAa,GAAG;EACtB,QACE,GAAG,YAAY,cACV,SACD,GAAG,YAAY,aACZ,kBACA;EACT,QAAQ;GACN,WAAW,GAAG;GACd,QAAQ,GAAG;GACX,SAAS,GAAG;GACZ,GAAG,GAAG;GACN,aAAa,GAAG,UAAU;GAC1B,OAAO,GAAG,GAAG;GACb,QAAQ,GAAG,GAAG;GACd,mBAAmB,GAAG;GACtB,gBAAgB,GAAG;GACnB,SAAS,GAAG;GACZ,eAAe,GAAG;GAClB,gBAAgB,GAAG,UAAU;GAC7B,iBAAiB,GAAG,UAAU;GAC9B,YAAY,GAAG,UAAU;GACzB,eAAe,GAAG;GAClB,gBAAgB,GAAG;EACrB;CACF,EAAE;CAEF,MAAM,YAAY,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,WAAW;CACjE,MAAM,UAAU,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,UAAU;CAC9D,MAAM,WAAW,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,UAAU;CAE/D,IAAI;CACJ,MAAM,UAAoB,CAAC;CAC3B,IAAI,UAAU,SAAS,GAAG;EAIxB,WAAW;EACX,KAAK,MAAM,KAAK,WACd,QAAQ,KACN,cAAc,EAAE,KAAK,qCAAqC,EAAE,GAAG,IAAI,QAAQ,CAAC,EAAE,MAAM,EAAE,eAAe,MAAM,EAAE,EAAE,EACjH;CAEJ,OAAO,IAAI,QAAQ,SAAS,GAAG;EAG7B,WAAW;EACX,KAAK,MAAM,KAAK,SACd,QAAQ,KACN,cAAc,EAAE,KAAK,eAAe,EAAE,EAAE,2DAC1C;CAEJ,OAAO,IAAI,SAAS,SAAS,GAAG;EAG9B,WAAW;EACX,QAAQ,KACN,+CAA+C,SAC5C,KACE,MACC,IAAI,EAAE,KAAK,KAAK,EAAE,GAAG,MAAM,IAAI,EAAE,GAAG,IAAI,QAAQ,CAAC,IAAI,EAAE,UAAU,KAAK,QAAQ,CAAC,EAAE,WAAW,EAAE,GAAG,IAAI,QAAQ,CAAC,EAAE,EACpH,CAAC,CACA,KAAK,IAAI,EAAE,yBAChB;CACF,OAAO;EAGL,WAAW;EACX,QAAQ,KACN,0FACF;CACF;CAIA,MAAM,YAAY,GAAG,KAAK,MAAM,MAAM,EAAE,OAAO,SAAS,WAAW,KAAK,GAAG,KAAK;CAChF,OAAO;EAAE;EAAU;EAAS;EAAmB,OAAO,WAAW,UAAU;CAAO;AACpF;;;;;;AAiBA,SAAgB,uBACd,SAC4B;CAC5B,IAAI,QAAQ,WAAW,WAAW,GAChC,MAAM,IAAI,MAAM,uDAAuD;CAEzE,MAAM,SAAS,QAAQ,UAAU;CACjC,OAAO;EACL,MAAM,QAAQ,QAAQ;EACtB,MAAM,OAAO,KAA6D;GACxE,MAAM,KAAK,oBAAoB,KAAK,QAAQ,YAAY,OAAO;GAC/D,OAAO,OAAO,EAAE;EAClB;CACF;AACF"}
|
|
@@ -2,7 +2,7 @@ import { c as VerificationError, s as ValidationError } from "./errors-Dngq5h35.
|
|
|
2
2
|
import { t as mulberry32 } from "./random-Dn5fPWkt.js";
|
|
3
3
|
import { t as FAILURE_CLASSES } from "./schema-CSf6qWgZ.js";
|
|
4
4
|
import { r as observedSplitScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
|
|
5
|
-
import { c as validateRunRecord } from "./run-record-
|
|
5
|
+
import { c as validateRunRecord } from "./run-record-DQpSf7t-.js";
|
|
6
6
|
//#region src/promotion-gate.ts
|
|
7
7
|
/**
|
|
8
8
|
* Bootstrap-CI promotion gate.
|
|
@@ -523,4 +523,4 @@ function fmt(x) {
|
|
|
523
523
|
//#endregion
|
|
524
524
|
export { judgeReplayGate as i, evaluateReleaseConfidence as n, bootstrapCi as r, assertReleaseConfidence as t };
|
|
525
525
|
|
|
526
|
-
//# sourceMappingURL=release-confidence-
|
|
526
|
+
//# sourceMappingURL=release-confidence-BcGCclTB.js.map
|