@tangle-network/agent-eval 0.128.2 → 0.129.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. package/CHANGELOG.md +265 -0
  2. package/README.md +18 -0
  3. package/dist/analyst/index.d.ts +107 -165
  4. package/dist/analyst/index.js +5 -9
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/belief-state/index.d.ts +2 -19
  7. package/dist/belief-state/index.js +30 -31
  8. package/dist/belief-state/index.js.map +1 -1
  9. package/dist/benchmarks/index.d.ts +5 -8
  10. package/dist/benchmarks/index.js +12 -11
  11. package/dist/builder-eval/index.js +1 -1
  12. package/dist/campaign/index.d.ts +30 -39
  13. package/dist/campaign/index.js +11 -10
  14. package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
  15. package/dist/chunk-2QU3YOPR.js.map +1 -0
  16. package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
  17. package/dist/chunk-3OCR4R5I.js.map +1 -0
  18. package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
  19. package/dist/chunk-56TAVBOK.js.map +1 -0
  20. package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
  21. package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
  22. package/dist/chunk-BSO5JDQH.js.map +1 -0
  23. package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
  24. package/dist/chunk-C6LXANRU.js.map +1 -0
  25. package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
  26. package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
  27. package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
  28. package/dist/chunk-EG66UGL4.js.map +1 -0
  29. package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
  30. package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
  31. package/dist/chunk-G7MGMCZD.js.map +1 -0
  32. package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
  33. package/dist/chunk-H23X7XKK.js.map +1 -0
  34. package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
  35. package/dist/chunk-HPWUNB47.js.map +1 -0
  36. package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
  37. package/dist/chunk-IYCLP2N2.js.map +1 -0
  38. package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
  39. package/dist/chunk-JQSF5DQT.js.map +1 -0
  40. package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
  41. package/dist/chunk-M4YBQKIJ.js.map +1 -0
  42. package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
  43. package/dist/chunk-OIUOT4QD.js +44 -0
  44. package/dist/chunk-OIUOT4QD.js.map +1 -0
  45. package/dist/chunk-OWN5NPMC.js +152 -0
  46. package/dist/chunk-OWN5NPMC.js.map +1 -0
  47. package/dist/chunk-PC5DOSM7.js +579 -0
  48. package/dist/chunk-PC5DOSM7.js.map +1 -0
  49. package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
  50. package/dist/chunk-QB6BDBP2.js.map +1 -0
  51. package/dist/chunk-RXHCETDZ.js +536 -0
  52. package/dist/chunk-RXHCETDZ.js.map +1 -0
  53. package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
  54. package/dist/chunk-SFLLL76A.js.map +1 -0
  55. package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
  56. package/dist/chunk-T6RLYGAD.js.map +1 -0
  57. package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
  58. package/dist/chunk-TJVT4QFF.js.map +1 -0
  59. package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
  60. package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
  61. package/dist/chunk-U4L7JRPZ.js.map +1 -0
  62. package/dist/chunk-U4PHLT2N.js +419 -0
  63. package/dist/chunk-U4PHLT2N.js.map +1 -0
  64. package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
  65. package/dist/chunk-VCZ5FQYW.js.map +1 -0
  66. package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
  67. package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
  68. package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
  69. package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
  70. package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
  71. package/dist/chunk-ZHTZ4EYI.js.map +1 -0
  72. package/dist/cli.js +6 -5
  73. package/dist/cli.js.map +1 -1
  74. package/dist/contract/index.d.ts +47 -87
  75. package/dist/contract/index.js +14 -13
  76. package/dist/contract/index.js.map +1 -1
  77. package/dist/control.js +3 -2
  78. package/dist/fuzz.js +3 -2
  79. package/dist/fuzz.js.map +1 -1
  80. package/dist/index.d.ts +659 -203
  81. package/dist/index.js +145 -117
  82. package/dist/index.js.map +1 -1
  83. package/dist/meta-eval/index.js +2 -2
  84. package/dist/multishot/index.d.ts +3 -4
  85. package/dist/multishot/index.js.map +1 -1
  86. package/dist/openapi.json +1 -1
  87. package/dist/pipelines/index.js +5 -5
  88. package/dist/reporting.d.ts +14 -0
  89. package/dist/reporting.js +7 -6
  90. package/dist/rl.d.ts +652 -82
  91. package/dist/rl.js +415 -171
  92. package/dist/rl.js.map +1 -1
  93. package/dist/rollout/index.d.ts +1071 -32
  94. package/dist/rollout/index.js +68 -10
  95. package/dist/run-campaign-OJJ7CZF4.js +18 -0
  96. package/dist/supervisor-run/index.d.ts +114 -4
  97. package/dist/supervisor-run/index.js +4 -3
  98. package/dist/traces.d.ts +1 -1
  99. package/dist/traces.js +6 -5
  100. package/dist/wire/index.d.ts +10 -11
  101. package/dist/wire/index.js +3 -3
  102. package/docs/feature-guide.md +1 -1
  103. package/docs/rollout.md +116 -2
  104. package/package.json +4 -4
  105. package/dist/chunk-2JX3CFMB.js.map +0 -1
  106. package/dist/chunk-DJKY2TSY.js.map +0 -1
  107. package/dist/chunk-EJGRPCO3.js.map +0 -1
  108. package/dist/chunk-EOSZT7PL.js.map +0 -1
  109. package/dist/chunk-EZJEIH2R.js.map +0 -1
  110. package/dist/chunk-IHQDPH7D.js.map +0 -1
  111. package/dist/chunk-MHELPNRP.js.map +0 -1
  112. package/dist/chunk-NACAGYSY.js.map +0 -1
  113. package/dist/chunk-NKAGIDE2.js.map +0 -1
  114. package/dist/chunk-NYLOYM6N.js.map +0 -1
  115. package/dist/chunk-PBE2LOSS.js.map +0 -1
  116. package/dist/chunk-TT4KNT67.js +0 -124
  117. package/dist/chunk-TT4KNT67.js.map +0 -1
  118. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  119. package/dist/chunk-UWZZKKU7.js +0 -237
  120. package/dist/chunk-UWZZKKU7.js.map +0 -1
  121. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  122. package/dist/chunk-VGRCHJON.js.map +0 -1
  123. package/dist/chunk-VLOATJQ2.js.map +0 -1
  124. package/dist/chunk-VZSRQ272.js.map +0 -1
  125. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  126. package/dist/chunk-XDWDC2MP.js.map +0 -1
  127. package/dist/chunk-XPRT64IE.js.map +0 -1
  128. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  129. /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
  130. /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
  131. /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
  132. /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
  133. /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
  134. /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
  135. /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
  136. /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
  137. /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
@@ -13,7 +13,7 @@ import {
13
13
  runCampaign,
14
14
  summarizeAgentReceiptIntegrity,
15
15
  tryAcquireAtomicFileLock
16
- } from "./chunk-EZJEIH2R.js";
16
+ } from "./chunk-C6LXANRU.js";
17
17
  import {
18
18
  clamp01,
19
19
  combineAbortSignals,
@@ -21,25 +21,25 @@ import {
21
21
  } from "./chunk-WGXIEX7P.js";
22
22
  import {
23
23
  detectRewardHacking
24
- } from "./chunk-NYLOYM6N.js";
24
+ } from "./chunk-EG66UGL4.js";
25
25
  import {
26
26
  campaignCellExecutionEvidence,
27
27
  projectCampaignCellQuality
28
- } from "./chunk-2MKQIFS4.js";
28
+ } from "./chunk-E7QXT7SX.js";
29
29
  import {
30
30
  costReceiptFromLlm,
31
31
  costReceiptFromLlmError,
32
32
  maximumChargeForLlmRequest,
33
33
  stripFencedJson
34
- } from "./chunk-PBE2LOSS.js";
34
+ } from "./chunk-SFLLL76A.js";
35
35
  import {
36
36
  pairedBootstrap,
37
37
  weightedComposite
38
- } from "./chunk-MHELPNRP.js";
38
+ } from "./chunk-ZHTZ4EYI.js";
39
39
  import {
40
40
  CostLedger,
41
41
  costForTokenPricing
42
- } from "./chunk-WS3NZZQQ.js";
42
+ } from "./chunk-VCZ5FQYW.js";
43
43
  import {
44
44
  DEFAULT_REDACTION_RULES
45
45
  } from "./chunk-GGE4NNQT.js";
@@ -48,36 +48,10 @@ import {
48
48
  ValidationError
49
49
  } from "./chunk-ONWEPEDO.js";
50
50
 
51
- // src/tcloud-cost.ts
52
- function maximumChargeForTCloudRequest(request, maximumAttempts) {
53
- if (maximumAttempts === void 0) return void 0;
54
- return maximumChargeForLlmRequest(request, { maxRetries: maximumAttempts });
55
- }
56
- function costReceiptFromTCloud(response, requestedModel) {
57
- const usage = response.usage;
58
- const inputTokens = tokenCount(usage?.prompt_tokens);
59
- const outputTokens = tokenCount(usage?.completion_tokens);
60
- const totalTokens = tokenCount(usage?.total_tokens);
61
- const usageUnknown = inputTokens === void 0 || outputTokens === void 0 || totalTokens !== void 0 && totalTokens !== inputTokens + outputTokens;
62
- return {
63
- model: response.model || requestedModel,
64
- inputTokens: inputTokens ?? 0,
65
- outputTokens: outputTokens ?? 0,
66
- costUnknown: usageUnknown,
67
- usageUnknown
68
- };
69
- }
70
- function tokenCount(value) {
71
- return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : void 0;
72
- }
73
-
74
51
  // src/judges.ts
75
52
  var JudgeParseError = class extends JudgeError {
76
- /** Name of the judge whose response failed to parse. */
77
53
  judgeName;
78
- /** The raw (truncated) model response that failed to parse. */
79
54
  raw;
80
- /** Paid-call metadata remains available even when the verdict is unusable. */
81
55
  llmCall;
82
56
  constructor(judgeName, raw, options) {
83
57
  super(`judge '${judgeName}' returned an unparseable response: ${raw.slice(0, 200)}`, options);
@@ -86,224 +60,6 @@ var JudgeParseError = class extends JudgeError {
86
60
  this.llmCall = options?.llmCall;
87
61
  }
88
62
  };
89
- function createDomainExpertJudge(domain) {
90
- return async (tc, input) => {
91
- const { scenario, turns } = input;
92
- const conversation = turns.map(
93
- (t, i) => `Turn ${i + 1}:
94
- User: ${t.userMessage}
95
- Agent: ${t.agentResponse.slice(0, 2e3)}`
96
- ).join("\n\n---\n\n");
97
- const resp = await runJudgeChat(tc, input, "domain_expert", {
98
- model: "gpt-4o",
99
- messages: [
100
- {
101
- role: "system",
102
- content: `You are a senior ${domain} professional with 20+ years of experience. You are evaluating an AI agent's responses for professional accuracy and depth.
103
-
104
- Score STRICTLY. A 5 means "a junior professional could do this." An 8 means "solid mid-career work." A 10 means "I would hire this agent."
105
-
106
- Evaluate:
107
- 1. **domain_accuracy** (0-10): Are the technical terms correct? Are the recommendations what you'd actually do? Would this advice cause problems if followed?
108
- 2. **professional_depth** (0-10): Does it go beyond surface-level? Does it consider practical constraints, edge cases, industry standards? Or is it generic textbook advice?
109
-
110
- Respond with JSON only: [{"dimension":"domain_accuracy","score":N,"reasoning":"...","evidence":"quote from response"},{"dimension":"professional_depth","score":N,"reasoning":"...","evidence":"quote"}]`
111
- },
112
- {
113
- role: "user",
114
- content: `Persona: ${scenario.persona} (${scenario.label})
115
- Scenario: ${scenario.thesis}
116
-
117
- ${conversation}`
118
- }
119
- ],
120
- temperature: 0.1,
121
- maxTokens: 800
122
- });
123
- return parseJudgeResponse("domain_expert", resp);
124
- };
125
- }
126
- var codeExecutionJudge = async (tc, input) => {
127
- const { scenario, artifacts } = input;
128
- const codeBlocks = artifacts.codeBlocks;
129
- if (codeBlocks.length === 0) {
130
- return [
131
- {
132
- judgeName: "code_execution",
133
- dimension: "code_execution",
134
- score: 0,
135
- reasoning: "No code blocks found in agent response."
136
- }
137
- ];
138
- }
139
- const codeText = codeBlocks.map(
140
- (b, i) => `Block ${i + 1} (${b.language}):
141
- \`\`\`${b.language}
142
- ${b.code.slice(0, 3e3)}
143
- \`\`\``
144
- ).join("\n\n");
145
- const resp = await runJudgeChat(tc, input, "code_execution", {
146
- model: "gpt-4o",
147
- messages: [
148
- {
149
- role: "system",
150
- content: `You are a principal software engineer reviewing code written by an AI agent.
151
-
152
- Score STRICTLY:
153
- 1. **executability** (0-10): Would this code run without errors? Check: import errors, undefined variables, missing deps, syntax errors. A 5 means "would run with minor fixes." A 10 means "copy-paste and it works."
154
- 2. **completeness** (0-10): Does it handle the FULL task, or just the happy path? A 5 means "handles the main case." A 10 means "production-ready."
155
- 3. **reusability** (0-10): Could this be saved as a tool and reused? A 5 means "works for this case." A 10 means "general-purpose tool."
156
-
157
- Respond with JSON only: [{"dimension":"executability","score":N,"reasoning":"...","evidence":"specific line/issue"},{"dimension":"completeness","score":N,"reasoning":"...","evidence":"..."},{"dimension":"reusability","score":N,"reasoning":"...","evidence":"..."}]`
158
- },
159
- {
160
- role: "user",
161
- content: `Task: ${scenario.thesis}
162
-
163
- ${codeText}`
164
- }
165
- ],
166
- temperature: 0.1,
167
- maxTokens: 1e3
168
- });
169
- return parseJudgeResponse("code_execution", resp);
170
- };
171
- var coherenceJudge = async (tc, input) => {
172
- const { scenario, turns } = input;
173
- if (turns.length < 2) {
174
- return [];
175
- }
176
- const conversation = turns.map(
177
- (t, i) => `Turn ${i + 1}:
178
- User: ${t.userMessage}
179
- Agent (${t.agentResponse.length} chars): ${t.agentResponse.slice(0, 1500)}`
180
- ).join("\n\n---\n\n");
181
- const resp = await runJudgeChat(tc, input, "coherence", {
182
- model: "gpt-4o",
183
- messages: [
184
- {
185
- role: "system",
186
- content: `You evaluate whether an AI agent maintains coherence across a multi-turn conversation.
187
-
188
- Score STRICTLY:
189
- 1. **consistency** (0-10): Does the agent contradict itself across turns? Does it remember what it said/built earlier?
190
- 2. **progression** (0-10): Does each turn BUILD on the previous? Or does it start fresh? A 5 means "vaguely related." A 10 means "each turn clearly advances the work."
191
- 3. **feedback_integration** (0-10): When the user gives feedback, does the agent demonstrate it HEARD the feedback?
192
-
193
- Respond with JSON only: [{"dimension":"consistency","score":N,"reasoning":"..."},{"dimension":"progression","score":N,"reasoning":"..."},{"dimension":"feedback_integration","score":N,"reasoning":"..."}]`
194
- },
195
- {
196
- role: "user",
197
- content: `Scenario: ${scenario.thesis}
198
-
199
- ${conversation}`
200
- }
201
- ],
202
- temperature: 0.1,
203
- maxTokens: 800
204
- });
205
- return parseJudgeResponse("coherence", resp);
206
- };
207
- var adversarialJudge = async (tc, input) => {
208
- const { scenario, turns } = input;
209
- const conversation = turns.map(
210
- (t, i) => `Turn ${i + 1}:
211
- User: ${t.userMessage}
212
- Agent: ${t.agentResponse.slice(0, 1500)}`
213
- ).join("\n\n---\n\n");
214
- const resp = await runJudgeChat(tc, input, "adversarial", {
215
- model: "gpt-4o",
216
- messages: [
217
- {
218
- role: "system",
219
- content: `You are a red-team evaluator. Find the WEAKEST point in the AI agent's responses. Be adversarial.
220
-
221
- 1. **hallucination** (0-10, inverted \u2014 10 means NO hallucination): Did the agent make up facts, cite nonexistent tools, invent standards?
222
- 2. **false_confidence** (0-10, inverted \u2014 10 means appropriate uncertainty): Did the agent present uncertain information as fact?
223
- 3. **worst_failure** (0-10, inverted \u2014 10 means no critical failures): What is the single worst thing in the response?
224
-
225
- Be harsh. If everything is genuinely good, say so \u2014 but look hard first.
226
-
227
- Respond with JSON only: [{"dimension":"hallucination","score":N,"reasoning":"...","evidence":"specific quote"},{"dimension":"false_confidence","score":N,"reasoning":"...","evidence":"..."},{"dimension":"worst_failure","score":N,"reasoning":"...","evidence":"..."}]`
228
- },
229
- {
230
- role: "user",
231
- content: `Persona: ${scenario.persona}
232
- Scenario: ${scenario.thesis}
233
-
234
- ${conversation}`
235
- }
236
- ],
237
- temperature: 0.2,
238
- maxTokens: 800
239
- });
240
- return parseJudgeResponse("adversarial", resp);
241
- };
242
- function createCustomJudge(name, systemPrompt, opts) {
243
- return async (tc, input) => {
244
- const { scenario, turns } = input;
245
- const conversation = turns.map(
246
- (t, i) => `Turn ${i + 1}:
247
- User: ${t.userMessage}
248
- Agent: ${t.agentResponse.slice(0, 2e3)}`
249
- ).join("\n\n---\n\n");
250
- const resp = await runJudgeChat(tc, input, name, {
251
- model: opts?.model ?? "gpt-4o",
252
- messages: [
253
- {
254
- role: "system",
255
- content: systemPrompt
256
- },
257
- {
258
- role: "user",
259
- content: `Persona: ${scenario.persona} (${scenario.label})
260
- Scenario: ${scenario.thesis}
261
-
262
- ${conversation}`
263
- }
264
- ],
265
- temperature: opts?.temperature ?? 0.1,
266
- maxTokens: opts?.maxTokens ?? 1e3
267
- });
268
- return parseJudgeResponse(name, resp);
269
- };
270
- }
271
- function defaultJudges(domain) {
272
- return [createDomainExpertJudge(domain), codeExecutionJudge, coherenceJudge, adversarialJudge];
273
- }
274
- function parseJudgeResponse(judgeName, resp) {
275
- const content = resp.choices?.[0]?.message?.content ?? "";
276
- try {
277
- let cleaned = content.replace(/```json\n?|\n?```/g, "").trim();
278
- const arrayMatch = cleaned.match(/\[[\s\S]*\]/);
279
- if (arrayMatch) cleaned = arrayMatch[0];
280
- const parsed = JSON.parse(cleaned);
281
- return parsed.map((p) => ({
282
- judgeName,
283
- dimension: p.dimension,
284
- score: Math.max(0, Math.min(10, p.score)),
285
- reasoning: p.reasoning ?? "",
286
- evidence: p.evidence
287
- }));
288
- } catch (err) {
289
- throw new JudgeParseError(judgeName, content, { cause: err });
290
- }
291
- }
292
- async function runJudgeChat(tc, input, judgeName, request) {
293
- const paid = await (input.costLedger ?? new CostLedger()).runPaidCall({
294
- channel: "judge",
295
- phase: input.costPhase ?? "judge",
296
- actor: `legacy-judge.${judgeName}`,
297
- model: request.model,
298
- maximumCharge: maximumChargeForTCloudRequest(request, input.tcloudMaximumAttempts),
299
- tags: input.costTags,
300
- signal: input.signal,
301
- execute: () => tc.chat(request),
302
- receipt: (response) => costReceiptFromTCloud(response, request.model)
303
- });
304
- if (!paid.succeeded) throw paid.error;
305
- return paid.value;
306
- }
307
63
 
308
64
  // src/pareto.ts
309
65
  function dominates(a, b, objectives) {
@@ -473,7 +229,7 @@ ${renderContract(dimensions, scale)}`;
473
229
  actor: name,
474
230
  model,
475
231
  maximumCharge: opts.chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, {
476
- maxRetries: opts.chat.maximumAttempts
232
+ maximumAttempts: opts.chat.maximumAttempts
477
233
  }),
478
234
  tags: { ...costTags, scenarioId: scenario.id },
479
235
  signal,
@@ -1194,7 +950,7 @@ function renderPrBody(result, gate, diff) {
1194
950
  lines.push(
1195
951
  `**Cells**: executed ${result.aggregates.cellsExecuted}, cached ${result.aggregates.cellsCached}, skipped ${result.aggregates.cellsSkipped}, failed ${result.aggregates.cellsFailed}`
1196
952
  );
1197
- lines.push(`**Total spend**: $${result.aggregates.totalCostUsd.toFixed(2)}`);
953
+ lines.push(`**Total spend**: $${result.aggregates.cost.totalCostUsd.toFixed(2)}`);
1198
954
  lines.push("");
1199
955
  lines.push(`### Gate verdict: \`${gate.decision}\``);
1200
956
  lines.push("");
@@ -2142,12 +1898,6 @@ function assertComparisonControls(opts, seed, resamples, confidence) {
2142
1898
  }
2143
1899
  }
2144
1900
  function assertComparisonPartitions(opts) {
2145
- const legacy = opts;
2146
- if (legacy.holdoutScenarios !== void 0) {
2147
- throw new Error(
2148
- "compareOptimizationMethods: holdoutScenarios is ambiguous and no longer accepted. Provide disjoint trainScenarios, selectionScenarios, and testScenarios; selection may be reused adaptively, test must remain untouched."
2149
- );
2150
- }
2151
1901
  const partitions = [
2152
1902
  { name: "trainScenarios", scenarios: opts.trainScenarios },
2153
1903
  { name: "selectionScenarios", scenarios: opts.selectionScenarios },
@@ -2329,7 +2079,7 @@ function assertComparisonCost(cost, label) {
2329
2079
 
2330
2080
  // src/campaign/single-run-lock.ts
2331
2081
  function assertAvailable(path) {
2332
- const unavailable = probeAtomicFileLock({ lockPath: path, acceptLegacyPid: true });
2082
+ const unavailable = probeAtomicFileLock({ lockPath: path });
2333
2083
  if (unavailable) throw unavailableError(path, unavailable);
2334
2084
  }
2335
2085
  function unavailableError(path, unavailable) {
@@ -2349,8 +2099,7 @@ function acquireSingleRunLock(opts) {
2349
2099
  for (const path of opts.alsoCheck ?? []) assertAvailable(path);
2350
2100
  const acquisition = tryAcquireAtomicFileLock({
2351
2101
  lockPath: opts.lockPath,
2352
- pid,
2353
- acceptLegacyPid: true
2102
+ pid
2354
2103
  });
2355
2104
  if (!acquisition.acquired) throw unavailableError(opts.lockPath, acquisition);
2356
2105
  const release = () => {
@@ -6349,7 +6098,7 @@ async function runImprovementLoop(opts) {
6349
6098
  const dispatchTimeoutMs = opts.dispatchTimeoutMs ?? DEFAULT_DISPATCH_TIMEOUT_MS;
6350
6099
  const optimization = await runOptimization({ ...opts, dispatchTimeoutMs, costLedger });
6351
6100
  const winnerIsBaseline = optimization.winnerSurfaceHash === surfaceHash(opts.baselineSurface);
6352
- const { runCampaign: runCampaign2 } = await import("./run-campaign-ISHFZ7FJ.js");
6101
+ const { runCampaign: runCampaign2 } = await import("./run-campaign-OJJ7CZF4.js");
6353
6102
  const holdoutDeferred = (opts.holdout ?? "measured") === "deferred";
6354
6103
  const baselineOnHoldout = holdoutDeferred ? await runCampaign2({
6355
6104
  ...opts,
@@ -6464,8 +6213,8 @@ async function runImprovementLoop(opts) {
6464
6213
  neutralizedJudgeScores,
6465
6214
  scenarios: opts.holdoutScenarios,
6466
6215
  cost: {
6467
- candidate: winnerOnHoldout.aggregates.totalCostUsd,
6468
- baseline: baselineOnHoldout.aggregates.totalCostUsd
6216
+ candidate: winnerOnHoldout.aggregates.cost.totalCostUsd,
6217
+ baseline: baselineOnHoldout.aggregates.cost.totalCostUsd
6469
6218
  },
6470
6219
  costLedger,
6471
6220
  costPhase: "promotion.gate",
@@ -7094,7 +6843,7 @@ function snapshotFromHoldout(index, surfaceHash2, surface, campaign) {
7094
6843
  surface,
7095
6844
  cells,
7096
6845
  compositeMean: campaignMeanCompositeOrNull(campaign),
7097
- costUsd: campaign.aggregates.totalCostUsd,
6846
+ costUsd: campaign.aggregates.cost.totalCostUsd,
7098
6847
  durationMs: campaign.durationMs
7099
6848
  };
7100
6849
  }
@@ -7553,15 +7302,7 @@ function skillOptOptimizationMethod(config) {
7553
7302
  }
7554
7303
 
7555
7304
  export {
7556
- maximumChargeForTCloudRequest,
7557
- costReceiptFromTCloud,
7558
7305
  JudgeParseError,
7559
- createDomainExpertJudge,
7560
- codeExecutionJudge,
7561
- coherenceJudge,
7562
- adversarialJudge,
7563
- createCustomJudge,
7564
- defaultJudges,
7565
7306
  recoverTruncatedJson,
7566
7307
  dominates,
7567
7308
  paretoFrontier,
@@ -7630,4 +7371,4 @@ export {
7630
7371
  emitLoopProvenance,
7631
7372
  skillOptOptimizationMethod
7632
7373
  };
7633
- //# sourceMappingURL=chunk-NKAGIDE2.js.map
7374
+ //# sourceMappingURL=chunk-2QU3YOPR.js.map