@tangle-network/agent-eval 0.115.3 → 0.117.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. package/CHANGELOG.md +55 -0
  2. package/dist/analyst/index.d.ts +16 -11
  3. package/dist/analyst/index.js +33 -25
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
  6. package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
  7. package/dist/belief-state/index.d.ts +6 -6
  8. package/dist/belief-state/index.js +1 -1
  9. package/dist/benchmarks/index.d.ts +12 -5
  10. package/dist/benchmarks/index.js +11 -10
  11. package/dist/builder-eval/index.d.ts +4 -4
  12. package/dist/builder-eval/index.js +1 -1
  13. package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
  14. package/dist/campaign/index.d.ts +247 -34
  15. package/dist/campaign/index.js +33 -13
  16. package/dist/chunk-3YYRZDON.js +45 -0
  17. package/dist/chunk-3YYRZDON.js.map +1 -0
  18. package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
  19. package/dist/{chunk-WSBUZMBU.js → chunk-CCZIVI3F.js} +54 -115
  20. package/dist/chunk-CCZIVI3F.js.map +1 -0
  21. package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
  22. package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
  23. package/dist/chunk-HHWE3POT.js +94 -0
  24. package/dist/chunk-HHWE3POT.js.map +1 -0
  25. package/dist/{chunk-ADYLPOSX.js → chunk-HQPHZGL6.js} +1112 -135
  26. package/dist/chunk-HQPHZGL6.js.map +1 -0
  27. package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
  28. package/dist/chunk-IDZTTFRR.js.map +1 -0
  29. package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
  30. package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
  31. package/dist/chunk-LQUTGLOZ.js.map +1 -0
  32. package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
  33. package/dist/chunk-LTVG32KX.js.map +1 -0
  34. package/dist/{chunk-5S5NJ63F.js → chunk-MGEHEHSN.js} +807 -15
  35. package/dist/chunk-MGEHEHSN.js.map +1 -0
  36. package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
  37. package/dist/chunk-NJC7U437.js.map +1 -0
  38. package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
  39. package/dist/chunk-ODVOOEWQ.js.map +1 -0
  40. package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
  41. package/dist/chunk-S2F4J57L.js.map +1 -0
  42. package/dist/chunk-VCTY3W6J.js +798 -0
  43. package/dist/chunk-VCTY3W6J.js.map +1 -0
  44. package/dist/chunk-VF3XSYTI.js +545 -0
  45. package/dist/chunk-VF3XSYTI.js.map +1 -0
  46. package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
  47. package/dist/chunk-YZPO4UHR.js.map +1 -0
  48. package/dist/{chunk-KG4TD7EQ.js → chunk-ZUXV7UWZ.js} +1425 -697
  49. package/dist/chunk-ZUXV7UWZ.js.map +1 -0
  50. package/dist/cli.js +4 -2
  51. package/dist/cli.js.map +1 -1
  52. package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
  53. package/dist/contract/index.d.ts +45 -31
  54. package/dist/contract/index.js +58 -19
  55. package/dist/contract/index.js.map +1 -1
  56. package/dist/{control-CcBiAEnn.d.ts → control-6vuGfmDH.d.ts} +5 -5
  57. package/dist/control.d.ts +6 -6
  58. package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
  59. package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DaK8b3fv.d.ts} +2 -1
  60. package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
  61. package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
  62. package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
  63. package/dist/fuzz.d.ts +8 -16
  64. package/dist/fuzz.js +72 -42
  65. package/dist/fuzz.js.map +1 -1
  66. package/dist/{gepa-dne9JDPL.d.ts → gepa-eESocoDi.d.ts} +64 -12
  67. package/dist/hosted/index.d.ts +14 -7
  68. package/dist/{index-BTEpx9He.d.ts → index-PdX4VnPA.d.ts} +3 -3
  69. package/dist/index.d.ts +97 -55
  70. package/dist/index.js +343 -244
  71. package/dist/index.js.map +1 -1
  72. package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
  73. package/dist/{integrity-qemeBAyx.d.ts → integrity-DqlBiLyK.d.ts} +1 -1
  74. package/dist/kind-factory-ClZmO25A.d.ts +171 -0
  75. package/dist/{llm-client-DyqEH4jH.d.ts → llm-client-qoDd18Qz.d.ts} +27 -3
  76. package/dist/meta-eval/index.d.ts +8 -7
  77. package/dist/meta-eval/index.js +1 -1
  78. package/dist/multishot/index.d.ts +10 -3
  79. package/dist/openapi.json +1 -1
  80. package/dist/pipelines/index.d.ts +16 -6
  81. package/dist/pipelines/index.js +119 -23
  82. package/dist/pipelines/index.js.map +1 -1
  83. package/dist/{kind-factory-DcNg13sZ.d.ts → policy-edit-wG9uFEFm.d.ts} +114 -167
  84. package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration-BWQhJ3vz.d.ts} +23 -4
  85. package/dist/{provenance-Bibyg1U9.d.ts → provenance-DpjwyseI.d.ts} +28 -16
  86. package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
  87. package/dist/{release-report-CCtzajxP.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
  88. package/dist/reporting.d.ts +10 -9
  89. package/dist/{researcher-Dq-EtpbE.d.ts → researcher-C8XyxQsu.d.ts} +7 -7
  90. package/dist/rl.d.ts +17 -12
  91. package/dist/rl.js +2 -2
  92. package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
  93. package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
  94. package/dist/{run-record-B7RTi_ix.d.ts → run-record-BDH49H2E.d.ts} +2 -2
  95. package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
  96. package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
  97. package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +23 -5
  98. package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
  99. package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
  100. package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
  101. package/dist/storyboard/index.d.ts +1 -1
  102. package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
  103. package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
  104. package/dist/traces.d.ts +19 -10
  105. package/dist/traces.js +16 -4
  106. package/dist/{types-C5gJrOVT.d.ts → types-BSw1rOUB.d.ts} +97 -38
  107. package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
  108. package/dist/wire/index.d.ts +28 -19
  109. package/dist/wire/index.js +4 -2
  110. package/docs/design/loop-taxonomy.md +1 -2
  111. package/docs/distributed-driver.md +1 -1
  112. package/package.json +3 -3
  113. package/dist/chunk-4D5RVB3W.js.map +0 -1
  114. package/dist/chunk-5S5NJ63F.js.map +0 -1
  115. package/dist/chunk-ADYLPOSX.js.map +0 -1
  116. package/dist/chunk-FAOEFFRT.js.map +0 -1
  117. package/dist/chunk-GY4SYVPJ.js.map +0 -1
  118. package/dist/chunk-I6LVHOV3.js +0 -205
  119. package/dist/chunk-I6LVHOV3.js.map +0 -1
  120. package/dist/chunk-KG4TD7EQ.js.map +0 -1
  121. package/dist/chunk-LNQEP766.js.map +0 -1
  122. package/dist/chunk-MHNQWM4I.js.map +0 -1
  123. package/dist/chunk-NYFUT3B3.js.map +0 -1
  124. package/dist/chunk-QMXXSNC4.js +0 -761
  125. package/dist/chunk-QMXXSNC4.js.map +0 -1
  126. package/dist/chunk-TLDB7WRY.js.map +0 -1
  127. package/dist/chunk-WSBUZMBU.js.map +0 -1
  128. package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
  129. package/dist/policy-edit-RLn8GWof.d.ts +0 -103
  130. /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
  131. /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
  132. /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
  133. /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
  134. /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
package/dist/index.js CHANGED
@@ -9,12 +9,12 @@ import {
9
9
  checkBehavioralCanary,
10
10
  checkCanaries,
11
11
  runBehavioralCanaries
12
- } from "./chunk-J6P6PK2R.js";
12
+ } from "./chunk-FQNLDL4D.js";
13
13
  import {
14
14
  BENCHMARK_SPLIT_SEED,
15
15
  benchmarks_exports,
16
16
  deterministicSplit
17
- } from "./chunk-3LXTCTWL.js";
17
+ } from "./chunk-JSDVRFAP.js";
18
18
  import {
19
19
  DEFAULT_RULES,
20
20
  buildTrajectory,
@@ -23,7 +23,7 @@ import {
23
23
  computeToolUseMetrics,
24
24
  iqr,
25
25
  welchsTTest
26
- } from "./chunk-NYFUT3B3.js";
26
+ } from "./chunk-ODVOOEWQ.js";
27
27
  import {
28
28
  analyzeSeries
29
29
  } from "./chunk-BOD4O7OF.js";
@@ -40,40 +40,45 @@ import {
40
40
  import {
41
41
  CODING_HARNESSES,
42
42
  HARNESS_NATIVE_MODEL,
43
- JudgeParseError,
44
- adversarialJudge,
45
43
  agentProfileHash,
46
44
  agentProfileId,
47
45
  agentProfileModelId,
48
- codeExecutionJudge,
49
- coherenceJudge,
50
46
  comparePairedArms,
51
47
  completionVerdict,
52
- createCustomJudge,
53
- createDomainExpertJudge,
54
48
  createLlmCorrectnessChecker,
55
49
  createTokenRecallChecker,
56
- defaultJudges,
57
50
  expandProfileAxes,
58
51
  extractProducedState,
59
52
  harnessAxisOf,
60
- llmJudge,
61
53
  pairArms,
62
54
  parseCorrectnessResponse,
63
55
  verifyCompletion
64
- } from "./chunk-KG4TD7EQ.js";
56
+ } from "./chunk-ZUXV7UWZ.js";
65
57
  import {
66
58
  DEFAULT_MUTATION_PRIMITIVES,
67
59
  DEFAULT_RED_TEAM_CORPUS,
68
60
  Dataset,
69
61
  HoldoutLockedError,
62
+ JudgeParseError,
63
+ REFERENCE_EQUIVALENCE_INPUT_LIMITS,
64
+ REFERENCE_EQUIVALENCE_JUDGE_VERSION,
65
+ adversarialJudge,
70
66
  buildReflectionPrompt,
71
67
  campaignMeanComposite,
68
+ codeExecutionJudge,
69
+ coherenceJudge,
70
+ costReceiptFromTCloud,
71
+ createCustomJudge,
72
+ createDomainExpertJudge,
73
+ createReferenceEquivalenceJudge,
72
74
  crowdingDistance,
75
+ defaultJudges,
73
76
  dominates,
74
77
  gepaProposer,
75
78
  hashScenarios,
76
79
  heldOutGate,
80
+ llmJudge,
81
+ maximumChargeForTCloudRequest,
77
82
  paretoFrontier,
78
83
  paretoFrontierWithCrowding,
79
84
  parseReflectionResponse,
@@ -81,20 +86,12 @@ import {
81
86
  redTeamReport,
82
87
  runCanaries,
83
88
  runImprovementLoop,
89
+ runReferenceEquivalenceJudge,
84
90
  scalarScore,
85
91
  scoreRedTeamOutput,
86
92
  surfaceContentHash,
87
93
  toolNamesForRun
88
- } from "./chunk-ADYLPOSX.js";
89
- import {
90
- MODEL_PRICING,
91
- MetricsCollector,
92
- TokenCounter,
93
- estimateCost,
94
- estimateTokens,
95
- isModelPriced,
96
- resolveModelPricing
97
- } from "./chunk-VI2UW6B6.js";
94
+ } from "./chunk-HQPHZGL6.js";
98
95
  import {
99
96
  BackendIntegrityError,
100
97
  assertRealBackend,
@@ -104,7 +101,7 @@ import {
104
101
  fileVerdictCache,
105
102
  inMemoryVerdictCache,
106
103
  summarizeBackendIntegrity
107
- } from "./chunk-FAOEFFRT.js";
104
+ } from "./chunk-IDZTTFRR.js";
108
105
  import {
109
106
  DEFAULT_COMPLEXITY_WEIGHTS,
110
107
  FindingsStore,
@@ -114,46 +111,50 @@ import {
114
111
  SKILL_USAGE_ANALYST,
115
112
  SkillUsageAnalyst,
116
113
  createAnalystAi,
117
- createChatClient,
118
114
  createSemanticConceptJudge,
119
115
  defaultIsMaterial,
120
116
  diffFindings,
121
117
  runSemanticConceptJudge
122
- } from "./chunk-WSBUZMBU.js";
118
+ } from "./chunk-CCZIVI3F.js";
123
119
  import {
124
120
  buildDefaultAnalystRegistry,
125
- computeTraceMetrics
126
- } from "./chunk-I6LVHOV3.js";
121
+ computeTraceMetrics,
122
+ createChatClient
123
+ } from "./chunk-VF3XSYTI.js";
124
+ import "./chunk-HHWE3POT.js";
125
+ import {
126
+ Mutex
127
+ } from "./chunk-3YYRZDON.js";
127
128
  import {
129
+ AnalystRegistry,
128
130
  DEFAULT_RUN_SCORE_WEIGHTS,
129
- Mutex,
131
+ DEFAULT_TRACE_ANALYST_KINDS,
132
+ FAILURE_MODE_KIND_SPEC,
133
+ IMPROVEMENT_KIND_SPEC,
134
+ KNOWLEDGE_GAP_KIND_SPEC,
135
+ KNOWLEDGE_POISONING_KIND_SPEC,
130
136
  POLICY_EDIT_AXES,
137
+ POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
131
138
  POLICY_EDIT_TARGET_SURFACES,
132
139
  PolicyEditValidationError,
133
140
  admitPolicyEdit,
134
141
  aggregateRunScore,
135
142
  applyPolicyEditToSurface,
136
143
  clamp01,
144
+ computeFindingId,
137
145
  computePolicyEditId,
146
+ createTraceAnalystKind,
138
147
  isPolicyEdit,
148
+ makeFinding,
139
149
  makePolicyEdit,
150
+ makePolicyEditCandidateRecord,
140
151
  policyEditFromFinding,
141
152
  policyEditsFromFindings,
153
+ renderPriorFindings,
142
154
  scorePolicyEditReadiness,
143
- validatePolicyEdit
144
- } from "./chunk-QMXXSNC4.js";
145
- import {
146
- AnalystRegistry,
147
- DEFAULT_TRACE_ANALYST_KINDS,
148
- FAILURE_MODE_KIND_SPEC,
149
- IMPROVEMENT_KIND_SPEC,
150
- KNOWLEDGE_GAP_KIND_SPEC,
151
- KNOWLEDGE_POISONING_KIND_SPEC,
152
- computeFindingId,
153
- createTraceAnalystKind,
154
- makeFinding,
155
- renderPriorFindings
156
- } from "./chunk-5S5NJ63F.js";
155
+ validatePolicyEdit,
156
+ validatePolicyEditCandidateRecord
157
+ } from "./chunk-MGEHEHSN.js";
157
158
  import {
158
159
  allCriticalPassed,
159
160
  controlFailureClassFromVerification,
@@ -184,7 +185,7 @@ import {
184
185
  } from "./chunk-MOXWMGPC.js";
185
186
  import {
186
187
  runEvalCampaign
187
- } from "./chunk-ONM6PEAE.js";
188
+ } from "./chunk-GQCZRZ7L.js";
188
189
  import "./chunk-ARU2PZFM.js";
189
190
  import {
190
191
  evaluateInterimReleaseConfidence,
@@ -267,13 +268,14 @@ import {
267
268
  scoreTraceInsightReadiness,
268
269
  tokenizeDomainWords,
269
270
  traceAnalystOnRunComplete
270
- } from "./chunk-TLDB7WRY.js";
271
+ } from "./chunk-YZPO4UHR.js";
271
272
  import {
272
273
  FAILURE_CLASSES,
273
274
  TRACE_SCHEMA_VERSION,
274
275
  aggregateLlm,
275
276
  argHash,
276
277
  groupBy,
278
+ hasCapturedToolArgs,
277
279
  isJudgeSpan,
278
280
  isLlmSpan,
279
281
  isRetrievalSpan,
@@ -284,13 +286,13 @@ import {
284
286
  runFailureClass,
285
287
  runsForScenario,
286
288
  toolSpans
287
- } from "./chunk-MHNQWM4I.js";
289
+ } from "./chunk-LQUTGLOZ.js";
288
290
  import {
289
291
  TRACE_ANALYST_ACTOR_DESCRIPTION,
290
292
  TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
291
293
  TRACE_ANALYST_SUBAGENT_DESCRIPTION,
292
294
  analyzeTraces
293
- } from "./chunk-RPDDVKI7.js";
295
+ } from "./chunk-4JLWXDYA.js";
294
296
  import {
295
297
  DEFAULT_REDACTION_RULES,
296
298
  REDACTION_VERSION,
@@ -299,6 +301,7 @@ import {
299
301
  } from "./chunk-GGE4NNQT.js";
300
302
  import {
301
303
  DEFAULT_TRACE_ANALYST_BUDGETS,
304
+ INPUT_VALUE,
302
305
  LLM_CACHED_TOKENS,
303
306
  LLM_CACHED_TOKEN_ATTR_KEYS,
304
307
  LLM_COST_ATTR_KEYS,
@@ -310,14 +313,18 @@ import {
310
313
  LLM_OUTPUT_TOKENS,
311
314
  LLM_OUTPUT_TOKEN_ATTR_KEYS,
312
315
  OPENINFERENCE_SPAN_KIND,
316
+ OUTPUT_VALUE,
313
317
  OtlpFileTraceStore,
314
318
  SPAN_KIND_ATTR_KEYS,
315
319
  SpanNotFoundError,
320
+ TOOL_ARGS_CAPTURED,
321
+ TOOL_LATENCY_MS,
316
322
  TOOL_NAME,
317
323
  TOOL_NAME_ATTR_KEYS,
318
324
  TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
319
325
  TraceFileMissingError,
320
326
  TraceNotFoundError,
327
+ applyToolSpanOtlpAttributes,
321
328
  asNumber,
322
329
  asString,
323
330
  buildTraceAnalystTools,
@@ -330,7 +337,7 @@ import {
330
337
  stringField,
331
338
  traceAnalystFunctionGroup,
332
339
  traceSpanKindToOpenInferenceKind
333
- } from "./chunk-LNQEP766.js";
340
+ } from "./chunk-S2F4J57L.js";
334
341
  import {
335
342
  RunIntegrityError,
336
343
  assertRunCaptured,
@@ -373,15 +380,39 @@ import {
373
380
  import {
374
381
  LlmCallError,
375
382
  LlmClient,
383
+ LlmResponseError,
376
384
  LlmRouteAssertionError,
377
385
  assertLlmRoute,
378
386
  backoffMs,
379
387
  callLlm,
380
388
  callLlmJson,
389
+ costReceiptFromLlm,
390
+ costReceiptFromLlmError,
381
391
  isTransientLlmError,
392
+ maximumChargeForLlmRequest,
382
393
  probeLlm,
383
394
  stripFencedJson
384
- } from "./chunk-GY4SYVPJ.js";
395
+ } from "./chunk-NJC7U437.js";
396
+ import {
397
+ CostAccountingIncompleteError,
398
+ CostCallConflictError,
399
+ CostCeilingReachedError,
400
+ CostLedger,
401
+ CostLedgerPersistenceError,
402
+ CostReceiptCaptureError,
403
+ CostReservationExceededError,
404
+ costForUsage,
405
+ modelPriceKey
406
+ } from "./chunk-VCTY3W6J.js";
407
+ import {
408
+ MODEL_PRICING,
409
+ MetricsCollector,
410
+ TokenCounter,
411
+ estimateCost,
412
+ estimateTokens,
413
+ isModelPriced,
414
+ resolveModelPricing
415
+ } from "./chunk-VI2UW6B6.js";
385
416
  import {
386
417
  FileSystemRawProviderSink,
387
418
  InMemoryRawProviderSink,
@@ -707,6 +738,12 @@ function errMessage(err) {
707
738
  async function executeScenario(tc, scenario, config) {
708
739
  const startTime = Date.now();
709
740
  const model = config.model ?? "gpt-4o";
741
+ const costLedger = config.costLedger ?? new CostLedger();
742
+ const costTags = {
743
+ ...config.costTags,
744
+ scenarioId: scenario.id,
745
+ executionId: globalThis.crypto.randomUUID()
746
+ };
710
747
  const systemPrompt = [config.systemPrompt, scenario.systemPromptAppend ?? ""].filter(Boolean).join("\n\n");
711
748
  const messages = [{ role: "system", content: systemPrompt }];
712
749
  const turns = [];
@@ -718,12 +755,25 @@ async function executeScenario(tc, scenario, config) {
718
755
  const turn = scenario.turns[i];
719
756
  const turnStart = Date.now();
720
757
  messages.push({ role: "user", content: turn.user });
721
- const resp = await tc.chat({
758
+ const request = {
722
759
  model,
723
760
  messages,
724
761
  temperature: 0.4,
725
762
  maxTokens: 3e3
763
+ };
764
+ const paid = await costLedger.runPaidCall({
765
+ channel: "agent",
766
+ phase: config.costPhase ?? "benchmark.agent",
767
+ actor: "scenario-agent",
768
+ model,
769
+ maximumCharge: maximumChargeForTCloudRequest(request, config.tcloudMaximumAttempts),
770
+ tags: costTags,
771
+ signal: config.signal,
772
+ execute: () => tc.chat(request),
773
+ receipt: (response) => costReceiptFromTCloud(response, model)
726
774
  });
775
+ if (!paid.succeeded) throw paid.error;
776
+ const resp = paid.value;
727
777
  const message = resp.choices?.[0]?.message;
728
778
  const rawContent = message?.content;
729
779
  if (message === void 0 || message === null || typeof rawContent !== "string") {
@@ -809,7 +859,16 @@ async function executeScenario(tc, scenario, config) {
809
859
  };
810
860
  }
811
861
  });
812
- const judgeInput = { scenario, turns, artifacts };
862
+ const judgeInput = {
863
+ scenario,
864
+ turns,
865
+ artifacts,
866
+ costLedger,
867
+ costPhase: config.costPhase ?? "benchmark.judge",
868
+ costTags,
869
+ signal: config.signal,
870
+ tcloudMaximumAttempts: config.tcloudMaximumAttempts
871
+ };
813
872
  const judgeResults = [];
814
873
  let failedJudges = 0;
815
874
  const judgeFailures = [];
@@ -878,7 +937,8 @@ async function executeScenario(tc, scenario, config) {
878
937
  judgeErrors: errorScores.length + failedJudges,
879
938
  overallScore,
880
939
  totalDurationMs: Date.now() - startTime,
881
- artifacts
940
+ artifacts,
941
+ cost: costLedger.summary({ tags: costTags })
882
942
  };
883
943
  if (judgeFailures.length > 0) result.judgeFailures = judgeFailures;
884
944
  return result;
@@ -895,6 +955,8 @@ var BenchmarkRunner = class {
895
955
  async run(scenarios) {
896
956
  const toRun = scenarios ?? this.config.scenarios;
897
957
  const passThreshold = this.config.passThreshold ?? 6;
958
+ const costLedger = this.config.costLedger ?? new CostLedger();
959
+ const costTags = { benchmarkRunId: globalThis.crypto.randomUUID() };
898
960
  console.log("=".repeat(70));
899
961
  console.log(" AGENT EVAL \u2014 BENCHMARK");
900
962
  console.log(" Multi-turn scenarios x Multi-judge panel");
@@ -912,7 +974,10 @@ var BenchmarkRunner = class {
912
974
  const result = await executeScenario(this.tc, scenario, {
913
975
  systemPrompt: this.config.systemPrompt,
914
976
  model: this.config.model,
915
- judges: this.config.judges
977
+ judges: this.config.judges,
978
+ costLedger,
979
+ costTags,
980
+ tcloudMaximumAttempts: this.config.tcloudMaximumAttempts
916
981
  });
917
982
  results.push(result);
918
983
  for (const turn of result.turns) {
@@ -1012,6 +1077,7 @@ var BenchmarkRunner = class {
1012
1077
  promptVersion: this.config.promptVersion ?? "v1",
1013
1078
  scenarioCount: toRun.length,
1014
1079
  results,
1080
+ cost: costLedger.summary({ tags: costTags }),
1015
1081
  summary: { overallAvg, byPersona, byDimension, weakest, strongest }
1016
1082
  };
1017
1083
  }
@@ -1618,11 +1684,15 @@ var AgentDriver = class {
1618
1684
  client;
1619
1685
  driverModel;
1620
1686
  productContext;
1687
+ costLedger;
1688
+ tcloudMaximumAttempts;
1621
1689
  constructor(tc, config) {
1622
1690
  this.tc = tc;
1623
1691
  this.client = config.client;
1624
1692
  this.driverModel = config.driverModel ?? "claude-sonnet-4-6";
1625
1693
  this.productContext = config.productContext ?? "";
1694
+ this.costLedger = config.costLedger ?? new CostLedger();
1695
+ this.tcloudMaximumAttempts = config.tcloudMaximumAttempts;
1626
1696
  }
1627
1697
  /**
1628
1698
  * Run a persona through the product.
@@ -1631,6 +1701,7 @@ var AgentDriver = class {
1631
1701
  * quality curve, and convergence curve.
1632
1702
  */
1633
1703
  async run(persona) {
1704
+ const costTags = { driverRunId: globalThis.crypto.randomUUID() };
1634
1705
  const email = `eval-driver-${Date.now()}@test.agent-eval.local`;
1635
1706
  await this.client.signup(`Driver ${persona.role}`, email, "eval-driver-pass");
1636
1707
  await this.client.login(email, "eval-driver-pass");
@@ -1645,7 +1716,12 @@ var AgentDriver = class {
1645
1716
  let criteriaMetAtTurn = null;
1646
1717
  for (let turn = 1; turn <= persona.maxTurns; turn++) {
1647
1718
  const state = await metrics.getState();
1648
- const userMessage = await this.decideNextMessage(persona, state, conversationHistory);
1719
+ const userMessage = await this.decideNextMessage(
1720
+ persona,
1721
+ state,
1722
+ conversationHistory,
1723
+ costTags
1724
+ );
1649
1725
  if (userMessage === "DONE") {
1650
1726
  completed = true;
1651
1727
  turnsToCompletion = turn - 1;
@@ -1693,18 +1769,21 @@ var AgentDriver = class {
1693
1769
  metrics: turnMetrics,
1694
1770
  finalState,
1695
1771
  convergenceCurve: convergence.getCurve(),
1696
- totalCostUsd: 0,
1772
+ totalCostUsd: this.costLedger.summary({ tags: costTags }).totalCostUsd,
1697
1773
  finalQualityScore: null
1698
1774
  };
1699
1775
  }
1700
1776
  /** Use the driver LLM to decide what the "user" says next */
1701
- async decideNextMessage(persona, state, history) {
1777
+ async decideNextMessage(persona, state, history, costTags) {
1702
1778
  return decideNextUserTurn(this.tc, {
1703
1779
  persona,
1704
1780
  state,
1705
1781
  history,
1706
1782
  productContext: this.productContext,
1707
- model: this.driverModel
1783
+ model: this.driverModel,
1784
+ costLedger: this.costLedger,
1785
+ costTags,
1786
+ tcloudMaximumAttempts: this.tcloudMaximumAttempts
1708
1787
  });
1709
1788
  }
1710
1789
  /** Handle pending approvals based on persona feedback patterns */
@@ -1813,7 +1892,7 @@ async function decideNextUserTurn(tc, opts) {
1813
1892
  const { persona, state, history, productContext = "", model = "claude-sonnet-4-6" } = opts;
1814
1893
  const lastResponse = history.length > 0 ? history[history.length - 1].content.slice(0, 2e3) : "(no conversation yet \u2014 this is the first message)";
1815
1894
  const recentHistory = history.slice(-6).map((h) => `${h.role}: ${h.content.slice(0, 500)}`).join("\n\n");
1816
- const resp = await tc.chat({
1895
+ const request = {
1817
1896
  model,
1818
1897
  messages: [
1819
1898
  { role: "system", content: buildDriverSystemPrompt(persona, state, productContext) },
@@ -1828,7 +1907,19 @@ ${lastResponse}` : "No conversation yet. Send your opening message \u2014 in cha
1828
1907
  ],
1829
1908
  temperature: 0.5,
1830
1909
  maxTokens: 700
1910
+ };
1911
+ const paid = await (opts.costLedger ?? new CostLedger()).runPaidCall({
1912
+ channel: "driver",
1913
+ phase: "driver-turn",
1914
+ actor: "decideNextUserTurn",
1915
+ model,
1916
+ tags: opts.costTags,
1917
+ maximumCharge: maximumChargeForTCloudRequest(request, opts.tcloudMaximumAttempts),
1918
+ execute: () => tc.chat(request),
1919
+ receipt: (response) => costReceiptFromTCloud(response, model)
1831
1920
  });
1921
+ if (!paid.succeeded) throw paid.error;
1922
+ const resp = paid.value;
1832
1923
  const content = resp.choices?.[0]?.message?.content ?? "";
1833
1924
  return content.trim();
1834
1925
  }
@@ -3791,7 +3882,6 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
3791
3882
  const dimAcc = {};
3792
3883
  for (const d of dimensionKeys) dimAcc[d] = [];
3793
3884
  let rationale = "";
3794
- let costUsd = 0;
3795
3885
  const seenCount = /* @__PURE__ */ new Map();
3796
3886
  const keyFor = (model) => {
3797
3887
  const n = (seenCount.get(model) ?? 0) + 1;
@@ -3799,7 +3889,6 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
3799
3889
  return n === 1 ? model : `${model}#${n}`;
3800
3890
  };
3801
3891
  for (const v of verdicts) {
3802
- costUsd += v.costUsd ?? 0;
3803
3892
  const key = keyFor(v.model);
3804
3893
  if (!v.perDimension) {
3805
3894
  failedJudges.push(key);
@@ -3841,7 +3930,6 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
3841
3930
  perJudge,
3842
3931
  maxDisagreement,
3843
3932
  failedJudges,
3844
- costUsd,
3845
3933
  rationale: rationale || "llm-judge",
3846
3934
  verdicts: [...verdicts]
3847
3935
  };
@@ -3921,37 +4009,92 @@ function ensembleJudge(opts) {
3921
4009
  if (opts.crossFamily !== false) {
3922
4010
  assertCrossFamily(opts.models);
3923
4011
  }
3924
- const scoreOne = async (model, input) => {
3925
- if (opts.retry) {
3926
- const outcome = await withJudgeRetry((m) => opts.scoreWith(m, input), {
3927
- ...opts.retry,
3928
- models: [model]
3929
- });
3930
- if (!outcome.succeeded || outcome.value === null) {
3931
- return {
4012
+ const declaredJudgeVersion = opts.judgeVersion?.trim();
4013
+ if (opts.judgeVersion !== void 0 && !declaredJudgeVersion) {
4014
+ throw new Error(`ensembleJudge '${opts.name}': judgeVersion must be non-empty when provided`);
4015
+ }
4016
+ const judgeVersion = declaredJudgeVersion ?? contentHash({
4017
+ kind: "ensembleJudge",
4018
+ models: opts.models,
4019
+ dimensions: opts.dimensions,
4020
+ weights: opts.weights ?? null,
4021
+ crossFamily: opts.crossFamily ?? true,
4022
+ maximumCharge: typeof opts.maximumCharge === "function" ? opts.maximumCharge.toString() : opts.maximumCharge ?? null,
4023
+ retry: opts.retry ? {
4024
+ maxAttempts: opts.retry.maxAttempts ?? null,
4025
+ timeoutMs: opts.retry.timeoutMs ?? null,
4026
+ models: opts.retry.models ?? null,
4027
+ backoffMs: opts.retry.backoffMs?.toString() ?? null,
4028
+ isRetryable: opts.retry.isRetryable?.toString() ?? null
4029
+ } : null,
4030
+ scoreWith: opts.scoreWith.toString()
4031
+ });
4032
+ const directCostLedger = opts.costLedger ?? new CostLedger();
4033
+ const scoreOne = async (args) => {
4034
+ const outcome = await withJudgeRetry(
4035
+ async (model, retrySignal) => {
4036
+ const paid = await args.costLedger.runPaidCall({
4037
+ channel: "judge",
4038
+ phase: args.costPhase,
4039
+ actor: `${opts.name}.${model}`,
3932
4040
  model,
3933
- perDimension: null,
3934
- rationale: outcome.error?.message ?? "judge failed after retries"
3935
- };
3936
- }
3937
- return outcome.value;
3938
- }
3939
- try {
3940
- return await opts.scoreWith(model, input);
3941
- } catch (err) {
4041
+ maximumCharge: typeof opts.maximumCharge === "function" ? opts.maximumCharge(model) : opts.maximumCharge,
4042
+ tags: args.costTags,
4043
+ signal: AbortSignal.any([args.signal, retrySignal]),
4044
+ execute: (signal) => opts.scoreWith(model, { artifact: args.artifact, scenario: args.scenario, signal }),
4045
+ receipt: (verdict) => {
4046
+ const cachedTokens = verdict.usage?.cachedPromptTokens ?? 0;
4047
+ const usageUnknown = !verdict.usage || verdict.usage.captured === false;
4048
+ return {
4049
+ model: verdict.model,
4050
+ inputTokens: Math.max(0, (verdict.usage?.promptTokens ?? 0) - cachedTokens),
4051
+ outputTokens: verdict.usage?.completionTokens ?? 0,
4052
+ cachedTokens: cachedTokens > 0 ? cachedTokens : void 0,
4053
+ usageUnknown,
4054
+ ...verdict.costUsd === void 0 ? {} : { actualCostUsd: verdict.costUsd }
4055
+ };
4056
+ },
4057
+ receiptFromError: (error) => opts.receiptFromError?.(error, model)
4058
+ });
4059
+ if (!paid.succeeded) throw paid.error;
4060
+ return paid.value;
4061
+ },
4062
+ opts.retry ? { ...opts.retry, models: [args.model] } : { maxAttempts: 1, models: [args.model], isRetryable: () => false }
4063
+ );
4064
+ if (!outcome.succeeded || outcome.value === null) {
3942
4065
  return {
3943
- model,
4066
+ model: args.model,
3944
4067
  perDimension: null,
3945
- rationale: err instanceof Error ? err.message : String(err)
4068
+ rationale: outcome.error?.message ?? "judge failed"
3946
4069
  };
3947
4070
  }
4071
+ return outcome.value;
3948
4072
  };
3949
4073
  return {
3950
4074
  name: opts.name,
3951
4075
  dimensions: opts.dimensions.map((d) => ({ key: d, description: d })),
3952
- async score({ artifact, scenario }) {
3953
- const input = { artifact, scenario };
3954
- const verdicts = await Promise.all(opts.models.map((model) => scoreOne(model, input)));
4076
+ judgeVersion,
4077
+ async score({
4078
+ artifact,
4079
+ scenario,
4080
+ signal,
4081
+ costLedger,
4082
+ costPhase,
4083
+ costTags
4084
+ }) {
4085
+ const verdicts = await Promise.all(
4086
+ opts.models.map(
4087
+ (model) => scoreOne({
4088
+ model,
4089
+ artifact,
4090
+ scenario,
4091
+ signal,
4092
+ costLedger: costLedger ?? directCostLedger,
4093
+ costPhase: costPhase ?? "judge",
4094
+ costTags
4095
+ })
4096
+ )
4097
+ );
3955
4098
  const agg = aggregateJudgeVerdicts(verdicts, opts.dimensions, opts.weights);
3956
4099
  const score = {
3957
4100
  dimensions: agg.perDimension,
@@ -4567,118 +4710,13 @@ function clampUnit(value) {
4567
4710
  return Math.max(0, Math.min(1, value));
4568
4711
  }
4569
4712
 
4570
- // src/cost-ledger.ts
4571
- function modelPriceKey(model) {
4572
- return isModelPriced(model) ? model : null;
4573
- }
4574
- function costForUsage(model, usage) {
4575
- assertNonNegative(usage.inputTokens, "inputTokens");
4576
- assertNonNegative(usage.outputTokens, "outputTokens");
4577
- if (usage.cachedTokens !== void 0) assertNonNegative(usage.cachedTokens, "cachedTokens");
4578
- const pricing = resolveModelPricing(model);
4579
- if (!pricing) return { costUsd: 0, costUnknown: true };
4580
- const billedInput = usage.inputTokens + (usage.cachedTokens ?? 0);
4581
- return { costUsd: estimateCost(billedInput, usage.outputTokens, model), costUnknown: false };
4582
- }
4583
- var CostLedger = class {
4584
- entries = [];
4585
- completedTasks = 0;
4586
- /**
4587
- * Record one LLM call. The cost is computed from pricing unless
4588
- * `actualCostUsd` is supplied (a finite observed cost from the provider
4589
- * response), in which case `costUnknown` is false regardless of pricing.
4590
- */
4591
- record(input) {
4592
- const { costUsd, costUnknown } = costForUsage(input.model, input.usage);
4593
- const hasActual = typeof input.actualCostUsd === "number" && Number.isFinite(input.actualCostUsd);
4594
- if (hasActual) assertNonNegative(input.actualCostUsd, "actualCostUsd");
4595
- const entry = {
4596
- model: input.model,
4597
- channel: input.channel,
4598
- inputTokens: input.usage.inputTokens,
4599
- outputTokens: input.usage.outputTokens,
4600
- cachedTokens: input.usage.cachedTokens,
4601
- costUsd: hasActual ? input.actualCostUsd : costUsd,
4602
- costUnknown: hasActual ? false : costUnknown,
4603
- actualCostUsd: hasActual ? input.actualCostUsd : void 0,
4604
- tags: input.tags,
4605
- timestamp: input.timestamp ?? Date.now()
4606
- };
4607
- this.entries.push(entry);
4608
- return entry;
4609
- }
4610
- /** Increment the completed-task counter (used for cost-per-completed-task). */
4611
- markCompleted(count = 1) {
4612
- if (!Number.isInteger(count) || count < 0) {
4613
- throw new ValidationError(
4614
- `CostLedger.markCompleted: count must be a non-negative integer, got ${count}`
4615
- );
4616
- }
4617
- this.completedTasks += count;
4618
- }
4619
- list() {
4620
- return [...this.entries];
4621
- }
4622
- summary() {
4623
- const byChannel = /* @__PURE__ */ new Map();
4624
- const unpriced = /* @__PURE__ */ new Set();
4625
- let totalCost = 0;
4626
- let inputTokens = 0;
4627
- let outputTokens = 0;
4628
- let cachedTokens = 0;
4629
- for (const e of this.entries) {
4630
- totalCost += e.costUsd;
4631
- inputTokens += e.inputTokens;
4632
- outputTokens += e.outputTokens;
4633
- cachedTokens += e.cachedTokens ?? 0;
4634
- if (e.costUnknown) unpriced.add(e.model);
4635
- const roll = byChannel.get(e.channel) ?? {
4636
- channel: e.channel,
4637
- calls: 0,
4638
- inputTokens: 0,
4639
- outputTokens: 0,
4640
- cachedTokens: 0,
4641
- costUsd: 0,
4642
- unpricedCalls: 0
4643
- };
4644
- roll.calls += 1;
4645
- roll.inputTokens += e.inputTokens;
4646
- roll.outputTokens += e.outputTokens;
4647
- roll.cachedTokens += e.cachedTokens ?? 0;
4648
- roll.costUsd += e.costUsd;
4649
- if (e.costUnknown) roll.unpricedCalls += 1;
4650
- byChannel.set(e.channel, roll);
4651
- }
4652
- return {
4653
- totalCalls: this.entries.length,
4654
- inputTokens,
4655
- outputTokens,
4656
- cachedTokens,
4657
- totalCostUsd: totalCost,
4658
- byChannel: [...byChannel.values()].sort((a, b) => a.channel.localeCompare(b.channel)),
4659
- unpricedModels: [...unpriced].sort(),
4660
- fullyPriced: unpriced.size === 0
4661
- };
4662
- }
4663
- /** Total spend divided by completed tasks; null when nothing completed. */
4664
- costPerCompletedTask() {
4665
- if (this.completedTasks === 0) return null;
4666
- return this.summary().totalCostUsd / this.completedTasks;
4667
- }
4668
- };
4669
- function assertNonNegative(n, name) {
4670
- if (!Number.isFinite(n) || n < 0) {
4671
- throw new ValidationError(`CostLedger: ${name} must be a non-negative finite number, got ${n}`);
4672
- }
4673
- }
4674
-
4675
4713
  // src/cost-tracker.ts
4676
4714
  var CostTracker = class {
4677
4715
  byScenario = /* @__PURE__ */ new Map();
4678
4716
  record(entry) {
4679
4717
  const full = { timestamp: entry.timestamp ?? Date.now(), ...entry };
4680
- assertNonNegative2(full.inputTokens, "inputTokens");
4681
- assertNonNegative2(full.outputTokens, "outputTokens");
4718
+ assertNonNegative(full.inputTokens, "inputTokens");
4719
+ assertNonNegative(full.outputTokens, "outputTokens");
4682
4720
  let bucket = this.byScenario.get(full.scenarioId);
4683
4721
  if (!bucket) {
4684
4722
  bucket = {
@@ -4757,7 +4795,7 @@ function costFor(entry) {
4757
4795
  }
4758
4796
  return estimateCost(entry.inputTokens, entry.outputTokens, entry.model);
4759
4797
  }
4760
- function assertNonNegative2(n, name) {
4798
+ function assertNonNegative(n, name) {
4761
4799
  if (!Number.isFinite(n) || n < 0) {
4762
4800
  throw new Error(`CostTracker: ${name} must be a non-negative finite number, got ${n}`);
4763
4801
  }
@@ -7983,6 +8021,7 @@ function flowLayer(input) {
7983
8021
  var INTENT_MATCH_JUDGE_VERSION = "intent-match-judge-v1-2026-04-24";
7984
8022
  var DEFAULT_MODEL = "claude-sonnet-4-6";
7985
8023
  var DEFAULT_TIMEOUT = 3e5;
8024
+ var DEFAULT_MAX_TOKENS = 800;
7986
8025
  var DEFAULT_MAX_SOURCE = 25e3;
7987
8026
  var DEFAULT_MAX_PER_FILE = 12e3;
7988
8027
  var DEFAULT_MAX_HTML = 2e4;
@@ -8052,10 +8091,14 @@ async function runIntentMatchJudge(input, options = {}) {
8052
8091
  const opts = {
8053
8092
  model: options.model ?? DEFAULT_MODEL,
8054
8093
  timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,
8094
+ maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,
8055
8095
  maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,
8056
8096
  maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,
8057
8097
  maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,
8058
- llm: options.llm ?? {}
8098
+ llm: options.llm ?? {},
8099
+ costLedger: options.costLedger ?? new CostLedger(),
8100
+ costPhase: options.costPhase ?? "judge.intent-match",
8101
+ signal: options.signal ?? new AbortController().signal
8059
8102
  };
8060
8103
  if (input.sourceFiles.length === 0 && !input.servedHtml) {
8061
8104
  return {
@@ -8069,23 +8112,40 @@ async function runIntentMatchJudge(input, options = {}) {
8069
8112
  error: "no input artifact"
8070
8113
  };
8071
8114
  }
8115
+ let receipt;
8072
8116
  try {
8073
- const { value, result } = await callLlmJson(
8074
- {
8075
- model: opts.model,
8076
- messages: [
8077
- {
8078
- role: "system",
8079
- content: "You are a holistic code reviewer answering one question: did the agent build the right app for the user. Return strict JSON. No prose outside."
8080
- },
8081
- { role: "user", content: buildPrompt(input, opts) }
8082
- ],
8083
- jsonSchema: { name: "intent_match_judge", schema: INTENT_SCHEMA },
8084
- temperature: 0,
8085
- timeoutMs: opts.timeoutMs
8086
- },
8087
- opts.llm
8088
- );
8117
+ const request = {
8118
+ model: opts.model,
8119
+ messages: [
8120
+ {
8121
+ role: "system",
8122
+ content: "You are a holistic code reviewer answering one question: did the agent build the right app for the user. Return strict JSON. No prose outside."
8123
+ },
8124
+ { role: "user", content: buildPrompt(input, opts) }
8125
+ ],
8126
+ jsonSchema: { name: "intent_match_judge", schema: INTENT_SCHEMA },
8127
+ temperature: 0,
8128
+ maxTokens: opts.maxTokens,
8129
+ timeoutMs: opts.timeoutMs
8130
+ };
8131
+ const paid = await opts.costLedger.runPaidCall({
8132
+ channel: "judge",
8133
+ phase: opts.costPhase,
8134
+ actor: "intent-match",
8135
+ model: opts.model,
8136
+ maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
8137
+ signal: opts.signal,
8138
+ execute: (signal, callId) => callLlmJson(request, {
8139
+ ...opts.llm,
8140
+ signal,
8141
+ idempotencyKey: callId
8142
+ }),
8143
+ receipt: ({ result }) => costReceiptFromLlm(result),
8144
+ receiptFromError: costReceiptFromLlmError
8145
+ });
8146
+ receipt = paid.receipt;
8147
+ if (!paid.succeeded) throw paid.error;
8148
+ const { value } = paid.value;
8089
8149
  const score = Math.max(0, Math.min(1, Number(value?.score ?? 0)));
8090
8150
  return {
8091
8151
  kind: "intent-match",
@@ -8093,7 +8153,7 @@ async function runIntentMatchJudge(input, options = {}) {
8093
8153
  score: Number(score.toFixed(3)),
8094
8154
  evidence: String(value?.evidence ?? "").slice(0, 400),
8095
8155
  durationMs: Date.now() - start,
8096
- costUsd: result.costUsd ?? null,
8156
+ costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,
8097
8157
  available: true
8098
8158
  };
8099
8159
  } catch (err) {
@@ -8103,7 +8163,7 @@ async function runIntentMatchJudge(input, options = {}) {
8103
8163
  score: 0,
8104
8164
  evidence: "",
8105
8165
  durationMs: Date.now() - start,
8106
- costUsd: null,
8166
+ costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,
8107
8167
  available: false,
8108
8168
  error: err instanceof Error ? err.message : String(err)
8109
8169
  };
@@ -9143,23 +9203,42 @@ function createDefaultReviewer(options) {
9143
9203
  };
9144
9204
  const promptBuilder = options.promptBuilder ?? buildReviewerPrompt;
9145
9205
  const timeoutMs = options.timeoutMs ?? 3e5;
9206
+ const maxTokens = options.maxTokens ?? 4e3;
9207
+ const costLedger = options.costLedger ?? new CostLedger();
9146
9208
  return async (input) => {
9147
9209
  const start = Date.now();
9148
9210
  const { system, user } = promptBuilder(input);
9211
+ let receipt;
9149
9212
  try {
9150
- const { value, result } = await callLlmJson(
9151
- {
9152
- model: options.model,
9153
- messages: [
9154
- { role: "system", content: system },
9155
- { role: "user", content: user }
9156
- ],
9157
- jsonSchema: { name: "reviewer_output", schema: REVIEWER_SCHEMA },
9158
- temperature: 0,
9159
- timeoutMs
9160
- },
9161
- options.llm ?? {}
9162
- );
9213
+ const request = {
9214
+ model: options.model,
9215
+ messages: [
9216
+ { role: "system", content: system },
9217
+ { role: "user", content: user }
9218
+ ],
9219
+ jsonSchema: { name: "reviewer_output", schema: REVIEWER_SCHEMA },
9220
+ temperature: 0,
9221
+ maxTokens,
9222
+ timeoutMs
9223
+ };
9224
+ const paid = await costLedger.runPaidCall({
9225
+ channel: "analyst",
9226
+ phase: options.costPhase ?? "review",
9227
+ actor: "default-reviewer",
9228
+ model: options.model,
9229
+ maximumCharge: maximumChargeForLlmRequest(request, options.llm),
9230
+ signal: options.signal,
9231
+ execute: (signal, callId) => callLlmJson(request, {
9232
+ ...options.llm,
9233
+ signal,
9234
+ idempotencyKey: callId
9235
+ }),
9236
+ receipt: ({ result }) => costReceiptFromLlm(result),
9237
+ receiptFromError: costReceiptFromLlmError
9238
+ });
9239
+ receipt = paid.receipt;
9240
+ if (!paid.succeeded) throw paid.error;
9241
+ const { value } = paid.value;
9163
9242
  return {
9164
9243
  shot: input.shot,
9165
9244
  observations: String(value.observations ?? softFail2.observations),
@@ -9167,7 +9246,7 @@ function createDefaultReviewer(options) {
9167
9246
  nextShotInstruction: String(value.nextShotInstruction ?? softFail2.nextShotInstruction),
9168
9247
  shouldContinue: Boolean(value.shouldContinue),
9169
9248
  confidence: Math.max(0, Math.min(1, Number(value.confidence ?? softFail2.confidence))),
9170
- costUsd: result.costUsd ?? null,
9249
+ costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,
9171
9250
  durationMs: Date.now() - start,
9172
9251
  available: true
9173
9252
  };
@@ -9179,7 +9258,7 @@ function createDefaultReviewer(options) {
9179
9258
  nextShotInstruction: softFail2.nextShotInstruction,
9180
9259
  shouldContinue: softFail2.shouldContinue,
9181
9260
  confidence: softFail2.confidence,
9182
- costUsd: null,
9261
+ costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,
9183
9262
  durationMs: Date.now() - start,
9184
9263
  available: false,
9185
9264
  error: err instanceof Error ? err.message : String(err)
@@ -10259,18 +10338,23 @@ async function runDistillation(opts) {
10259
10338
  input: scenario.input,
10260
10339
  scenarioId: scenario.id
10261
10340
  });
10262
- const response = await chat.chat(
10263
- {
10264
- model: opts.studentModel,
10265
- messages: prompt,
10266
- jsonMode: true,
10267
- temperature: studentTemperature,
10268
- maxTokens: studentMaxTokens
10269
- },
10270
- { signal: ctx.signal }
10271
- );
10272
- reportUsage(ctx.cost, response);
10273
- return parse(response.content, scenario.id);
10341
+ const request = {
10342
+ model: opts.studentModel,
10343
+ messages: prompt,
10344
+ jsonMode: true,
10345
+ temperature: studentTemperature,
10346
+ maxTokens: studentMaxTokens
10347
+ };
10348
+ const paid = await ctx.cost.runPaidCall({
10349
+ actor: "distillation-student",
10350
+ model: opts.studentModel,
10351
+ maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maxRetries: chat.maximumAttempts }),
10352
+ execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),
10353
+ receipt: costReceiptFromLlm,
10354
+ receiptFromError: costReceiptFromLlmError
10355
+ });
10356
+ if (!paid.succeeded) throw paid.error;
10357
+ return parse(paid.value.content, scenario.id);
10274
10358
  }
10275
10359
  });
10276
10360
  const winnerPrompt = typeof loop.winnerSurface === "string" ? loop.winnerSurface : opts.baselinePrompt;
@@ -10282,14 +10366,6 @@ async function runDistillation(opts) {
10282
10366
  holdoutAgreement: { baseline, winner, delta: winner - baseline }
10283
10367
  };
10284
10368
  }
10285
- function reportUsage(cost, response) {
10286
- if (typeof response.costUsd === "number") cost.observe(response.costUsd, "distillation-student");
10287
- cost.observeTokens({
10288
- input: response.usage.promptTokens,
10289
- output: response.usage.completionTokens,
10290
- cached: response.usage.cachedPromptTokens
10291
- });
10292
- }
10293
10369
  var DEFAULT_MUTATION_PRIMITIVES2 = [
10294
10370
  "Add an explicit output-schema instruction so the model emits exactly the gold label fields as JSON.",
10295
10371
  "Add a one-line decision rule for each verdict field the student keeps getting wrong.",
@@ -11452,7 +11528,13 @@ export {
11452
11528
  CaptureIntegrityError,
11453
11529
  ConfigError,
11454
11530
  ConvergenceTracker,
11531
+ CostAccountingIncompleteError,
11532
+ CostCallConflictError,
11533
+ CostCeilingReachedError,
11455
11534
  CostLedger,
11535
+ CostLedgerPersistenceError,
11536
+ CostReceiptCaptureError,
11537
+ CostReservationExceededError,
11456
11538
  CostTracker,
11457
11539
  CrossFamilyError,
11458
11540
  DEFAULT_AGENT_SLOS,
@@ -11487,6 +11569,7 @@ export {
11487
11569
  HoldoutAuditor,
11488
11570
  HoldoutLockedError,
11489
11571
  IMPROVEMENT_KIND_SPEC,
11572
+ INPUT_VALUE,
11490
11573
  INTENT_MATCH_JUDGE_VERSION,
11491
11574
  InMemoryFeedbackTrajectoryStore,
11492
11575
  InMemoryRawProviderSink,
@@ -11509,6 +11592,7 @@ export {
11509
11592
  LLM_OUTPUT_TOKEN_ATTR_KEYS,
11510
11593
  LlmCallError,
11511
11594
  LlmClient,
11595
+ LlmResponseError,
11512
11596
  LlmRouteAssertionError,
11513
11597
  LockedJsonlAppender,
11514
11598
  MODEL_PRICING,
@@ -11521,14 +11605,18 @@ export {
11521
11605
  NotFoundError,
11522
11606
  OPENINFERENCE_SPAN_KIND,
11523
11607
  OTEL_AGENT_EVAL_SCOPE,
11608
+ OUTPUT_VALUE,
11524
11609
  OtlpFileTraceStore,
11525
11610
  POLICY_EDIT_AXES,
11611
+ POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
11526
11612
  POLICY_EDIT_TARGET_SURFACES,
11527
11613
  PairwiseSteeringOptimizer,
11528
11614
  PolicyEditValidationError,
11529
11615
  ProductClient,
11530
11616
  PromptRegistry,
11531
11617
  REDACTION_VERSION,
11618
+ REFERENCE_EQUIVALENCE_INPUT_LIMITS,
11619
+ REFERENCE_EQUIVALENCE_JUDGE_VERSION,
11532
11620
  RESEARCH_REPORT_HARD_PAIR_FLOOR,
11533
11621
  ReplayCache,
11534
11622
  ReplayCacheMissError,
@@ -11546,6 +11634,8 @@ export {
11546
11634
  SkillUsageAnalyst,
11547
11635
  SpanNotFoundError,
11548
11636
  SubprocessSandboxDriver,
11637
+ TOOL_ARGS_CAPTURED,
11638
+ TOOL_LATENCY_MS,
11549
11639
  TOOL_NAME,
11550
11640
  TOOL_NAME_ATTR_KEYS,
11551
11641
  TRACE_ANALYST_ACTOR_DESCRIPTION,
@@ -11582,6 +11672,7 @@ export {
11582
11672
  analyzeTraces,
11583
11673
  appendScorecard,
11584
11674
  applyPolicyEditToSurface,
11675
+ applyToolSpanOtlpAttributes,
11585
11676
  argHash,
11586
11677
  asNumber,
11587
11678
  asString,
@@ -11674,6 +11765,8 @@ export {
11674
11765
  corpusInterRaterAgreement,
11675
11766
  corpusInterRaterAgreementFromJudgeScores,
11676
11767
  costForUsage,
11768
+ costReceiptFromLlm,
11769
+ costReceiptFromLlmError,
11677
11770
  costReport,
11678
11771
  createAnalystAi,
11679
11772
  createAntiSlopJudge,
@@ -11687,6 +11780,7 @@ export {
11687
11780
  createLlmReviewer,
11688
11781
  createOtelExporter,
11689
11782
  createOtelTracingStore,
11783
+ createReferenceEquivalenceJudge,
11690
11784
  createReplayFetch,
11691
11785
  createSandboxPool,
11692
11786
  createSemanticConceptJudge,
@@ -11777,6 +11871,7 @@ export {
11777
11871
  groupBy,
11778
11872
  groupRunsByAgentProfileCell,
11779
11873
  harnessAxisOf,
11874
+ hasCapturedToolArgs,
11780
11875
  hashContent,
11781
11876
  hashJson,
11782
11877
  hashScenarios,
@@ -11832,9 +11927,11 @@ export {
11832
11927
  makeEvalTools,
11833
11928
  makeFinding,
11834
11929
  makePolicyEdit,
11930
+ makePolicyEditCandidateRecord,
11835
11931
  mannWhitneyU,
11836
11932
  matchGoldens,
11837
11933
  matchSpan,
11934
+ maximumChargeForLlmRequest,
11838
11935
  mcnemar,
11839
11936
  mcnemarPower,
11840
11937
  mcnemarRequiredN,
@@ -11950,6 +12047,7 @@ export {
11950
12047
  runProposeReview,
11951
12048
  runProposeReviewAsControlLoop,
11952
12049
  runRecordToProductBenchmarkRecord,
12050
+ runReferenceEquivalenceJudge,
11953
12051
  runReferenceReplay,
11954
12052
  runScore,
11955
12053
  runSelfPlay,
@@ -12009,6 +12107,7 @@ export {
12009
12107
  userQuestionsForKnowledgeGaps,
12010
12108
  validateAgentProfileCell,
12011
12109
  validatePolicyEdit,
12110
+ validatePolicyEditCandidateRecord,
12012
12111
  validateProductBenchmarkManifest,
12013
12112
  validateProductBenchmarkRecord,
12014
12113
  validateProductBenchmarkRun,