@tangle-network/agent-eval 0.116.0 → 0.117.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. package/CHANGELOG.md +38 -0
  2. package/dist/analyst/index.d.ts +18 -11
  3. package/dist/analyst/index.js +10 -7
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyst-CFBc14Wc.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
  6. package/dist/{analyze-runs-0rz_m29H.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
  7. package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
  8. package/dist/belief-state/index.d.ts +6 -6
  9. package/dist/belief-state/index.js +1 -1
  10. package/dist/benchmarks/index.d.ts +11 -8
  11. package/dist/benchmarks/index.js +11 -10
  12. package/dist/builder-eval/index.d.ts +4 -4
  13. package/dist/builder-eval/index.js +1 -1
  14. package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
  15. package/dist/campaign/index.d.ts +54 -30
  16. package/dist/campaign/index.js +18 -13
  17. package/dist/chunk-3YYRZDON.js +45 -0
  18. package/dist/chunk-3YYRZDON.js.map +1 -0
  19. package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
  20. package/dist/{chunk-NBSS5NDZ.js → chunk-CCZIVI3F.js} +54 -115
  21. package/dist/chunk-CCZIVI3F.js.map +1 -0
  22. package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
  23. package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
  24. package/dist/chunk-HHWE3POT.js +94 -0
  25. package/dist/chunk-HHWE3POT.js.map +1 -0
  26. package/dist/{chunk-3274WNK7.js → chunk-HQPHZGL6.js} +687 -44
  27. package/dist/chunk-HQPHZGL6.js.map +1 -0
  28. package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
  29. package/dist/chunk-IDZTTFRR.js.map +1 -0
  30. package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
  31. package/dist/{chunk-GSW3OBHK.js → chunk-JSJZ4PJ6.js} +406 -726
  32. package/dist/chunk-JSJZ4PJ6.js.map +1 -0
  33. package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
  34. package/dist/chunk-LQUTGLOZ.js.map +1 -0
  35. package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
  36. package/dist/chunk-LTVG32KX.js.map +1 -0
  37. package/dist/{chunk-CIUOICJT.js → chunk-MGEHEHSN.js} +62 -15
  38. package/dist/chunk-MGEHEHSN.js.map +1 -0
  39. package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
  40. package/dist/chunk-NJC7U437.js.map +1 -0
  41. package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
  42. package/dist/chunk-ODVOOEWQ.js.map +1 -0
  43. package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
  44. package/dist/chunk-S2F4J57L.js.map +1 -0
  45. package/dist/chunk-VCTY3W6J.js +798 -0
  46. package/dist/chunk-VCTY3W6J.js.map +1 -0
  47. package/dist/chunk-VF3XSYTI.js +545 -0
  48. package/dist/chunk-VF3XSYTI.js.map +1 -0
  49. package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
  50. package/dist/chunk-YZPO4UHR.js.map +1 -0
  51. package/dist/cli.js +4 -2
  52. package/dist/cli.js.map +1 -1
  53. package/dist/{code-agent-session-CdxteG0y.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
  54. package/dist/contract/index.d.ts +43 -29
  55. package/dist/contract/index.js +56 -19
  56. package/dist/contract/index.js.map +1 -1
  57. package/dist/{control-DbcDxouY.d.ts → control-6vuGfmDH.d.ts} +5 -5
  58. package/dist/control.d.ts +6 -6
  59. package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
  60. package/dist/{default-registry-DDfv22MQ.d.ts → default-registry-DaK8b3fv.d.ts} +2 -2
  61. package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
  62. package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
  63. package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
  64. package/dist/fuzz.d.ts +8 -16
  65. package/dist/fuzz.js +72 -42
  66. package/dist/fuzz.js.map +1 -1
  67. package/dist/{gepa-CQelRtuC.d.ts → gepa-eESocoDi.d.ts} +56 -6
  68. package/dist/hosted/index.d.ts +13 -10
  69. package/dist/{index-DbCXJfZ1.d.ts → index-PdX4VnPA.d.ts} +3 -3
  70. package/dist/index.d.ts +102 -57
  71. package/dist/index.js +328 -235
  72. package/dist/index.js.map +1 -1
  73. package/dist/{insight-report-oMVxDTxl.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
  74. package/dist/{integrity-C6PZ73iC.d.ts → integrity-DqlBiLyK.d.ts} +2 -2
  75. package/dist/{kind-factory-DWOvXjR_.d.ts → kind-factory-ClZmO25A.d.ts} +2 -2
  76. package/dist/llm-client-qoDd18Qz.d.ts +289 -0
  77. package/dist/meta-eval/index.d.ts +8 -7
  78. package/dist/meta-eval/index.js +1 -1
  79. package/dist/multishot/index.d.ts +9 -6
  80. package/dist/openapi.json +1 -1
  81. package/dist/pipelines/index.d.ts +16 -6
  82. package/dist/pipelines/index.js +119 -23
  83. package/dist/pipelines/index.js.map +1 -1
  84. package/dist/{policy-edit-Clb2v6Oa.d.ts → policy-edit-wG9uFEFm.d.ts} +13 -266
  85. package/dist/{pre-registration--vU0mMtD.d.ts → pre-registration-BWQhJ3vz.d.ts} +24 -5
  86. package/dist/{provenance-BbVagC68.d.ts → provenance-DpjwyseI.d.ts} +6 -6
  87. package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
  88. package/dist/raw-provider-sink-C46HDghv.d.ts +132 -0
  89. package/dist/{release-report-CamNDe90.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
  90. package/dist/reporting.d.ts +10 -9
  91. package/dist/{researcher-Dwbo_Fxx.d.ts → researcher-C8XyxQsu.d.ts} +8 -8
  92. package/dist/rl.d.ts +18 -15
  93. package/dist/rl.js +2 -2
  94. package/dist/{rubric-predictive-validity-BIdf9h4R.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
  95. package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
  96. package/dist/{run-record-CZmcpWPo.d.ts → run-record-BDH49H2E.d.ts} +1 -1
  97. package/dist/{runtime-trajectory-CC0jx9ql.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
  98. package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
  99. package/dist/{semantic-concept-judge-CKjePUMh.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +24 -6
  100. package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
  101. package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
  102. package/dist/{store-9cAScOcb.d.ts → store-C1YxJDEK.d.ts} +1 -132
  103. package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
  104. package/dist/storyboard/index.d.ts +1 -1
  105. package/dist/{summary-report-DTNgQycC.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
  106. package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
  107. package/dist/traces.d.ts +25 -14
  108. package/dist/traces.js +16 -4
  109. package/dist/{types-Ca_63YSD.d.ts → types-BSw1rOUB.d.ts} +41 -39
  110. package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
  111. package/dist/wire/index.d.ts +28 -19
  112. package/dist/wire/index.js +4 -2
  113. package/docs/distributed-driver.md +1 -1
  114. package/package.json +3 -3
  115. package/dist/chunk-3274WNK7.js.map +0 -1
  116. package/dist/chunk-4D5RVB3W.js.map +0 -1
  117. package/dist/chunk-7GKEAIAD.js +0 -205
  118. package/dist/chunk-7GKEAIAD.js.map +0 -1
  119. package/dist/chunk-CIUOICJT.js.map +0 -1
  120. package/dist/chunk-FAOEFFRT.js.map +0 -1
  121. package/dist/chunk-GSW3OBHK.js.map +0 -1
  122. package/dist/chunk-GY4SYVPJ.js.map +0 -1
  123. package/dist/chunk-LNQEP766.js.map +0 -1
  124. package/dist/chunk-MHNQWM4I.js.map +0 -1
  125. package/dist/chunk-MPHTT5HE.js +0 -74
  126. package/dist/chunk-MPHTT5HE.js.map +0 -1
  127. package/dist/chunk-NBSS5NDZ.js.map +0 -1
  128. package/dist/chunk-NYFUT3B3.js.map +0 -1
  129. package/dist/chunk-TLDB7WRY.js.map +0 -1
  130. package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
  131. /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
  132. /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
  133. /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
  134. /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
  135. /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
package/dist/index.js CHANGED
@@ -9,12 +9,12 @@ import {
9
9
  checkBehavioralCanary,
10
10
  checkCanaries,
11
11
  runBehavioralCanaries
12
- } from "./chunk-J6P6PK2R.js";
12
+ } from "./chunk-FQNLDL4D.js";
13
13
  import {
14
14
  BENCHMARK_SPLIT_SEED,
15
15
  benchmarks_exports,
16
16
  deterministicSplit
17
- } from "./chunk-3LXTCTWL.js";
17
+ } from "./chunk-JSDVRFAP.js";
18
18
  import {
19
19
  DEFAULT_RULES,
20
20
  buildTrajectory,
@@ -23,7 +23,7 @@ import {
23
23
  computeToolUseMetrics,
24
24
  iqr,
25
25
  welchsTTest
26
- } from "./chunk-NYFUT3B3.js";
26
+ } from "./chunk-ODVOOEWQ.js";
27
27
  import {
28
28
  analyzeSeries
29
29
  } from "./chunk-BOD4O7OF.js";
@@ -40,40 +40,45 @@ import {
40
40
  import {
41
41
  CODING_HARNESSES,
42
42
  HARNESS_NATIVE_MODEL,
43
- JudgeParseError,
44
- adversarialJudge,
45
43
  agentProfileHash,
46
44
  agentProfileId,
47
45
  agentProfileModelId,
48
- codeExecutionJudge,
49
- coherenceJudge,
50
46
  comparePairedArms,
51
47
  completionVerdict,
52
- createCustomJudge,
53
- createDomainExpertJudge,
54
48
  createLlmCorrectnessChecker,
55
49
  createTokenRecallChecker,
56
- defaultJudges,
57
50
  expandProfileAxes,
58
51
  extractProducedState,
59
52
  harnessAxisOf,
60
- llmJudge,
61
53
  pairArms,
62
54
  parseCorrectnessResponse,
63
55
  verifyCompletion
64
- } from "./chunk-GSW3OBHK.js";
56
+ } from "./chunk-JSJZ4PJ6.js";
65
57
  import {
66
58
  DEFAULT_MUTATION_PRIMITIVES,
67
59
  DEFAULT_RED_TEAM_CORPUS,
68
60
  Dataset,
69
61
  HoldoutLockedError,
62
+ JudgeParseError,
63
+ REFERENCE_EQUIVALENCE_INPUT_LIMITS,
64
+ REFERENCE_EQUIVALENCE_JUDGE_VERSION,
65
+ adversarialJudge,
70
66
  buildReflectionPrompt,
71
67
  campaignMeanComposite,
68
+ codeExecutionJudge,
69
+ coherenceJudge,
70
+ costReceiptFromTCloud,
71
+ createCustomJudge,
72
+ createDomainExpertJudge,
73
+ createReferenceEquivalenceJudge,
72
74
  crowdingDistance,
75
+ defaultJudges,
73
76
  dominates,
74
77
  gepaProposer,
75
78
  hashScenarios,
76
79
  heldOutGate,
80
+ llmJudge,
81
+ maximumChargeForTCloudRequest,
77
82
  paretoFrontier,
78
83
  paretoFrontierWithCrowding,
79
84
  parseReflectionResponse,
@@ -81,20 +86,12 @@ import {
81
86
  redTeamReport,
82
87
  runCanaries,
83
88
  runImprovementLoop,
89
+ runReferenceEquivalenceJudge,
84
90
  scalarScore,
85
91
  scoreRedTeamOutput,
86
92
  surfaceContentHash,
87
93
  toolNamesForRun
88
- } from "./chunk-3274WNK7.js";
89
- import {
90
- MODEL_PRICING,
91
- MetricsCollector,
92
- TokenCounter,
93
- estimateCost,
94
- estimateTokens,
95
- isModelPriced,
96
- resolveModelPricing
97
- } from "./chunk-VI2UW6B6.js";
94
+ } from "./chunk-HQPHZGL6.js";
98
95
  import {
99
96
  BackendIntegrityError,
100
97
  assertRealBackend,
@@ -104,7 +101,7 @@ import {
104
101
  fileVerdictCache,
105
102
  inMemoryVerdictCache,
106
103
  summarizeBackendIntegrity
107
- } from "./chunk-FAOEFFRT.js";
104
+ } from "./chunk-IDZTTFRR.js";
108
105
  import {
109
106
  DEFAULT_COMPLEXITY_WEIGHTS,
110
107
  FindingsStore,
@@ -114,24 +111,23 @@ import {
114
111
  SKILL_USAGE_ANALYST,
115
112
  SkillUsageAnalyst,
116
113
  createAnalystAi,
117
- createChatClient,
118
114
  createSemanticConceptJudge,
119
115
  defaultIsMaterial,
120
116
  diffFindings,
121
117
  runSemanticConceptJudge
122
- } from "./chunk-NBSS5NDZ.js";
118
+ } from "./chunk-CCZIVI3F.js";
123
119
  import {
124
120
  buildDefaultAnalystRegistry,
125
- computeTraceMetrics
126
- } from "./chunk-7GKEAIAD.js";
121
+ computeTraceMetrics,
122
+ createChatClient
123
+ } from "./chunk-VF3XSYTI.js";
124
+ import "./chunk-HHWE3POT.js";
127
125
  import {
128
- DEFAULT_RUN_SCORE_WEIGHTS,
129
- Mutex,
130
- aggregateRunScore,
131
- clamp01
132
- } from "./chunk-MPHTT5HE.js";
126
+ Mutex
127
+ } from "./chunk-3YYRZDON.js";
133
128
  import {
134
129
  AnalystRegistry,
130
+ DEFAULT_RUN_SCORE_WEIGHTS,
135
131
  DEFAULT_TRACE_ANALYST_KINDS,
136
132
  FAILURE_MODE_KIND_SPEC,
137
133
  IMPROVEMENT_KIND_SPEC,
@@ -142,7 +138,9 @@ import {
142
138
  POLICY_EDIT_TARGET_SURFACES,
143
139
  PolicyEditValidationError,
144
140
  admitPolicyEdit,
141
+ aggregateRunScore,
145
142
  applyPolicyEditToSurface,
143
+ clamp01,
146
144
  computeFindingId,
147
145
  computePolicyEditId,
148
146
  createTraceAnalystKind,
@@ -156,7 +154,7 @@ import {
156
154
  scorePolicyEditReadiness,
157
155
  validatePolicyEdit,
158
156
  validatePolicyEditCandidateRecord
159
- } from "./chunk-CIUOICJT.js";
157
+ } from "./chunk-MGEHEHSN.js";
160
158
  import {
161
159
  allCriticalPassed,
162
160
  controlFailureClassFromVerification,
@@ -187,7 +185,7 @@ import {
187
185
  } from "./chunk-MOXWMGPC.js";
188
186
  import {
189
187
  runEvalCampaign
190
- } from "./chunk-ONM6PEAE.js";
188
+ } from "./chunk-GQCZRZ7L.js";
191
189
  import "./chunk-ARU2PZFM.js";
192
190
  import {
193
191
  evaluateInterimReleaseConfidence,
@@ -270,13 +268,14 @@ import {
270
268
  scoreTraceInsightReadiness,
271
269
  tokenizeDomainWords,
272
270
  traceAnalystOnRunComplete
273
- } from "./chunk-TLDB7WRY.js";
271
+ } from "./chunk-YZPO4UHR.js";
274
272
  import {
275
273
  FAILURE_CLASSES,
276
274
  TRACE_SCHEMA_VERSION,
277
275
  aggregateLlm,
278
276
  argHash,
279
277
  groupBy,
278
+ hasCapturedToolArgs,
280
279
  isJudgeSpan,
281
280
  isLlmSpan,
282
281
  isRetrievalSpan,
@@ -287,13 +286,13 @@ import {
287
286
  runFailureClass,
288
287
  runsForScenario,
289
288
  toolSpans
290
- } from "./chunk-MHNQWM4I.js";
289
+ } from "./chunk-LQUTGLOZ.js";
291
290
  import {
292
291
  TRACE_ANALYST_ACTOR_DESCRIPTION,
293
292
  TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
294
293
  TRACE_ANALYST_SUBAGENT_DESCRIPTION,
295
294
  analyzeTraces
296
- } from "./chunk-RPDDVKI7.js";
295
+ } from "./chunk-4JLWXDYA.js";
297
296
  import {
298
297
  DEFAULT_REDACTION_RULES,
299
298
  REDACTION_VERSION,
@@ -302,6 +301,7 @@ import {
302
301
  } from "./chunk-GGE4NNQT.js";
303
302
  import {
304
303
  DEFAULT_TRACE_ANALYST_BUDGETS,
304
+ INPUT_VALUE,
305
305
  LLM_CACHED_TOKENS,
306
306
  LLM_CACHED_TOKEN_ATTR_KEYS,
307
307
  LLM_COST_ATTR_KEYS,
@@ -313,14 +313,18 @@ import {
313
313
  LLM_OUTPUT_TOKENS,
314
314
  LLM_OUTPUT_TOKEN_ATTR_KEYS,
315
315
  OPENINFERENCE_SPAN_KIND,
316
+ OUTPUT_VALUE,
316
317
  OtlpFileTraceStore,
317
318
  SPAN_KIND_ATTR_KEYS,
318
319
  SpanNotFoundError,
320
+ TOOL_ARGS_CAPTURED,
321
+ TOOL_LATENCY_MS,
319
322
  TOOL_NAME,
320
323
  TOOL_NAME_ATTR_KEYS,
321
324
  TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
322
325
  TraceFileMissingError,
323
326
  TraceNotFoundError,
327
+ applyToolSpanOtlpAttributes,
324
328
  asNumber,
325
329
  asString,
326
330
  buildTraceAnalystTools,
@@ -333,7 +337,7 @@ import {
333
337
  stringField,
334
338
  traceAnalystFunctionGroup,
335
339
  traceSpanKindToOpenInferenceKind
336
- } from "./chunk-LNQEP766.js";
340
+ } from "./chunk-S2F4J57L.js";
337
341
  import {
338
342
  RunIntegrityError,
339
343
  assertRunCaptured,
@@ -376,15 +380,39 @@ import {
376
380
  import {
377
381
  LlmCallError,
378
382
  LlmClient,
383
+ LlmResponseError,
379
384
  LlmRouteAssertionError,
380
385
  assertLlmRoute,
381
386
  backoffMs,
382
387
  callLlm,
383
388
  callLlmJson,
389
+ costReceiptFromLlm,
390
+ costReceiptFromLlmError,
384
391
  isTransientLlmError,
392
+ maximumChargeForLlmRequest,
385
393
  probeLlm,
386
394
  stripFencedJson
387
- } from "./chunk-GY4SYVPJ.js";
395
+ } from "./chunk-NJC7U437.js";
396
+ import {
397
+ CostAccountingIncompleteError,
398
+ CostCallConflictError,
399
+ CostCeilingReachedError,
400
+ CostLedger,
401
+ CostLedgerPersistenceError,
402
+ CostReceiptCaptureError,
403
+ CostReservationExceededError,
404
+ costForUsage,
405
+ modelPriceKey
406
+ } from "./chunk-VCTY3W6J.js";
407
+ import {
408
+ MODEL_PRICING,
409
+ MetricsCollector,
410
+ TokenCounter,
411
+ estimateCost,
412
+ estimateTokens,
413
+ isModelPriced,
414
+ resolveModelPricing
415
+ } from "./chunk-VI2UW6B6.js";
388
416
  import {
389
417
  FileSystemRawProviderSink,
390
418
  InMemoryRawProviderSink,
@@ -710,6 +738,12 @@ function errMessage(err) {
710
738
  async function executeScenario(tc, scenario, config) {
711
739
  const startTime = Date.now();
712
740
  const model = config.model ?? "gpt-4o";
741
+ const costLedger = config.costLedger ?? new CostLedger();
742
+ const costTags = {
743
+ ...config.costTags,
744
+ scenarioId: scenario.id,
745
+ executionId: globalThis.crypto.randomUUID()
746
+ };
713
747
  const systemPrompt = [config.systemPrompt, scenario.systemPromptAppend ?? ""].filter(Boolean).join("\n\n");
714
748
  const messages = [{ role: "system", content: systemPrompt }];
715
749
  const turns = [];
@@ -721,12 +755,25 @@ async function executeScenario(tc, scenario, config) {
721
755
  const turn = scenario.turns[i];
722
756
  const turnStart = Date.now();
723
757
  messages.push({ role: "user", content: turn.user });
724
- const resp = await tc.chat({
758
+ const request = {
725
759
  model,
726
760
  messages,
727
761
  temperature: 0.4,
728
762
  maxTokens: 3e3
763
+ };
764
+ const paid = await costLedger.runPaidCall({
765
+ channel: "agent",
766
+ phase: config.costPhase ?? "benchmark.agent",
767
+ actor: "scenario-agent",
768
+ model,
769
+ maximumCharge: maximumChargeForTCloudRequest(request, config.tcloudMaximumAttempts),
770
+ tags: costTags,
771
+ signal: config.signal,
772
+ execute: () => tc.chat(request),
773
+ receipt: (response) => costReceiptFromTCloud(response, model)
729
774
  });
775
+ if (!paid.succeeded) throw paid.error;
776
+ const resp = paid.value;
730
777
  const message = resp.choices?.[0]?.message;
731
778
  const rawContent = message?.content;
732
779
  if (message === void 0 || message === null || typeof rawContent !== "string") {
@@ -812,7 +859,16 @@ async function executeScenario(tc, scenario, config) {
812
859
  };
813
860
  }
814
861
  });
815
- const judgeInput = { scenario, turns, artifacts };
862
+ const judgeInput = {
863
+ scenario,
864
+ turns,
865
+ artifacts,
866
+ costLedger,
867
+ costPhase: config.costPhase ?? "benchmark.judge",
868
+ costTags,
869
+ signal: config.signal,
870
+ tcloudMaximumAttempts: config.tcloudMaximumAttempts
871
+ };
816
872
  const judgeResults = [];
817
873
  let failedJudges = 0;
818
874
  const judgeFailures = [];
@@ -881,7 +937,8 @@ async function executeScenario(tc, scenario, config) {
881
937
  judgeErrors: errorScores.length + failedJudges,
882
938
  overallScore,
883
939
  totalDurationMs: Date.now() - startTime,
884
- artifacts
940
+ artifacts,
941
+ cost: costLedger.summary({ tags: costTags })
885
942
  };
886
943
  if (judgeFailures.length > 0) result.judgeFailures = judgeFailures;
887
944
  return result;
@@ -898,6 +955,8 @@ var BenchmarkRunner = class {
898
955
  async run(scenarios) {
899
956
  const toRun = scenarios ?? this.config.scenarios;
900
957
  const passThreshold = this.config.passThreshold ?? 6;
958
+ const costLedger = this.config.costLedger ?? new CostLedger();
959
+ const costTags = { benchmarkRunId: globalThis.crypto.randomUUID() };
901
960
  console.log("=".repeat(70));
902
961
  console.log(" AGENT EVAL \u2014 BENCHMARK");
903
962
  console.log(" Multi-turn scenarios x Multi-judge panel");
@@ -915,7 +974,10 @@ var BenchmarkRunner = class {
915
974
  const result = await executeScenario(this.tc, scenario, {
916
975
  systemPrompt: this.config.systemPrompt,
917
976
  model: this.config.model,
918
- judges: this.config.judges
977
+ judges: this.config.judges,
978
+ costLedger,
979
+ costTags,
980
+ tcloudMaximumAttempts: this.config.tcloudMaximumAttempts
919
981
  });
920
982
  results.push(result);
921
983
  for (const turn of result.turns) {
@@ -1015,6 +1077,7 @@ var BenchmarkRunner = class {
1015
1077
  promptVersion: this.config.promptVersion ?? "v1",
1016
1078
  scenarioCount: toRun.length,
1017
1079
  results,
1080
+ cost: costLedger.summary({ tags: costTags }),
1018
1081
  summary: { overallAvg, byPersona, byDimension, weakest, strongest }
1019
1082
  };
1020
1083
  }
@@ -1621,11 +1684,15 @@ var AgentDriver = class {
1621
1684
  client;
1622
1685
  driverModel;
1623
1686
  productContext;
1687
+ costLedger;
1688
+ tcloudMaximumAttempts;
1624
1689
  constructor(tc, config) {
1625
1690
  this.tc = tc;
1626
1691
  this.client = config.client;
1627
1692
  this.driverModel = config.driverModel ?? "claude-sonnet-4-6";
1628
1693
  this.productContext = config.productContext ?? "";
1694
+ this.costLedger = config.costLedger ?? new CostLedger();
1695
+ this.tcloudMaximumAttempts = config.tcloudMaximumAttempts;
1629
1696
  }
1630
1697
  /**
1631
1698
  * Run a persona through the product.
@@ -1634,6 +1701,7 @@ var AgentDriver = class {
1634
1701
  * quality curve, and convergence curve.
1635
1702
  */
1636
1703
  async run(persona) {
1704
+ const costTags = { driverRunId: globalThis.crypto.randomUUID() };
1637
1705
  const email = `eval-driver-${Date.now()}@test.agent-eval.local`;
1638
1706
  await this.client.signup(`Driver ${persona.role}`, email, "eval-driver-pass");
1639
1707
  await this.client.login(email, "eval-driver-pass");
@@ -1648,7 +1716,12 @@ var AgentDriver = class {
1648
1716
  let criteriaMetAtTurn = null;
1649
1717
  for (let turn = 1; turn <= persona.maxTurns; turn++) {
1650
1718
  const state = await metrics.getState();
1651
- const userMessage = await this.decideNextMessage(persona, state, conversationHistory);
1719
+ const userMessage = await this.decideNextMessage(
1720
+ persona,
1721
+ state,
1722
+ conversationHistory,
1723
+ costTags
1724
+ );
1652
1725
  if (userMessage === "DONE") {
1653
1726
  completed = true;
1654
1727
  turnsToCompletion = turn - 1;
@@ -1696,18 +1769,21 @@ var AgentDriver = class {
1696
1769
  metrics: turnMetrics,
1697
1770
  finalState,
1698
1771
  convergenceCurve: convergence.getCurve(),
1699
- totalCostUsd: 0,
1772
+ totalCostUsd: this.costLedger.summary({ tags: costTags }).totalCostUsd,
1700
1773
  finalQualityScore: null
1701
1774
  };
1702
1775
  }
1703
1776
  /** Use the driver LLM to decide what the "user" says next */
1704
- async decideNextMessage(persona, state, history) {
1777
+ async decideNextMessage(persona, state, history, costTags) {
1705
1778
  return decideNextUserTurn(this.tc, {
1706
1779
  persona,
1707
1780
  state,
1708
1781
  history,
1709
1782
  productContext: this.productContext,
1710
- model: this.driverModel
1783
+ model: this.driverModel,
1784
+ costLedger: this.costLedger,
1785
+ costTags,
1786
+ tcloudMaximumAttempts: this.tcloudMaximumAttempts
1711
1787
  });
1712
1788
  }
1713
1789
  /** Handle pending approvals based on persona feedback patterns */
@@ -1816,7 +1892,7 @@ async function decideNextUserTurn(tc, opts) {
1816
1892
  const { persona, state, history, productContext = "", model = "claude-sonnet-4-6" } = opts;
1817
1893
  const lastResponse = history.length > 0 ? history[history.length - 1].content.slice(0, 2e3) : "(no conversation yet \u2014 this is the first message)";
1818
1894
  const recentHistory = history.slice(-6).map((h) => `${h.role}: ${h.content.slice(0, 500)}`).join("\n\n");
1819
- const resp = await tc.chat({
1895
+ const request = {
1820
1896
  model,
1821
1897
  messages: [
1822
1898
  { role: "system", content: buildDriverSystemPrompt(persona, state, productContext) },
@@ -1831,7 +1907,19 @@ ${lastResponse}` : "No conversation yet. Send your opening message \u2014 in cha
1831
1907
  ],
1832
1908
  temperature: 0.5,
1833
1909
  maxTokens: 700
1910
+ };
1911
+ const paid = await (opts.costLedger ?? new CostLedger()).runPaidCall({
1912
+ channel: "driver",
1913
+ phase: "driver-turn",
1914
+ actor: "decideNextUserTurn",
1915
+ model,
1916
+ tags: opts.costTags,
1917
+ maximumCharge: maximumChargeForTCloudRequest(request, opts.tcloudMaximumAttempts),
1918
+ execute: () => tc.chat(request),
1919
+ receipt: (response) => costReceiptFromTCloud(response, model)
1834
1920
  });
1921
+ if (!paid.succeeded) throw paid.error;
1922
+ const resp = paid.value;
1835
1923
  const content = resp.choices?.[0]?.message?.content ?? "";
1836
1924
  return content.trim();
1837
1925
  }
@@ -3794,7 +3882,6 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
3794
3882
  const dimAcc = {};
3795
3883
  for (const d of dimensionKeys) dimAcc[d] = [];
3796
3884
  let rationale = "";
3797
- let costUsd = 0;
3798
3885
  const seenCount = /* @__PURE__ */ new Map();
3799
3886
  const keyFor = (model) => {
3800
3887
  const n = (seenCount.get(model) ?? 0) + 1;
@@ -3802,7 +3889,6 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
3802
3889
  return n === 1 ? model : `${model}#${n}`;
3803
3890
  };
3804
3891
  for (const v of verdicts) {
3805
- costUsd += v.costUsd ?? 0;
3806
3892
  const key = keyFor(v.model);
3807
3893
  if (!v.perDimension) {
3808
3894
  failedJudges.push(key);
@@ -3844,7 +3930,6 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
3844
3930
  perJudge,
3845
3931
  maxDisagreement,
3846
3932
  failedJudges,
3847
- costUsd,
3848
3933
  rationale: rationale || "llm-judge",
3849
3934
  verdicts: [...verdicts]
3850
3935
  };
@@ -3924,37 +4009,92 @@ function ensembleJudge(opts) {
3924
4009
  if (opts.crossFamily !== false) {
3925
4010
  assertCrossFamily(opts.models);
3926
4011
  }
3927
- const scoreOne = async (model, input) => {
3928
- if (opts.retry) {
3929
- const outcome = await withJudgeRetry((m) => opts.scoreWith(m, input), {
3930
- ...opts.retry,
3931
- models: [model]
3932
- });
3933
- if (!outcome.succeeded || outcome.value === null) {
3934
- return {
4012
+ const declaredJudgeVersion = opts.judgeVersion?.trim();
4013
+ if (opts.judgeVersion !== void 0 && !declaredJudgeVersion) {
4014
+ throw new Error(`ensembleJudge '${opts.name}': judgeVersion must be non-empty when provided`);
4015
+ }
4016
+ const judgeVersion = declaredJudgeVersion ?? contentHash({
4017
+ kind: "ensembleJudge",
4018
+ models: opts.models,
4019
+ dimensions: opts.dimensions,
4020
+ weights: opts.weights ?? null,
4021
+ crossFamily: opts.crossFamily ?? true,
4022
+ maximumCharge: typeof opts.maximumCharge === "function" ? opts.maximumCharge.toString() : opts.maximumCharge ?? null,
4023
+ retry: opts.retry ? {
4024
+ maxAttempts: opts.retry.maxAttempts ?? null,
4025
+ timeoutMs: opts.retry.timeoutMs ?? null,
4026
+ models: opts.retry.models ?? null,
4027
+ backoffMs: opts.retry.backoffMs?.toString() ?? null,
4028
+ isRetryable: opts.retry.isRetryable?.toString() ?? null
4029
+ } : null,
4030
+ scoreWith: opts.scoreWith.toString()
4031
+ });
4032
+ const directCostLedger = opts.costLedger ?? new CostLedger();
4033
+ const scoreOne = async (args) => {
4034
+ const outcome = await withJudgeRetry(
4035
+ async (model, retrySignal) => {
4036
+ const paid = await args.costLedger.runPaidCall({
4037
+ channel: "judge",
4038
+ phase: args.costPhase,
4039
+ actor: `${opts.name}.${model}`,
3935
4040
  model,
3936
- perDimension: null,
3937
- rationale: outcome.error?.message ?? "judge failed after retries"
3938
- };
3939
- }
3940
- return outcome.value;
3941
- }
3942
- try {
3943
- return await opts.scoreWith(model, input);
3944
- } catch (err) {
4041
+ maximumCharge: typeof opts.maximumCharge === "function" ? opts.maximumCharge(model) : opts.maximumCharge,
4042
+ tags: args.costTags,
4043
+ signal: AbortSignal.any([args.signal, retrySignal]),
4044
+ execute: (signal) => opts.scoreWith(model, { artifact: args.artifact, scenario: args.scenario, signal }),
4045
+ receipt: (verdict) => {
4046
+ const cachedTokens = verdict.usage?.cachedPromptTokens ?? 0;
4047
+ const usageUnknown = !verdict.usage || verdict.usage.captured === false;
4048
+ return {
4049
+ model: verdict.model,
4050
+ inputTokens: Math.max(0, (verdict.usage?.promptTokens ?? 0) - cachedTokens),
4051
+ outputTokens: verdict.usage?.completionTokens ?? 0,
4052
+ cachedTokens: cachedTokens > 0 ? cachedTokens : void 0,
4053
+ usageUnknown,
4054
+ ...verdict.costUsd === void 0 ? {} : { actualCostUsd: verdict.costUsd }
4055
+ };
4056
+ },
4057
+ receiptFromError: (error) => opts.receiptFromError?.(error, model)
4058
+ });
4059
+ if (!paid.succeeded) throw paid.error;
4060
+ return paid.value;
4061
+ },
4062
+ opts.retry ? { ...opts.retry, models: [args.model] } : { maxAttempts: 1, models: [args.model], isRetryable: () => false }
4063
+ );
4064
+ if (!outcome.succeeded || outcome.value === null) {
3945
4065
  return {
3946
- model,
4066
+ model: args.model,
3947
4067
  perDimension: null,
3948
- rationale: err instanceof Error ? err.message : String(err)
4068
+ rationale: outcome.error?.message ?? "judge failed"
3949
4069
  };
3950
4070
  }
4071
+ return outcome.value;
3951
4072
  };
3952
4073
  return {
3953
4074
  name: opts.name,
3954
4075
  dimensions: opts.dimensions.map((d) => ({ key: d, description: d })),
3955
- async score({ artifact, scenario }) {
3956
- const input = { artifact, scenario };
3957
- const verdicts = await Promise.all(opts.models.map((model) => scoreOne(model, input)));
4076
+ judgeVersion,
4077
+ async score({
4078
+ artifact,
4079
+ scenario,
4080
+ signal,
4081
+ costLedger,
4082
+ costPhase,
4083
+ costTags
4084
+ }) {
4085
+ const verdicts = await Promise.all(
4086
+ opts.models.map(
4087
+ (model) => scoreOne({
4088
+ model,
4089
+ artifact,
4090
+ scenario,
4091
+ signal,
4092
+ costLedger: costLedger ?? directCostLedger,
4093
+ costPhase: costPhase ?? "judge",
4094
+ costTags
4095
+ })
4096
+ )
4097
+ );
3958
4098
  const agg = aggregateJudgeVerdicts(verdicts, opts.dimensions, opts.weights);
3959
4099
  const score = {
3960
4100
  dimensions: agg.perDimension,
@@ -4570,118 +4710,13 @@ function clampUnit(value) {
4570
4710
  return Math.max(0, Math.min(1, value));
4571
4711
  }
4572
4712
 
4573
- // src/cost-ledger.ts
4574
- function modelPriceKey(model) {
4575
- return isModelPriced(model) ? model : null;
4576
- }
4577
- function costForUsage(model, usage) {
4578
- assertNonNegative(usage.inputTokens, "inputTokens");
4579
- assertNonNegative(usage.outputTokens, "outputTokens");
4580
- if (usage.cachedTokens !== void 0) assertNonNegative(usage.cachedTokens, "cachedTokens");
4581
- const pricing = resolveModelPricing(model);
4582
- if (!pricing) return { costUsd: 0, costUnknown: true };
4583
- const billedInput = usage.inputTokens + (usage.cachedTokens ?? 0);
4584
- return { costUsd: estimateCost(billedInput, usage.outputTokens, model), costUnknown: false };
4585
- }
4586
- var CostLedger = class {
4587
- entries = [];
4588
- completedTasks = 0;
4589
- /**
4590
- * Record one LLM call. The cost is computed from pricing unless
4591
- * `actualCostUsd` is supplied (a finite observed cost from the provider
4592
- * response), in which case `costUnknown` is false regardless of pricing.
4593
- */
4594
- record(input) {
4595
- const { costUsd, costUnknown } = costForUsage(input.model, input.usage);
4596
- const hasActual = typeof input.actualCostUsd === "number" && Number.isFinite(input.actualCostUsd);
4597
- if (hasActual) assertNonNegative(input.actualCostUsd, "actualCostUsd");
4598
- const entry = {
4599
- model: input.model,
4600
- channel: input.channel,
4601
- inputTokens: input.usage.inputTokens,
4602
- outputTokens: input.usage.outputTokens,
4603
- cachedTokens: input.usage.cachedTokens,
4604
- costUsd: hasActual ? input.actualCostUsd : costUsd,
4605
- costUnknown: hasActual ? false : costUnknown,
4606
- actualCostUsd: hasActual ? input.actualCostUsd : void 0,
4607
- tags: input.tags,
4608
- timestamp: input.timestamp ?? Date.now()
4609
- };
4610
- this.entries.push(entry);
4611
- return entry;
4612
- }
4613
- /** Increment the completed-task counter (used for cost-per-completed-task). */
4614
- markCompleted(count = 1) {
4615
- if (!Number.isInteger(count) || count < 0) {
4616
- throw new ValidationError(
4617
- `CostLedger.markCompleted: count must be a non-negative integer, got ${count}`
4618
- );
4619
- }
4620
- this.completedTasks += count;
4621
- }
4622
- list() {
4623
- return [...this.entries];
4624
- }
4625
- summary() {
4626
- const byChannel = /* @__PURE__ */ new Map();
4627
- const unpriced = /* @__PURE__ */ new Set();
4628
- let totalCost = 0;
4629
- let inputTokens = 0;
4630
- let outputTokens = 0;
4631
- let cachedTokens = 0;
4632
- for (const e of this.entries) {
4633
- totalCost += e.costUsd;
4634
- inputTokens += e.inputTokens;
4635
- outputTokens += e.outputTokens;
4636
- cachedTokens += e.cachedTokens ?? 0;
4637
- if (e.costUnknown) unpriced.add(e.model);
4638
- const roll = byChannel.get(e.channel) ?? {
4639
- channel: e.channel,
4640
- calls: 0,
4641
- inputTokens: 0,
4642
- outputTokens: 0,
4643
- cachedTokens: 0,
4644
- costUsd: 0,
4645
- unpricedCalls: 0
4646
- };
4647
- roll.calls += 1;
4648
- roll.inputTokens += e.inputTokens;
4649
- roll.outputTokens += e.outputTokens;
4650
- roll.cachedTokens += e.cachedTokens ?? 0;
4651
- roll.costUsd += e.costUsd;
4652
- if (e.costUnknown) roll.unpricedCalls += 1;
4653
- byChannel.set(e.channel, roll);
4654
- }
4655
- return {
4656
- totalCalls: this.entries.length,
4657
- inputTokens,
4658
- outputTokens,
4659
- cachedTokens,
4660
- totalCostUsd: totalCost,
4661
- byChannel: [...byChannel.values()].sort((a, b) => a.channel.localeCompare(b.channel)),
4662
- unpricedModels: [...unpriced].sort(),
4663
- fullyPriced: unpriced.size === 0
4664
- };
4665
- }
4666
- /** Total spend divided by completed tasks; null when nothing completed. */
4667
- costPerCompletedTask() {
4668
- if (this.completedTasks === 0) return null;
4669
- return this.summary().totalCostUsd / this.completedTasks;
4670
- }
4671
- };
4672
- function assertNonNegative(n, name) {
4673
- if (!Number.isFinite(n) || n < 0) {
4674
- throw new ValidationError(`CostLedger: ${name} must be a non-negative finite number, got ${n}`);
4675
- }
4676
- }
4677
-
4678
4713
  // src/cost-tracker.ts
4679
4714
  var CostTracker = class {
4680
4715
  byScenario = /* @__PURE__ */ new Map();
4681
4716
  record(entry) {
4682
4717
  const full = { timestamp: entry.timestamp ?? Date.now(), ...entry };
4683
- assertNonNegative2(full.inputTokens, "inputTokens");
4684
- assertNonNegative2(full.outputTokens, "outputTokens");
4718
+ assertNonNegative(full.inputTokens, "inputTokens");
4719
+ assertNonNegative(full.outputTokens, "outputTokens");
4685
4720
  let bucket = this.byScenario.get(full.scenarioId);
4686
4721
  if (!bucket) {
4687
4722
  bucket = {
@@ -4760,7 +4795,7 @@ function costFor(entry) {
4760
4795
  }
4761
4796
  return estimateCost(entry.inputTokens, entry.outputTokens, entry.model);
4762
4797
  }
4763
- function assertNonNegative2(n, name) {
4798
+ function assertNonNegative(n, name) {
4764
4799
  if (!Number.isFinite(n) || n < 0) {
4765
4800
  throw new Error(`CostTracker: ${name} must be a non-negative finite number, got ${n}`);
4766
4801
  }
@@ -7986,6 +8021,7 @@ function flowLayer(input) {
7986
8021
  var INTENT_MATCH_JUDGE_VERSION = "intent-match-judge-v1-2026-04-24";
7987
8022
  var DEFAULT_MODEL = "claude-sonnet-4-6";
7988
8023
  var DEFAULT_TIMEOUT = 3e5;
8024
+ var DEFAULT_MAX_TOKENS = 800;
7989
8025
  var DEFAULT_MAX_SOURCE = 25e3;
7990
8026
  var DEFAULT_MAX_PER_FILE = 12e3;
7991
8027
  var DEFAULT_MAX_HTML = 2e4;
@@ -8055,10 +8091,14 @@ async function runIntentMatchJudge(input, options = {}) {
8055
8091
  const opts = {
8056
8092
  model: options.model ?? DEFAULT_MODEL,
8057
8093
  timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,
8094
+ maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,
8058
8095
  maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,
8059
8096
  maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,
8060
8097
  maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,
8061
- llm: options.llm ?? {}
8098
+ llm: options.llm ?? {},
8099
+ costLedger: options.costLedger ?? new CostLedger(),
8100
+ costPhase: options.costPhase ?? "judge.intent-match",
8101
+ signal: options.signal ?? new AbortController().signal
8062
8102
  };
8063
8103
  if (input.sourceFiles.length === 0 && !input.servedHtml) {
8064
8104
  return {
@@ -8072,23 +8112,40 @@ async function runIntentMatchJudge(input, options = {}) {
8072
8112
  error: "no input artifact"
8073
8113
  };
8074
8114
  }
8115
+ let receipt;
8075
8116
  try {
8076
- const { value, result } = await callLlmJson(
8077
- {
8078
- model: opts.model,
8079
- messages: [
8080
- {
8081
- role: "system",
8082
- content: "You are a holistic code reviewer answering one question: did the agent build the right app for the user. Return strict JSON. No prose outside."
8083
- },
8084
- { role: "user", content: buildPrompt(input, opts) }
8085
- ],
8086
- jsonSchema: { name: "intent_match_judge", schema: INTENT_SCHEMA },
8087
- temperature: 0,
8088
- timeoutMs: opts.timeoutMs
8089
- },
8090
- opts.llm
8091
- );
8117
+ const request = {
8118
+ model: opts.model,
8119
+ messages: [
8120
+ {
8121
+ role: "system",
8122
+ content: "You are a holistic code reviewer answering one question: did the agent build the right app for the user. Return strict JSON. No prose outside."
8123
+ },
8124
+ { role: "user", content: buildPrompt(input, opts) }
8125
+ ],
8126
+ jsonSchema: { name: "intent_match_judge", schema: INTENT_SCHEMA },
8127
+ temperature: 0,
8128
+ maxTokens: opts.maxTokens,
8129
+ timeoutMs: opts.timeoutMs
8130
+ };
8131
+ const paid = await opts.costLedger.runPaidCall({
8132
+ channel: "judge",
8133
+ phase: opts.costPhase,
8134
+ actor: "intent-match",
8135
+ model: opts.model,
8136
+ maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
8137
+ signal: opts.signal,
8138
+ execute: (signal, callId) => callLlmJson(request, {
8139
+ ...opts.llm,
8140
+ signal,
8141
+ idempotencyKey: callId
8142
+ }),
8143
+ receipt: ({ result }) => costReceiptFromLlm(result),
8144
+ receiptFromError: costReceiptFromLlmError
8145
+ });
8146
+ receipt = paid.receipt;
8147
+ if (!paid.succeeded) throw paid.error;
8148
+ const { value } = paid.value;
8092
8149
  const score = Math.max(0, Math.min(1, Number(value?.score ?? 0)));
8093
8150
  return {
8094
8151
  kind: "intent-match",
@@ -8096,7 +8153,7 @@ async function runIntentMatchJudge(input, options = {}) {
8096
8153
  score: Number(score.toFixed(3)),
8097
8154
  evidence: String(value?.evidence ?? "").slice(0, 400),
8098
8155
  durationMs: Date.now() - start,
8099
- costUsd: result.costUsd ?? null,
8156
+ costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,
8100
8157
  available: true
8101
8158
  };
8102
8159
  } catch (err) {
@@ -8106,7 +8163,7 @@ async function runIntentMatchJudge(input, options = {}) {
8106
8163
  score: 0,
8107
8164
  evidence: "",
8108
8165
  durationMs: Date.now() - start,
8109
- costUsd: null,
8166
+ costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,
8110
8167
  available: false,
8111
8168
  error: err instanceof Error ? err.message : String(err)
8112
8169
  };
@@ -9146,23 +9203,42 @@ function createDefaultReviewer(options) {
9146
9203
  };
9147
9204
  const promptBuilder = options.promptBuilder ?? buildReviewerPrompt;
9148
9205
  const timeoutMs = options.timeoutMs ?? 3e5;
9206
+ const maxTokens = options.maxTokens ?? 4e3;
9207
+ const costLedger = options.costLedger ?? new CostLedger();
9149
9208
  return async (input) => {
9150
9209
  const start = Date.now();
9151
9210
  const { system, user } = promptBuilder(input);
9211
+ let receipt;
9152
9212
  try {
9153
- const { value, result } = await callLlmJson(
9154
- {
9155
- model: options.model,
9156
- messages: [
9157
- { role: "system", content: system },
9158
- { role: "user", content: user }
9159
- ],
9160
- jsonSchema: { name: "reviewer_output", schema: REVIEWER_SCHEMA },
9161
- temperature: 0,
9162
- timeoutMs
9163
- },
9164
- options.llm ?? {}
9165
- );
9213
+ const request = {
9214
+ model: options.model,
9215
+ messages: [
9216
+ { role: "system", content: system },
9217
+ { role: "user", content: user }
9218
+ ],
9219
+ jsonSchema: { name: "reviewer_output", schema: REVIEWER_SCHEMA },
9220
+ temperature: 0,
9221
+ maxTokens,
9222
+ timeoutMs
9223
+ };
9224
+ const paid = await costLedger.runPaidCall({
9225
+ channel: "analyst",
9226
+ phase: options.costPhase ?? "review",
9227
+ actor: "default-reviewer",
9228
+ model: options.model,
9229
+ maximumCharge: maximumChargeForLlmRequest(request, options.llm),
9230
+ signal: options.signal,
9231
+ execute: (signal, callId) => callLlmJson(request, {
9232
+ ...options.llm,
9233
+ signal,
9234
+ idempotencyKey: callId
9235
+ }),
9236
+ receipt: ({ result }) => costReceiptFromLlm(result),
9237
+ receiptFromError: costReceiptFromLlmError
9238
+ });
9239
+ receipt = paid.receipt;
9240
+ if (!paid.succeeded) throw paid.error;
9241
+ const { value } = paid.value;
9166
9242
  return {
9167
9243
  shot: input.shot,
9168
9244
  observations: String(value.observations ?? softFail2.observations),
@@ -9170,7 +9246,7 @@ function createDefaultReviewer(options) {
9170
9246
  nextShotInstruction: String(value.nextShotInstruction ?? softFail2.nextShotInstruction),
9171
9247
  shouldContinue: Boolean(value.shouldContinue),
9172
9248
  confidence: Math.max(0, Math.min(1, Number(value.confidence ?? softFail2.confidence))),
9173
- costUsd: result.costUsd ?? null,
9249
+ costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,
9174
9250
  durationMs: Date.now() - start,
9175
9251
  available: true
9176
9252
  };
@@ -9182,7 +9258,7 @@ function createDefaultReviewer(options) {
9182
9258
  nextShotInstruction: softFail2.nextShotInstruction,
9183
9259
  shouldContinue: softFail2.shouldContinue,
9184
9260
  confidence: softFail2.confidence,
9185
- costUsd: null,
9261
+ costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,
9186
9262
  durationMs: Date.now() - start,
9187
9263
  available: false,
9188
9264
  error: err instanceof Error ? err.message : String(err)
@@ -10262,18 +10338,23 @@ async function runDistillation(opts) {
10262
10338
  input: scenario.input,
10263
10339
  scenarioId: scenario.id
10264
10340
  });
10265
- const response = await chat.chat(
10266
- {
10267
- model: opts.studentModel,
10268
- messages: prompt,
10269
- jsonMode: true,
10270
- temperature: studentTemperature,
10271
- maxTokens: studentMaxTokens
10272
- },
10273
- { signal: ctx.signal }
10274
- );
10275
- reportUsage(ctx.cost, response);
10276
- return parse(response.content, scenario.id);
10341
+ const request = {
10342
+ model: opts.studentModel,
10343
+ messages: prompt,
10344
+ jsonMode: true,
10345
+ temperature: studentTemperature,
10346
+ maxTokens: studentMaxTokens
10347
+ };
10348
+ const paid = await ctx.cost.runPaidCall({
10349
+ actor: "distillation-student",
10350
+ model: opts.studentModel,
10351
+ maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maxRetries: chat.maximumAttempts }),
10352
+ execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),
10353
+ receipt: costReceiptFromLlm,
10354
+ receiptFromError: costReceiptFromLlmError
10355
+ });
10356
+ if (!paid.succeeded) throw paid.error;
10357
+ return parse(paid.value.content, scenario.id);
10277
10358
  }
10278
10359
  });
10279
10360
  const winnerPrompt = typeof loop.winnerSurface === "string" ? loop.winnerSurface : opts.baselinePrompt;
@@ -10285,14 +10366,6 @@ async function runDistillation(opts) {
10285
10366
  holdoutAgreement: { baseline, winner, delta: winner - baseline }
10286
10367
  };
10287
10368
  }
10288
- function reportUsage(cost, response) {
10289
- if (typeof response.costUsd === "number") cost.observe(response.costUsd, "distillation-student");
10290
- cost.observeTokens({
10291
- input: response.usage.promptTokens,
10292
- output: response.usage.completionTokens,
10293
- cached: response.usage.cachedPromptTokens
10294
- });
10295
- }
10296
10369
  var DEFAULT_MUTATION_PRIMITIVES2 = [
10297
10370
  "Add an explicit output-schema instruction so the model emits exactly the gold label fields as JSON.",
10298
10371
  "Add a one-line decision rule for each verdict field the student keeps getting wrong.",
@@ -11455,7 +11528,13 @@ export {
11455
11528
  CaptureIntegrityError,
11456
11529
  ConfigError,
11457
11530
  ConvergenceTracker,
11531
+ CostAccountingIncompleteError,
11532
+ CostCallConflictError,
11533
+ CostCeilingReachedError,
11458
11534
  CostLedger,
11535
+ CostLedgerPersistenceError,
11536
+ CostReceiptCaptureError,
11537
+ CostReservationExceededError,
11459
11538
  CostTracker,
11460
11539
  CrossFamilyError,
11461
11540
  DEFAULT_AGENT_SLOS,
@@ -11490,6 +11569,7 @@ export {
11490
11569
  HoldoutAuditor,
11491
11570
  HoldoutLockedError,
11492
11571
  IMPROVEMENT_KIND_SPEC,
11572
+ INPUT_VALUE,
11493
11573
  INTENT_MATCH_JUDGE_VERSION,
11494
11574
  InMemoryFeedbackTrajectoryStore,
11495
11575
  InMemoryRawProviderSink,
@@ -11512,6 +11592,7 @@ export {
11512
11592
  LLM_OUTPUT_TOKEN_ATTR_KEYS,
11513
11593
  LlmCallError,
11514
11594
  LlmClient,
11595
+ LlmResponseError,
11515
11596
  LlmRouteAssertionError,
11516
11597
  LockedJsonlAppender,
11517
11598
  MODEL_PRICING,
@@ -11524,6 +11605,7 @@ export {
11524
11605
  NotFoundError,
11525
11606
  OPENINFERENCE_SPAN_KIND,
11526
11607
  OTEL_AGENT_EVAL_SCOPE,
11608
+ OUTPUT_VALUE,
11527
11609
  OtlpFileTraceStore,
11528
11610
  POLICY_EDIT_AXES,
11529
11611
  POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
@@ -11533,6 +11615,8 @@ export {
11533
11615
  ProductClient,
11534
11616
  PromptRegistry,
11535
11617
  REDACTION_VERSION,
11618
+ REFERENCE_EQUIVALENCE_INPUT_LIMITS,
11619
+ REFERENCE_EQUIVALENCE_JUDGE_VERSION,
11536
11620
  RESEARCH_REPORT_HARD_PAIR_FLOOR,
11537
11621
  ReplayCache,
11538
11622
  ReplayCacheMissError,
@@ -11550,6 +11634,8 @@ export {
11550
11634
  SkillUsageAnalyst,
11551
11635
  SpanNotFoundError,
11552
11636
  SubprocessSandboxDriver,
11637
+ TOOL_ARGS_CAPTURED,
11638
+ TOOL_LATENCY_MS,
11553
11639
  TOOL_NAME,
11554
11640
  TOOL_NAME_ATTR_KEYS,
11555
11641
  TRACE_ANALYST_ACTOR_DESCRIPTION,
@@ -11586,6 +11672,7 @@ export {
11586
11672
  analyzeTraces,
11587
11673
  appendScorecard,
11588
11674
  applyPolicyEditToSurface,
11675
+ applyToolSpanOtlpAttributes,
11589
11676
  argHash,
11590
11677
  asNumber,
11591
11678
  asString,
@@ -11678,6 +11765,8 @@ export {
11678
11765
  corpusInterRaterAgreement,
11679
11766
  corpusInterRaterAgreementFromJudgeScores,
11680
11767
  costForUsage,
11768
+ costReceiptFromLlm,
11769
+ costReceiptFromLlmError,
11681
11770
  costReport,
11682
11771
  createAnalystAi,
11683
11772
  createAntiSlopJudge,
@@ -11691,6 +11780,7 @@ export {
11691
11780
  createLlmReviewer,
11692
11781
  createOtelExporter,
11693
11782
  createOtelTracingStore,
11783
+ createReferenceEquivalenceJudge,
11694
11784
  createReplayFetch,
11695
11785
  createSandboxPool,
11696
11786
  createSemanticConceptJudge,
@@ -11781,6 +11871,7 @@ export {
11781
11871
  groupBy,
11782
11872
  groupRunsByAgentProfileCell,
11783
11873
  harnessAxisOf,
11874
+ hasCapturedToolArgs,
11784
11875
  hashContent,
11785
11876
  hashJson,
11786
11877
  hashScenarios,
@@ -11840,6 +11931,7 @@ export {
11840
11931
  mannWhitneyU,
11841
11932
  matchGoldens,
11842
11933
  matchSpan,
11934
+ maximumChargeForLlmRequest,
11843
11935
  mcnemar,
11844
11936
  mcnemarPower,
11845
11937
  mcnemarRequiredN,
@@ -11955,6 +12047,7 @@ export {
11955
12047
  runProposeReview,
11956
12048
  runProposeReviewAsControlLoop,
11957
12049
  runRecordToProductBenchmarkRecord,
12050
+ runReferenceEquivalenceJudge,
11958
12051
  runReferenceReplay,
11959
12052
  runScore,
11960
12053
  runSelfPlay,