@tangle-network/agent-eval 0.115.3 → 0.117.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. package/CHANGELOG.md +55 -0
  2. package/dist/analyst/index.d.ts +16 -11
  3. package/dist/analyst/index.js +33 -25
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
  6. package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
  7. package/dist/belief-state/index.d.ts +6 -6
  8. package/dist/belief-state/index.js +1 -1
  9. package/dist/benchmarks/index.d.ts +12 -5
  10. package/dist/benchmarks/index.js +11 -10
  11. package/dist/builder-eval/index.d.ts +4 -4
  12. package/dist/builder-eval/index.js +1 -1
  13. package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
  14. package/dist/campaign/index.d.ts +247 -34
  15. package/dist/campaign/index.js +33 -13
  16. package/dist/chunk-3YYRZDON.js +45 -0
  17. package/dist/chunk-3YYRZDON.js.map +1 -0
  18. package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
  19. package/dist/{chunk-WSBUZMBU.js → chunk-CCZIVI3F.js} +54 -115
  20. package/dist/chunk-CCZIVI3F.js.map +1 -0
  21. package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
  22. package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
  23. package/dist/chunk-HHWE3POT.js +94 -0
  24. package/dist/chunk-HHWE3POT.js.map +1 -0
  25. package/dist/{chunk-ADYLPOSX.js → chunk-HQPHZGL6.js} +1112 -135
  26. package/dist/chunk-HQPHZGL6.js.map +1 -0
  27. package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
  28. package/dist/chunk-IDZTTFRR.js.map +1 -0
  29. package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
  30. package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
  31. package/dist/chunk-LQUTGLOZ.js.map +1 -0
  32. package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
  33. package/dist/chunk-LTVG32KX.js.map +1 -0
  34. package/dist/{chunk-5S5NJ63F.js → chunk-MGEHEHSN.js} +807 -15
  35. package/dist/chunk-MGEHEHSN.js.map +1 -0
  36. package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
  37. package/dist/chunk-NJC7U437.js.map +1 -0
  38. package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
  39. package/dist/chunk-ODVOOEWQ.js.map +1 -0
  40. package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
  41. package/dist/chunk-S2F4J57L.js.map +1 -0
  42. package/dist/chunk-VCTY3W6J.js +798 -0
  43. package/dist/chunk-VCTY3W6J.js.map +1 -0
  44. package/dist/chunk-VF3XSYTI.js +545 -0
  45. package/dist/chunk-VF3XSYTI.js.map +1 -0
  46. package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
  47. package/dist/chunk-YZPO4UHR.js.map +1 -0
  48. package/dist/{chunk-KG4TD7EQ.js → chunk-ZUXV7UWZ.js} +1425 -697
  49. package/dist/chunk-ZUXV7UWZ.js.map +1 -0
  50. package/dist/cli.js +4 -2
  51. package/dist/cli.js.map +1 -1
  52. package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
  53. package/dist/contract/index.d.ts +45 -31
  54. package/dist/contract/index.js +58 -19
  55. package/dist/contract/index.js.map +1 -1
  56. package/dist/{control-CcBiAEnn.d.ts → control-6vuGfmDH.d.ts} +5 -5
  57. package/dist/control.d.ts +6 -6
  58. package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
  59. package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DaK8b3fv.d.ts} +2 -1
  60. package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
  61. package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
  62. package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
  63. package/dist/fuzz.d.ts +8 -16
  64. package/dist/fuzz.js +72 -42
  65. package/dist/fuzz.js.map +1 -1
  66. package/dist/{gepa-dne9JDPL.d.ts → gepa-eESocoDi.d.ts} +64 -12
  67. package/dist/hosted/index.d.ts +14 -7
  68. package/dist/{index-BTEpx9He.d.ts → index-PdX4VnPA.d.ts} +3 -3
  69. package/dist/index.d.ts +97 -55
  70. package/dist/index.js +343 -244
  71. package/dist/index.js.map +1 -1
  72. package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
  73. package/dist/{integrity-qemeBAyx.d.ts → integrity-DqlBiLyK.d.ts} +1 -1
  74. package/dist/kind-factory-ClZmO25A.d.ts +171 -0
  75. package/dist/{llm-client-DyqEH4jH.d.ts → llm-client-qoDd18Qz.d.ts} +27 -3
  76. package/dist/meta-eval/index.d.ts +8 -7
  77. package/dist/meta-eval/index.js +1 -1
  78. package/dist/multishot/index.d.ts +10 -3
  79. package/dist/openapi.json +1 -1
  80. package/dist/pipelines/index.d.ts +16 -6
  81. package/dist/pipelines/index.js +119 -23
  82. package/dist/pipelines/index.js.map +1 -1
  83. package/dist/{kind-factory-DcNg13sZ.d.ts → policy-edit-wG9uFEFm.d.ts} +114 -167
  84. package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration-BWQhJ3vz.d.ts} +23 -4
  85. package/dist/{provenance-Bibyg1U9.d.ts → provenance-DpjwyseI.d.ts} +28 -16
  86. package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
  87. package/dist/{release-report-CCtzajxP.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
  88. package/dist/reporting.d.ts +10 -9
  89. package/dist/{researcher-Dq-EtpbE.d.ts → researcher-C8XyxQsu.d.ts} +7 -7
  90. package/dist/rl.d.ts +17 -12
  91. package/dist/rl.js +2 -2
  92. package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
  93. package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
  94. package/dist/{run-record-B7RTi_ix.d.ts → run-record-BDH49H2E.d.ts} +2 -2
  95. package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
  96. package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
  97. package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +23 -5
  98. package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
  99. package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
  100. package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
  101. package/dist/storyboard/index.d.ts +1 -1
  102. package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
  103. package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
  104. package/dist/traces.d.ts +19 -10
  105. package/dist/traces.js +16 -4
  106. package/dist/{types-C5gJrOVT.d.ts → types-BSw1rOUB.d.ts} +97 -38
  107. package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
  108. package/dist/wire/index.d.ts +28 -19
  109. package/dist/wire/index.js +4 -2
  110. package/docs/design/loop-taxonomy.md +1 -2
  111. package/docs/distributed-driver.md +1 -1
  112. package/package.json +3 -3
  113. package/dist/chunk-4D5RVB3W.js.map +0 -1
  114. package/dist/chunk-5S5NJ63F.js.map +0 -1
  115. package/dist/chunk-ADYLPOSX.js.map +0 -1
  116. package/dist/chunk-FAOEFFRT.js.map +0 -1
  117. package/dist/chunk-GY4SYVPJ.js.map +0 -1
  118. package/dist/chunk-I6LVHOV3.js +0 -205
  119. package/dist/chunk-I6LVHOV3.js.map +0 -1
  120. package/dist/chunk-KG4TD7EQ.js.map +0 -1
  121. package/dist/chunk-LNQEP766.js.map +0 -1
  122. package/dist/chunk-MHNQWM4I.js.map +0 -1
  123. package/dist/chunk-NYFUT3B3.js.map +0 -1
  124. package/dist/chunk-QMXXSNC4.js +0 -761
  125. package/dist/chunk-QMXXSNC4.js.map +0 -1
  126. package/dist/chunk-TLDB7WRY.js.map +0 -1
  127. package/dist/chunk-WSBUZMBU.js.map +0 -1
  128. package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
  129. package/dist/policy-edit-RLn8GWof.d.ts +0 -103
  130. /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
  131. /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
  132. /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
  133. /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
  134. /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
package/dist/traces.js CHANGED
@@ -28,13 +28,14 @@ import {
28
28
  scoreTraceInsightReadiness,
29
29
  tokenizeDomainWords,
30
30
  traceAnalystOnRunComplete
31
- } from "./chunk-TLDB7WRY.js";
31
+ } from "./chunk-YZPO4UHR.js";
32
32
  import {
33
33
  FAILURE_CLASSES,
34
34
  TRACE_SCHEMA_VERSION,
35
35
  aggregateLlm,
36
36
  argHash,
37
37
  groupBy,
38
+ hasCapturedToolArgs,
38
39
  isJudgeSpan,
39
40
  isLlmSpan,
40
41
  isRetrievalSpan,
@@ -45,13 +46,13 @@ import {
45
46
  runFailureClass,
46
47
  runsForScenario,
47
48
  toolSpans
48
- } from "./chunk-MHNQWM4I.js";
49
+ } from "./chunk-LQUTGLOZ.js";
49
50
  import {
50
51
  TRACE_ANALYST_ACTOR_DESCRIPTION,
51
52
  TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
52
53
  TRACE_ANALYST_SUBAGENT_DESCRIPTION,
53
54
  analyzeTraces
54
- } from "./chunk-RPDDVKI7.js";
55
+ } from "./chunk-4JLWXDYA.js";
55
56
  import {
56
57
  DEFAULT_REDACTION_RULES,
57
58
  REDACTION_VERSION,
@@ -60,6 +61,7 @@ import {
60
61
  } from "./chunk-GGE4NNQT.js";
61
62
  import {
62
63
  DEFAULT_TRACE_ANALYST_BUDGETS,
64
+ INPUT_VALUE,
63
65
  LLM_CACHED_TOKENS,
64
66
  LLM_CACHED_TOKEN_ATTR_KEYS,
65
67
  LLM_COST_ATTR_KEYS,
@@ -71,14 +73,18 @@ import {
71
73
  LLM_OUTPUT_TOKENS,
72
74
  LLM_OUTPUT_TOKEN_ATTR_KEYS,
73
75
  OPENINFERENCE_SPAN_KIND,
76
+ OUTPUT_VALUE,
74
77
  OtlpFileTraceStore,
75
78
  SPAN_KIND_ATTR_KEYS,
76
79
  SpanNotFoundError,
80
+ TOOL_ARGS_CAPTURED,
81
+ TOOL_LATENCY_MS,
77
82
  TOOL_NAME,
78
83
  TOOL_NAME_ATTR_KEYS,
79
84
  TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
80
85
  TraceFileMissingError,
81
86
  TraceNotFoundError,
87
+ applyToolSpanOtlpAttributes,
82
88
  asNumber,
83
89
  asString,
84
90
  buildTraceAnalystTools,
@@ -91,7 +97,7 @@ import {
91
97
  stringField,
92
98
  traceAnalystFunctionGroup,
93
99
  traceSpanKindToOpenInferenceKind
94
- } from "./chunk-LNQEP766.js";
100
+ } from "./chunk-S2F4J57L.js";
95
101
  import {
96
102
  RunIntegrityError,
97
103
  assertRunCaptured,
@@ -119,6 +125,7 @@ export {
119
125
  FAILURE_CLASSES,
120
126
  FileSystemRawProviderSink,
121
127
  FileSystemTraceStore,
128
+ INPUT_VALUE,
122
129
  InMemoryRawProviderSink,
123
130
  InMemoryTraceStore,
124
131
  LLM_CACHED_TOKENS,
@@ -134,6 +141,7 @@ export {
134
141
  NoopRawProviderSink,
135
142
  OPENINFERENCE_SPAN_KIND,
136
143
  OTEL_AGENT_EVAL_SCOPE,
144
+ OUTPUT_VALUE,
137
145
  OtlpFileTraceStore,
138
146
  REDACTION_VERSION,
139
147
  ReplayCache,
@@ -141,6 +149,8 @@ export {
141
149
  RunIntegrityError,
142
150
  SPAN_KIND_ATTR_KEYS,
143
151
  SpanNotFoundError,
152
+ TOOL_ARGS_CAPTURED,
153
+ TOOL_LATENCY_MS,
144
154
  TOOL_NAME,
145
155
  TOOL_NAME_ATTR_KEYS,
146
156
  TRACE_ANALYST_ACTOR_DESCRIPTION,
@@ -153,6 +163,7 @@ export {
153
163
  TraceNotFoundError,
154
164
  aggregateLlm,
155
165
  analyzeTraces,
166
+ applyToolSpanOtlpAttributes,
156
167
  argHash,
157
168
  asNumber,
158
169
  asString,
@@ -178,6 +189,7 @@ export {
178
189
  firstStringAttr,
179
190
  flattenOtlpExportToNdjson,
180
191
  groupBy,
192
+ hasCapturedToolArgs,
181
193
  inferDomainKeywords,
182
194
  inferOtlpKind,
183
195
  isJudgeSpan,
@@ -1,4 +1,7 @@
1
- import { d as RunTokenUsage } from './run-record-B7RTi_ix.js';
1
+ import { P as PolicyEditCandidateRecord } from './policy-edit-wG9uFEFm.js';
2
+ import { R as RunPaidCallInput, a as CostChannel, P as PaidCallResult, C as CostLedger, b as CostLedgerSummary } from './cost-ledger-DWy3XdJc.js';
3
+ import { L as LlmCallMetadata } from './llm-client-qoDd18Qz.js';
4
+ import { c as RunTokenUsage } from './run-record-BDH49H2E.js';
2
5
 
3
6
  /**
4
7
  * Pass A substrate types — `runCampaign` is the one primitive every
@@ -82,12 +85,20 @@ interface JudgeDimension {
82
85
  interface JudgeConfig<TArtifact, TScenario extends Scenario = Scenario> {
83
86
  name: string;
84
87
  dimensions: JudgeDimension[];
88
+ /** Stable scoring revision used by campaign resume and verdict caches.
89
+ * Built-in judges derive this from their prompt, model, and rubric. Custom
90
+ * judges should set it when closure state can change without changing code. */
91
+ judgeVersion?: string;
85
92
  /** Score one artifact. Throw on failure — a thrown judge is recorded as a
86
93
  * failed cell, never silently folded into a zero. */
87
94
  score(input: {
88
95
  artifact: TArtifact;
89
96
  scenario: TScenario;
90
97
  signal: AbortSignal;
98
+ /** Shared run spend account and receipt attribution phase. */
99
+ costLedger?: CostLedger;
100
+ costPhase?: string;
101
+ costTags?: Record<string, string>;
91
102
  }): JudgeScore | Promise<JudgeScore>;
92
103
  appliesTo?: (scenario: TScenario) => boolean;
93
104
  }
@@ -104,6 +115,8 @@ interface JudgeScore {
104
115
  dimensions: Record<string, number>;
105
116
  composite: number;
106
117
  notes: string;
118
+ /** Provider metadata for display and diagnostics; accounting uses CostLedger receipts. */
119
+ llmCall?: LlmCallMetadata;
107
120
  /** Set when the judge itself failed (call error, unparseable output).
108
121
  * `composite`/`dimensions` carry no signal — aggregators MUST exclude
109
122
  * failed scores from means instead of folding them into zeros. */
@@ -168,6 +181,9 @@ interface ProposedCandidate {
168
181
  * primitive it used. Survives to `GenerationCandidate.rationale` and the
169
182
  * emitted provenance record. */
170
183
  rationale: string;
184
+ /** Structured, JSON-safe cause for this exact candidate when the proposer
185
+ * can provide one. Policy edits retain the full validated edit here. */
186
+ candidateRecord?: PolicyEditCandidateRecord;
171
187
  }
172
188
  /** Type guard: a proposal carrying its rationale vs a bare
173
189
  * surface. The loop branches on this to populate `GenerationCandidate`. */
@@ -194,6 +210,28 @@ interface ParetoParent {
194
210
  label?: string;
195
211
  rationale?: string;
196
212
  }
213
+ /** Exact measured state for the surface an optimizer is learning from.
214
+ * Unlike a model-authored expected gain, every value here comes from a
215
+ * completed campaign over the designed denominator. */
216
+ interface ScoredSurfaceOutcome {
217
+ /** Optimization/search evidence only. Held-out results must never flow back
218
+ * into a proposer through this type. */
219
+ split: 'search';
220
+ /** Generation that actually measured this surface (`-1` for the baseline). */
221
+ generation: number;
222
+ surfaceHash: string;
223
+ composite: number;
224
+ dimensions: Record<string, number>;
225
+ scenarios: Array<{
226
+ scenarioId: string;
227
+ composite: number;
228
+ notes?: string;
229
+ }>;
230
+ coverage: {
231
+ expectedCells: number;
232
+ scorableCells: number;
233
+ };
234
+ }
197
235
  /** Stateless surface mutation — given findings + current
198
236
  * surface, return N candidate surfaces. Pure transform, no generation
199
237
  * awareness. Reflective-mutation and `AxGEPA` mutators conform. Wrapped by
@@ -221,6 +259,12 @@ interface ProposeContext<TFindings = unknown> {
221
259
  populationSize: number;
222
260
  generation: number;
223
261
  signal: AbortSignal;
262
+ /** Measured baseline for this optimization run. `runOptimization` always
263
+ * supplies it; optional for standalone proposer callers. */
264
+ baselineOutcome?: ScoredSurfaceOutcome;
265
+ /** Measured result for `currentSurface`, the complete global incumbent every
266
+ * new candidate mutates. `runOptimization` always supplies it. */
267
+ incumbentOutcome?: ScoredSurfaceOutcome;
224
268
  /** Optional analysis report produced before proposal. Opaque to the substrate:
225
269
  * the proposer that consumes it owns the shape. */
226
270
  report?: unknown;
@@ -238,6 +282,9 @@ interface ProposeContext<TFindings = unknown> {
238
282
  * scenarios) into a merged candidate. Proposers doing pure single-parent
239
283
  * reflection may ignore it. See {@link ParetoParent}. */
240
284
  paretoParents?: ParetoParent[];
285
+ /** Shared run spend account and receipt attribution phase. */
286
+ costLedger?: CostLedger;
287
+ costPhase?: string;
241
288
  /** FIREWALL (non-negotiable): the held-out judge is write-only — its verdicts
242
289
  * score the chosen output and gate promotion, and are NEVER an input to
243
290
  * proposal/steering (else the optimizer games the acceptance axis = an
@@ -319,6 +366,9 @@ interface GateContext<TArtifact, TScenario extends Scenario> {
319
366
  candidate: number;
320
367
  baseline: number;
321
368
  };
369
+ /** Shared run spend account and receipt attribution phase. */
370
+ costLedger?: CostLedger;
371
+ costPhase?: string;
322
372
  signal: AbortSignal;
323
373
  }
324
374
  interface GateResult {
@@ -357,35 +407,15 @@ interface CampaignArtifactWriter {
357
407
  * backend-integrity guard with ONE source of truth — a field added to
358
408
  * `RunTokenUsage` is a compile error here, not a silent drift. */
359
409
  type CampaignTokenUsage = RunTokenUsage;
360
- /** Cell-scoped cost meter. NOTHING is captured automatically
361
- * the substrate does not intercept the LLM call, so it cannot see cost or
362
- * tokens unless the dispatch reports them. Every LLM cost MUST be reported via
363
- * `observe` and every token count via `observeTokens`; a dispatch that reports
364
- * neither yields a `{cost:0, tokens:0}` cell, which the backend-integrity
365
- * guard (`assertRealBackend`) correctly reads as a stub. Also use `observe`
366
- * for non-LLM spend (sandbox time, tool costs). */
410
+ /** Cell-scoped paid-call entry point. The dispatch places every paid operation
411
+ * inside `runPaidCall`; the returned provider result supplies one receipt with
412
+ * cost, tokens, and resolved model. Calls made outside this method are not
413
+ * admitted or captured. */
367
414
  interface CampaignCostMeter {
368
- observe(amountUsd: number, source: string): void;
369
- /** Record LLM token usage for this cell; accumulates across calls. A cell
370
- * has `costUsd` but no token counts unless the dispatch reports them here —
371
- * and the backend-integrity guard (`assertRealBackend`) keys on
372
- * `tokenUsage`, so a cell that never reports tokens reads as a stub. Any
373
- * dispatch that calls an LLM MUST report its usage. */
374
- observeTokens(usage: CampaignTokenUsage): void;
375
- /** Record the concrete model the backend RESOLVED this cell to at runtime.
376
- * The substrate cannot see the LLM call, so it cannot know which model a
377
- * vendor-locked harness actually served — only the dispatch, reading the
378
- * backend's usage/terminal events, can. A dispatch whose profile declares a
379
- * runtime-resolved model (the `HARNESS_NATIVE_MODEL` sentinel) MUST report
380
- * the resolved, snapshot-bearing id here so the RunRecord pins a real model
381
- * instead of the sentinel. Last write wins (a cell issues one logical run);
382
- * optional because most dispatches declare a concrete model up front. */
383
- observeModel?(model: string): void;
384
- current(): number;
385
- /** Accumulated token usage for this cell (zeros if never observed). */
386
- tokens(): CampaignTokenUsage;
387
- /** The runtime-resolved model reported via `observeModel`, if any. */
388
- resolvedModel?(): string | undefined;
415
+ /** The only paid-call path. Returns a typed result; callers must inspect it. */
416
+ runPaidCall<T>(input: Omit<RunPaidCallInput<T>, 'channel' | 'phase' | 'tags'> & {
417
+ channel?: CostChannel;
418
+ }): Promise<PaidCallResult<T>>;
389
419
  }
390
420
  /** Source tag — required on every store write. Used by the
391
421
  * default training-source filter (production-trace samples NOT used as
@@ -478,15 +508,16 @@ interface CampaignCellResult<TArtifact> {
478
508
  artifact: TArtifact;
479
509
  judgeScores: Record<string, JudgeScore>;
480
510
  costUsd: number;
481
- /** LLM token usage the dispatch reported via `ctx.cost.observeTokens`.
482
- * `{ input: 0, output: 0 }` when the dispatch reported none — which the
483
- * backend-integrity guard reads as a stub. */
511
+ /** True when at least one priced receipt used the model table instead of a provider bill. */
512
+ costEstimated?: boolean;
513
+ /** Exact durable receipts required to reuse this cached result. */
514
+ costCallIds?: string[];
515
+ /** Agent-call token usage committed by `ctx.cost.runPaidCall`.
516
+ * `{ input: 0, output: 0 }` when no paid agent call was recorded. */
484
517
  tokenUsage: CampaignTokenUsage;
485
- /** The concrete model the backend resolved this cell to at runtime, reported
486
- * by the dispatch via `ctx.cost.observeModel`. Set only when the dispatch
487
- * reported it — a profile that declares a concrete model up front has no
488
- * need to. Consumed by `buildRunRecord` to pin the real model when the
489
- * declared model is the `HARNESS_NATIVE_MODEL` sentinel. */
518
+ /** Concrete model from the latest committed agent receipt. Consumed by
519
+ * `buildRunRecord` to pin the model when the declared profile uses a
520
+ * runtime-resolved sentinel. */
490
521
  resolvedModel?: string;
491
522
  durationMs: number;
492
523
  seed: number;
@@ -517,6 +548,29 @@ interface GenerationCandidate {
517
548
  surfaceHash: string;
518
549
  composite: number;
519
550
  ci95: [number, number];
551
+ /** Exact surface this candidate mutated. */
552
+ parentSurfaceHash?: string;
553
+ /** Measured search-split composite of the exact parent surface. */
554
+ parentComposite?: number;
555
+ /** Candidate composite minus its parent's composite. Present only when the
556
+ * candidate completed the designed denominator. */
557
+ observedDeltaFromParent?: number;
558
+ /** Whether this candidate had a scorable result for every designed campaign
559
+ * cell and was therefore eligible for ranking, promotion, and Pareto
560
+ * selection. Older externally-authored records may omit this field; loop
561
+ * records always populate it. */
562
+ eligibleForPromotion?: boolean;
563
+ /** Exact denominator receipt for selection eligibility. Scores stay
564
+ * descriptive: an incomplete candidate is retained with its observed score
565
+ * and errors instead of receiving an invented penalty. */
566
+ coverage?: {
567
+ expectedCells: number;
568
+ scorableCells: number;
569
+ unscorableCells: Array<{
570
+ cellId: string;
571
+ reason: string;
572
+ }>;
573
+ };
520
574
  /** Mean score per judge dimension across all cells (scenarios × reps ×
521
575
  * judges that reported the dimension). */
522
576
  dimensions: Record<string, number>;
@@ -538,10 +592,15 @@ interface GenerationCandidate {
538
592
  * "because rationale Z" the audit requires to survive to the result.
539
593
  * Present when the proposer returned a `ProposedCandidate`. */
540
594
  rationale?: string;
595
+ /** Exact structured cause threaded from the proposer, when available. */
596
+ candidateRecord?: PolicyEditCandidateRecord;
541
597
  }
542
598
  interface CampaignAggregates {
543
599
  byJudge: Record<string, JudgeAggregate>;
544
600
  byScenario: Record<string, ScenarioAggregate>;
601
+ /** Canonical campaign accounting, including worker and judge calls. */
602
+ cost: CostLedgerSummary;
603
+ /** Compatibility alias of `cost.totalCostUsd`. */
545
604
  totalCostUsd: number;
546
605
  cellsExecuted: number;
547
606
  cellsSkipped: number;
@@ -572,4 +631,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
572
631
  scenarios: Array<Pick<TScenario, 'id' | 'kind'>>;
573
632
  }
574
633
 
575
- export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchContext as D, isProposedCandidate as E, labelTrustRank as F, type GateResult as G, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizationProposer as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeDimension as a, type JudgeConfig as b, type DispatchFn as c, type CampaignTraceWriter as d, type GenerationRecord as e, type SurfaceProposer as f, type Gate as g, type GateDecision as h, type CampaignAggregates as i, type CampaignArtifactWriter as j, type CampaignCellResult as k, type CampaignCostMeter as l, type CodeSurface as m, type GateContext as n, type GenerationCandidate as o, type Mutator as p, type OptimizerConfig as q, type SessionScript as r, type LabeledScenarioWrite as s, type LabeledScenarioSampleArgs as t, type LabeledScenarioRecord as u, type LabelTrust as v, type ProposedCandidate as w, type ProposeContext as x, type LabeledScenarioSource as y, type CampaignTokenUsage as z };
634
+ export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchContext as D, type ScoredSurfaceOutcome as E, isProposedCandidate as F, type Gate as G, labelTrustRank as H, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizationProposer as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeDimension as a, type JudgeConfig as b, type SurfaceProposer as c, type GateDecision as d, type CampaignAggregates as e, type CampaignArtifactWriter as f, type CampaignCellResult as g, type CampaignCostMeter as h, type CampaignTraceWriter as i, type CodeSurface as j, type DispatchFn as k, type GateContext as l, type GateResult as m, type GenerationCandidate as n, type GenerationRecord as o, type Mutator as p, type OptimizerConfig as q, type SessionScript as r, type LabeledScenarioWrite as s, type LabeledScenarioSampleArgs as t, type LabeledScenarioRecord as u, type LabelTrust as v, type ProposedCandidate as w, type ProposeContext as x, type LabeledScenarioSource as y, type CampaignTokenUsage as z };
@@ -1,3 +1,4 @@
1
+ import { C as CostLedger, b as CostLedgerSummary } from './cost-ledger-DWy3XdJc.js';
1
2
  import { TCloud } from '@tangle-network/tcloud';
2
3
 
3
4
  interface Scenario {
@@ -50,6 +51,8 @@ interface ScenarioResult {
50
51
  overallScore: number;
51
52
  totalDurationMs: number;
52
53
  artifacts: CollectedArtifacts;
54
+ /** Agent and judge spend attributed to this scenario. */
55
+ cost?: CostLedgerSummary;
53
56
  }
54
57
  interface TurnResult {
55
58
  turnIndex: number;
@@ -96,6 +99,7 @@ interface BenchmarkReport {
96
99
  promptVersion: string;
97
100
  scenarioCount: number;
98
101
  results: ScenarioResult[];
102
+ cost?: CostLedgerSummary;
99
103
  summary: {
100
104
  overallAvg: number;
101
105
  byPersona: Record<string, {
@@ -266,11 +270,22 @@ interface BenchmarkRunnerConfig {
266
270
  passThreshold?: number;
267
271
  generation?: number;
268
272
  promptVersion?: string;
273
+ /** Shared ledger for agent and judge calls made by the benchmark. */
274
+ costLedger?: CostLedger;
275
+ /** Exact maximum provider attempts configured on the supplied TCloud client. */
276
+ tcloudMaximumAttempts?: number;
269
277
  }
270
278
  interface JudgeInput {
271
279
  scenario: Scenario;
272
280
  turns: TurnResult[];
273
281
  artifacts: CollectedArtifacts;
282
+ /** Shared ledger for paid built-in judges. Direct calls default to an uncapped ledger. */
283
+ costLedger?: CostLedger;
284
+ costPhase?: string;
285
+ costTags?: Record<string, string>;
286
+ signal?: AbortSignal;
287
+ /** Exact maximum provider attempts configured on the supplied TCloud client. */
288
+ tcloudMaximumAttempts?: number;
274
289
  }
275
290
  type JudgeFn = (tc: TCloud, input: JudgeInput) => Promise<JudgeScore[]>;
276
291
 
@@ -1,14 +1,17 @@
1
- import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-pDcz1lQ1.js';
2
- import { T as TraceStore } from '../store-BsVi7ncX.js';
1
+ import { C as CostLedger } from '../cost-ledger-DWy3XdJc.js';
2
+ import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-BUnM58xL.js';
3
+ import { a as LlmClientOptions } from '../llm-client-qoDd18Qz.js';
4
+ import { T as TraceStore } from '../store-DGqD0Pyo.js';
3
5
  import { z } from 'zod';
4
6
  import { OpenAPIObject } from 'openapi3-ts/oas31';
5
7
  import * as hono_types from 'hono/types';
6
8
  import { ServerType } from '@hono/node-server';
7
9
  import { Hono } from 'hono';
8
- import '../emitter-BRchAAAx.js';
9
- import '../schema-SGWcK9wa.js';
10
- import '../dataset-NENEzRgk.js';
11
10
  import '../errors-oeQrLqXC.js';
11
+ import '../emitter-CjD7vUwv.js';
12
+ import '../schema-B3Q3l9Z_.js';
13
+ import '../dataset-NENEzRgk.js';
14
+ import '../raw-provider-sink-C46HDghv.js';
12
15
 
13
16
  declare const RubricDimensionSchema: z.ZodObject<{
14
17
  id: z.ZodString;
@@ -122,9 +125,9 @@ declare const TraceEventSchema: z.ZodObject<{
122
125
  runId: z.ZodString;
123
126
  spanId: z.ZodOptional<z.ZodString>;
124
127
  kind: z.ZodEnum<{
125
- policy_violation: "policy_violation";
126
- custom: "custom";
127
128
  error: "error";
129
+ custom: "custom";
130
+ policy_violation: "policy_violation";
128
131
  log: "log";
129
132
  budget_decrement: "budget_decrement";
130
133
  budget_breach: "budget_breach";
@@ -140,9 +143,9 @@ declare const TracesIngestRequestSchema: z.ZodObject<{
140
143
  runId: z.ZodString;
141
144
  spanId: z.ZodOptional<z.ZodString>;
142
145
  kind: z.ZodEnum<{
143
- policy_violation: "policy_violation";
144
- custom: "custom";
145
146
  error: "error";
147
+ custom: "custom";
148
+ policy_violation: "policy_violation";
146
149
  log: "log";
147
150
  budget_decrement: "budget_decrement";
148
151
  budget_breach: "budget_breach";
@@ -165,8 +168,8 @@ declare const FeedbackLabelSchema: z.ZodObject<{
165
168
  id: z.ZodOptional<z.ZodString>;
166
169
  source: z.ZodEnum<{
167
170
  judge: "judge";
168
- system: "system";
169
171
  user: "user";
172
+ system: "system";
170
173
  policy: "policy";
171
174
  environment: "environment";
172
175
  metric: "metric";
@@ -198,10 +201,10 @@ declare const FeedbackAttemptSchema: z.ZodObject<{
198
201
  id: z.ZodString;
199
202
  stepIndex: z.ZodNumber;
200
203
  artifactType: z.ZodEnum<{
201
- action: "action";
202
- decision: "decision";
203
204
  text: "text";
204
205
  code: "code";
206
+ action: "action";
207
+ decision: "decision";
205
208
  plan: "plan";
206
209
  research: "research";
207
210
  ui: "ui";
@@ -226,8 +229,8 @@ declare const FeedbackAttemptSchema: z.ZodObject<{
226
229
  id: z.ZodOptional<z.ZodString>;
227
230
  source: z.ZodEnum<{
228
231
  judge: "judge";
229
- system: "system";
230
232
  user: "user";
233
+ system: "system";
231
234
  policy: "policy";
232
235
  environment: "environment";
233
236
  metric: "metric";
@@ -270,10 +273,10 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
270
273
  id: z.ZodString;
271
274
  stepIndex: z.ZodNumber;
272
275
  artifactType: z.ZodEnum<{
273
- action: "action";
274
- decision: "decision";
275
276
  text: "text";
276
277
  code: "code";
278
+ action: "action";
279
+ decision: "decision";
277
280
  plan: "plan";
278
281
  research: "research";
279
282
  ui: "ui";
@@ -298,8 +301,8 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
298
301
  id: z.ZodOptional<z.ZodString>;
299
302
  source: z.ZodEnum<{
300
303
  judge: "judge";
301
- system: "system";
302
304
  user: "user";
305
+ system: "system";
303
306
  policy: "policy";
304
307
  environment: "environment";
305
308
  metric: "metric";
@@ -334,8 +337,8 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
334
337
  id: z.ZodOptional<z.ZodString>;
335
338
  source: z.ZodEnum<{
336
339
  judge: "judge";
337
- system: "system";
338
340
  user: "user";
341
+ system: "system";
339
342
  policy: "policy";
340
343
  environment: "environment";
341
344
  metric: "metric";
@@ -438,7 +441,13 @@ declare class WireError extends Error {
438
441
  readonly details?: unknown | undefined;
439
442
  constructor(code: string, message: string, status?: number, details?: unknown | undefined);
440
443
  }
441
- declare function handleJudge(req: JudgeRequest): Promise<JudgeResult>;
444
+ interface HandleJudgeOptions {
445
+ costLedger?: CostLedger;
446
+ costPhase?: string;
447
+ llm?: LlmClientOptions;
448
+ signal?: AbortSignal;
449
+ }
450
+ declare function handleJudge(req: JudgeRequest, options?: HandleJudgeOptions): Promise<JudgeResult>;
442
451
  declare function handleListRubrics(): ListRubricsResponse;
443
452
  declare function handleVersion(): VersionResponse;
444
453
  /**
@@ -569,4 +578,4 @@ interface StartedServer {
569
578
  */
570
579
  declare function startServerAsync(opts?: ServeOptions): Promise<StartedServer>;
571
580
 
572
- export { BUILTIN_RUBRICS, type ErrorResponse, ErrorResponseSchema, type FailureMode, FailureModeSchema, FeedbackAttemptSchema, type FeedbackIngestResponse, FeedbackIngestResponseSchema, FeedbackLabelSchema, type FeedbackTrajectory, FeedbackTrajectorySchema, HealthResponseSchema, type IngestionStores, type JudgeRequest, JudgeRequestSchema, type JudgeResult, JudgeResultSchema, type ListRubricsResponse, ListRubricsResponseSchema, type Rubric, type RubricDimension, RubricDimensionSchema, type RubricInfo, RubricInfoSchema, RubricSchema, type ServeOptions, type StartedServer, type TraceEvent, TraceEventSchema, type TracesIngestRequest, TracesIngestRequestSchema, type TracesIngestResponse, TracesIngestResponseSchema, type VersionResponse, VersionResponseSchema, WIRE_VERSION, WireError, buildOpenApi, createApp, dispatchRpc, getBuiltinRubric, handleFeedbackIngest, handleJudge, handleListRubrics, handleTracesIngest, handleVersion, hashRubric, listBuiltinRubrics, runRpcBatch, runRpcOnce, startServer, startServerAsync };
581
+ export { BUILTIN_RUBRICS, type ErrorResponse, ErrorResponseSchema, type FailureMode, FailureModeSchema, FeedbackAttemptSchema, type FeedbackIngestResponse, FeedbackIngestResponseSchema, FeedbackLabelSchema, type FeedbackTrajectory, FeedbackTrajectorySchema, type HandleJudgeOptions, HealthResponseSchema, type IngestionStores, type JudgeRequest, JudgeRequestSchema, type JudgeResult, JudgeResultSchema, type ListRubricsResponse, ListRubricsResponseSchema, type Rubric, type RubricDimension, RubricDimensionSchema, type RubricInfo, RubricInfoSchema, RubricSchema, type ServeOptions, type StartedServer, type TraceEvent, TraceEventSchema, type TracesIngestRequest, TracesIngestRequestSchema, type TracesIngestResponse, TracesIngestResponseSchema, type VersionResponse, VersionResponseSchema, WIRE_VERSION, WireError, buildOpenApi, createApp, dispatchRpc, getBuiltinRubric, handleFeedbackIngest, handleJudge, handleListRubrics, handleTracesIngest, handleVersion, hashRubric, listBuiltinRubrics, runRpcBatch, runRpcOnce, startServer, startServerAsync };
@@ -34,8 +34,10 @@ import {
34
34
  runRpcOnce,
35
35
  startServer,
36
36
  startServerAsync
37
- } from "../chunk-4D5RVB3W.js";
38
- import "../chunk-GY4SYVPJ.js";
37
+ } from "../chunk-LTVG32KX.js";
38
+ import "../chunk-NJC7U437.js";
39
+ import "../chunk-VCTY3W6J.js";
40
+ import "../chunk-VI2UW6B6.js";
39
41
  import "../chunk-PC4UYEBM.js";
40
42
  import "../chunk-ONWEPEDO.js";
41
43
  import "../chunk-PZ5AY32C.js";
@@ -201,8 +201,7 @@ consume the **same dataset** the flywheel builds.
201
201
  agent-runtime.
202
202
  - **runCampaign** — a measurement: a surface scored over N scenarios × M reps.
203
203
  agent-eval. (A "campaign" = a coordinated batch of measurements.)
204
- - **runOptimization** — the improvement loop body: proposer suggests surfaces,
205
- each measured by a campaign, top-K promoted per generation. agent-eval.
204
+ - **runOptimization** — the improvement loop body: proposer suggests surfaces, each is measured, and only a candidate that beats the global incumbent is promoted. agent-eval.
206
205
  - **runImprovementLoop** — `runOptimization` + holdout re-score + release gate
207
206
  + optional PR. agent-eval.
208
207
  - **runAnalystLoop** — reflective autoresearch: findings + knowledge updates +
@@ -135,7 +135,7 @@ round-robin, region-affinity from a previous run, scheduling table).
135
135
  | **Auth** | Bearer token on `Authorization`; pluggable via `auth: string \| () => string \| Promise<string>` for rotation/refresh. |
136
136
  | **Payload size** | Server enforces `maxBodyBytes` (default 10 MB). |
137
137
  | **Traces** | Both ends emit OTel — if both point at the same OTLP collector, you get a unified trace per cell. See `docs/adapters-observability.md`. |
138
- | **Cost** | Worker's `ctx.cost.observe(usd, source)` is local to the worker process. Roll up server-side and attach to your worker-side telemetry; we don't (yet) forward cost back to the coordinator. Tracked as follow-up. |
138
+ | **Cost** | Worker's `ctx.cost.runPaidCall(...)` writes durable receipts in the worker process. Roll up those receipts server-side and attach them to worker telemetry; they are not forwarded to the coordinator automatically. |
139
139
 
140
140
  ## Running the reference example
141
141
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.115.3",
3
+ "version": "0.117.0",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {
@@ -148,7 +148,7 @@
148
148
  "@hono/node-server": "^2.0.0",
149
149
  "@tangle-network/agent-interface": "^0.22.0",
150
150
  "@tangle-network/tcloud": "^0.4.14",
151
- "hono": "^4.12.16",
151
+ "hono": "^4.12.30",
152
152
  "zod": "^4.3.6"
153
153
  },
154
154
  "devDependencies": {
@@ -165,7 +165,7 @@
165
165
  "minimumReleaseAge": 4320,
166
166
  "overrides": {
167
167
  "postcss@<8.5.10": "^8.5.10",
168
- "ws@>=8.0.0 <8.20.1": "^8.20.1"
168
+ "ws@>=8.0.0 <8.21.0": "^8.21.0"
169
169
  }
170
170
  },
171
171
  "engines": {