@tangle-network/agent-eval 0.128.1 → 0.129.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. package/CHANGELOG.md +271 -0
  2. package/README.md +18 -0
  3. package/dist/analyst/index.d.ts +107 -165
  4. package/dist/analyst/index.js +5 -9
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/belief-state/index.d.ts +2 -19
  7. package/dist/belief-state/index.js +30 -31
  8. package/dist/belief-state/index.js.map +1 -1
  9. package/dist/benchmarks/index.d.ts +5 -8
  10. package/dist/benchmarks/index.js +12 -11
  11. package/dist/builder-eval/index.js +1 -1
  12. package/dist/campaign/index.d.ts +30 -39
  13. package/dist/campaign/index.js +11 -10
  14. package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
  15. package/dist/chunk-2QU3YOPR.js.map +1 -0
  16. package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
  17. package/dist/chunk-3OCR4R5I.js.map +1 -0
  18. package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
  19. package/dist/chunk-56TAVBOK.js.map +1 -0
  20. package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
  21. package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
  22. package/dist/chunk-BSO5JDQH.js.map +1 -0
  23. package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
  24. package/dist/chunk-C6LXANRU.js.map +1 -0
  25. package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
  26. package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
  27. package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
  28. package/dist/chunk-EG66UGL4.js.map +1 -0
  29. package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
  30. package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
  31. package/dist/chunk-G7MGMCZD.js.map +1 -0
  32. package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
  33. package/dist/chunk-H23X7XKK.js.map +1 -0
  34. package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
  35. package/dist/chunk-HPWUNB47.js.map +1 -0
  36. package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
  37. package/dist/chunk-IYCLP2N2.js.map +1 -0
  38. package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
  39. package/dist/chunk-JQSF5DQT.js.map +1 -0
  40. package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
  41. package/dist/chunk-M4YBQKIJ.js.map +1 -0
  42. package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
  43. package/dist/chunk-OIUOT4QD.js +44 -0
  44. package/dist/chunk-OIUOT4QD.js.map +1 -0
  45. package/dist/chunk-OWN5NPMC.js +152 -0
  46. package/dist/chunk-OWN5NPMC.js.map +1 -0
  47. package/dist/chunk-PC5DOSM7.js +579 -0
  48. package/dist/chunk-PC5DOSM7.js.map +1 -0
  49. package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
  50. package/dist/chunk-QB6BDBP2.js.map +1 -0
  51. package/dist/chunk-RXHCETDZ.js +536 -0
  52. package/dist/chunk-RXHCETDZ.js.map +1 -0
  53. package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
  54. package/dist/chunk-SFLLL76A.js.map +1 -0
  55. package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
  56. package/dist/chunk-T6RLYGAD.js.map +1 -0
  57. package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
  58. package/dist/chunk-TJVT4QFF.js.map +1 -0
  59. package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
  60. package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
  61. package/dist/chunk-U4L7JRPZ.js.map +1 -0
  62. package/dist/chunk-U4PHLT2N.js +419 -0
  63. package/dist/chunk-U4PHLT2N.js.map +1 -0
  64. package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
  65. package/dist/chunk-VCZ5FQYW.js.map +1 -0
  66. package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
  67. package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
  68. package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
  69. package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
  70. package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
  71. package/dist/chunk-ZHTZ4EYI.js.map +1 -0
  72. package/dist/cli.js +6 -5
  73. package/dist/cli.js.map +1 -1
  74. package/dist/contract/index.d.ts +47 -87
  75. package/dist/contract/index.js +14 -13
  76. package/dist/contract/index.js.map +1 -1
  77. package/dist/control.js +3 -2
  78. package/dist/fuzz.js +3 -2
  79. package/dist/fuzz.js.map +1 -1
  80. package/dist/index.d.ts +659 -203
  81. package/dist/index.js +145 -117
  82. package/dist/index.js.map +1 -1
  83. package/dist/meta-eval/index.js +2 -2
  84. package/dist/multishot/index.d.ts +3 -4
  85. package/dist/multishot/index.js.map +1 -1
  86. package/dist/openapi.json +1 -1
  87. package/dist/pipelines/index.js +5 -5
  88. package/dist/reporting.d.ts +14 -0
  89. package/dist/reporting.js +7 -6
  90. package/dist/rl.d.ts +652 -82
  91. package/dist/rl.js +415 -171
  92. package/dist/rl.js.map +1 -1
  93. package/dist/rollout/index.d.ts +1071 -32
  94. package/dist/rollout/index.js +68 -10
  95. package/dist/run-campaign-OJJ7CZF4.js +18 -0
  96. package/dist/supervisor-run/index.d.ts +114 -4
  97. package/dist/supervisor-run/index.js +4 -3
  98. package/dist/traces.d.ts +1 -1
  99. package/dist/traces.js +6 -5
  100. package/dist/wire/index.d.ts +10 -11
  101. package/dist/wire/index.js +3 -3
  102. package/docs/feature-guide.md +1 -1
  103. package/docs/rollout.md +116 -2
  104. package/package.json +4 -4
  105. package/dist/chunk-2JX3CFMB.js.map +0 -1
  106. package/dist/chunk-DJKY2TSY.js.map +0 -1
  107. package/dist/chunk-EJGRPCO3.js.map +0 -1
  108. package/dist/chunk-EOSZT7PL.js.map +0 -1
  109. package/dist/chunk-EZJEIH2R.js.map +0 -1
  110. package/dist/chunk-IHQDPH7D.js.map +0 -1
  111. package/dist/chunk-MHELPNRP.js.map +0 -1
  112. package/dist/chunk-NACAGYSY.js.map +0 -1
  113. package/dist/chunk-NKAGIDE2.js.map +0 -1
  114. package/dist/chunk-NYLOYM6N.js.map +0 -1
  115. package/dist/chunk-PBE2LOSS.js.map +0 -1
  116. package/dist/chunk-TT4KNT67.js +0 -124
  117. package/dist/chunk-TT4KNT67.js.map +0 -1
  118. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  119. package/dist/chunk-UWZZKKU7.js +0 -237
  120. package/dist/chunk-UWZZKKU7.js.map +0 -1
  121. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  122. package/dist/chunk-VGRCHJON.js.map +0 -1
  123. package/dist/chunk-VLOATJQ2.js.map +0 -1
  124. package/dist/chunk-VZSRQ272.js.map +0 -1
  125. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  126. package/dist/chunk-XDWDC2MP.js.map +0 -1
  127. package/dist/chunk-XPRT64IE.js.map +0 -1
  128. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  129. /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
  130. /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
  131. /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
  132. /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
  133. /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
  134. /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
  135. /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
  136. /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
  137. /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
@@ -216,9 +216,8 @@ type CostLedgerHandle = Pick<CostLedger, Exclude<keyof CostLedger, 'listPending'
216
216
  * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
217
217
  * )
218
218
  *
219
- * This is THE llm-calling seam for agent-eval primitives that need structured
220
- * output (semantic concept judge, reviewer directives, critic scores). Primitives
221
- * that need free-form text use `callLlm` and parse output themselves.
219
+ * `createChatClient` wraps this implementation for provider-neutral package
220
+ * entry points. Direct callers can use `callLlm` or `callLlmJson`.
222
221
  */
223
222
 
224
223
  interface LlmMessage {
@@ -652,7 +651,7 @@ interface JudgeConfig<TArtifact, TScenario extends Scenario$1 = Scenario$1> {
652
651
  /** The canonical judge verdict shape — one declaration, shared by campaign
653
652
  * judges and the multishot judge runner (which re-exports this type).
654
653
  *
655
- * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the legacy
654
+ * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
656
655
  * multishot runner emits 0-10. Cross-scale comparison must go through
657
656
  * `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
658
657
  * promotion-policy) — never renormalize a producer's values in place, as
@@ -872,9 +871,6 @@ interface SurfaceProposer<TFindings = unknown> {
872
871
  reason?: string;
873
872
  };
874
873
  }
875
- /** Optional vocabulary alias. The loop is the optimizer; this object is the
876
- * proposer inside that loop. */
877
- type OptimizationProposer<TFindings = unknown> = SurfaceProposer<TFindings>;
878
874
  interface OptimizerConfigBase {
879
875
  populationSize: number;
880
876
  maxGenerations: number;
@@ -1152,8 +1148,6 @@ interface CampaignAggregates {
1152
1148
  byScenario: Record<string, ScenarioAggregate>;
1153
1149
  /** Canonical campaign accounting, including worker and judge calls. */
1154
1150
  cost: CostLedgerSummary;
1155
- /** Compatibility alias of `cost.totalCostUsd`. */
1156
- totalCostUsd: number;
1157
1151
  /** Cells whose dispatch completed, including cells whose later judge failed. */
1158
1152
  cellsExecuted: number;
1159
1153
  cellsSkipped: number;
@@ -1164,7 +1158,7 @@ interface CampaignAggregates {
1164
1158
  cellsDispatchFailed?: number;
1165
1159
  /** Present on results that record failure stages. */
1166
1160
  cellsJudgeFailed?: number;
1167
- /** Legacy failures whose stage was not recorded. */
1161
+ /** Failures whose stage could not be classified. */
1168
1162
  cellsUnclassifiedFailed?: number;
1169
1163
  }
1170
1164
  interface CampaignResult<TArtifact = unknown, TScenario extends Scenario$1 = Scenario$1> {
@@ -1222,7 +1216,7 @@ interface CampaignStorage {
1222
1216
  write(path: string, content: string | Uint8Array): void;
1223
1217
  /** Append only when the current UTF-8 byte length matches `expectedBytes`.
1224
1218
  * Returns the new length, or undefined when another writer won. */
1225
- append?(path: string, content: string, expectedBytes: number): number | undefined;
1219
+ append(path: string, content: string, expectedBytes: number): number | undefined;
1226
1220
  }
1227
1221
  /** Node-filesystem storage — the default. Lazily requires `node:fs` so the
1228
1222
  * module imports cleanly in non-Node runtimes (where the caller passes
@@ -1358,9 +1352,9 @@ interface RunCampaignOptions<TScenario extends Scenario$1, TArtifact> {
1358
1352
  }) => string | undefined;
1359
1353
  }
1360
1354
  /** Durable `<cell>/failure-receipt.json` written before a failed cell can
1361
- * trigger campaign-wide cancellation. The cell keeps its dispatch-only usage
1362
- * fields for compatibility; `cost` covers every settled agent and judge call
1363
- * attributed to this exact run attempt. */
1355
+ * trigger campaign-wide cancellation. The cell records dispatch measurements;
1356
+ * `cost` covers every settled agent and judge call attributed to this exact run
1357
+ * attempt. */
1364
1358
  interface CampaignCellFailureReceipt<TArtifact = unknown> {
1365
1359
  schemaVersion: 1;
1366
1360
  runAttemptId: string;
@@ -1632,42 +1626,27 @@ interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario$1> exte
1632
1626
  declare function runImprovementLoop<TScenario extends Scenario$1, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
1633
1627
 
1634
1628
  /**
1635
- * ChatClient the single LLM abstraction analysts call.
1636
- *
1637
- * agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
1638
- * graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
1639
- * mixed patterns force every analyst author to pick a transport, which
1640
- * couples analyst code to runtime concerns (cli-bridge vs router vs
1641
- * sandbox-sdk) it shouldn't know about.
1642
- *
1643
- * `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
1644
- * The operator decides at the registry boundary which transport binds
1645
- * to it. Analyst code stays transport-agnostic; swapping production
1646
- * (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
1647
- * line factory call.
1648
- *
1649
- * Designed to coexist: existing `LlmClient` callers and existing
1650
- * `TCloud`-based judges keep working untouched. New analyst code uses
1651
- * `ChatClient`. When old call sites migrate, they pick up budgeting,
1652
- * cancellation, and unified telemetry for free.
1629
+ * Provider-neutral chat contract for every model call made by agent-eval.
1630
+ *
1631
+ * Callers choose the transport at the package boundary with `createChatClient`.
1632
+ * Evaluation code receives canonical requests and results without importing a
1633
+ * provider SDK.
1653
1634
  */
1654
1635
 
1655
1636
  /**
1656
- * Unified chat interface. Mirrors LlmCallRequest/Result so the OpenAI-
1657
- * compatible mental model stays. Two methods: a one-shot `chat()` and
1658
- * an `streamChat()` for future agentic loops (not yet exposed).
1637
+ * Unified chat interface using the package's canonical LLM request and result.
1659
1638
  */
1660
1639
  interface ChatClient {
1661
- /** Display name of the bound transport included in telemetry. */
1640
+ /** Display name of the bound transport, included in telemetry. */
1662
1641
  readonly transport: ChatTransport;
1663
- /** Default model when caller omits — operators bind this per environment. */
1642
+ /** Default model when the caller omits one. */
1664
1643
  readonly defaultModel?: string;
1665
1644
  /** Total provider attempts this transport can make for one chat call. */
1666
1645
  readonly maximumAttempts?: number;
1667
1646
  /** Implementations must enforce `req.maxTokens` when it is present. */
1668
1647
  chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
1669
1648
  }
1670
- type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
1649
+ type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'custom' | 'mock';
1671
1650
  interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
1672
1651
  /** Optional — falls back to ChatClient.defaultModel. */
1673
1652
  model?: string;
@@ -1683,7 +1662,7 @@ interface ChatCallOpts {
1683
1662
  /** Stable provider idempotency key for retries/redrives of one paid call. */
1684
1663
  idempotencyKey?: string;
1685
1664
  }
1686
- type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
1665
+ type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | CustomTransportOpts | MockTransportOpts;
1687
1666
  interface BaseTransportOpts {
1688
1667
  defaultModel?: string;
1689
1668
  /** Total provider attempts. Required for opaque transports used in capped runs. */
@@ -1705,15 +1684,18 @@ interface DirectProviderTransportOpts extends BaseTransportOpts {
1705
1684
  apiKey: string;
1706
1685
  }
1707
1686
  /**
1708
- * Sandbox-SDK transport. Provided as a thin pass-through: the caller
1709
- * supplies a callable that mimics LlmClient.chat() against an already-
1710
- * configured Sandbox handle. We don't import the SDK here to keep
1711
- * agent-eval dep-free of @tangle-network/sandbox.
1687
+ * Sandbox-SDK transport. The caller supplies a canonical chat function for an
1688
+ * already-configured Sandbox handle, so agent-eval does not import the SDK.
1712
1689
  */
1713
1690
  interface SandboxSdkTransportOpts extends BaseTransportOpts {
1714
1691
  transport: 'sandbox-sdk';
1715
1692
  chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
1716
1693
  }
1694
+ /** Caller-adapted SDK or transport returning the canonical ChatResponse shape. */
1695
+ interface CustomTransportOpts extends BaseTransportOpts {
1696
+ transport: 'custom';
1697
+ chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
1698
+ }
1717
1699
  /**
1718
1700
  * Mock transport for tests. The handler receives the request and returns
1719
1701
  * whatever the test wants. No retries, no JSON-schema degrade.
@@ -2550,6 +2532,18 @@ interface VerifiableRewardExtractionOptions {
2550
2532
  * doesn't report one. Default `0.7`.
2551
2533
  */
2552
2534
  judgeConfidenceFloor?: number;
2535
+ /**
2536
+ * Whether the anti-Goodhart realness gate applies. Default `true`, and the
2537
+ * default is the one every training path must keep.
2538
+ *
2539
+ * Set `false` ONLY for detection and analysis. `rl/reward-hacking.ts` does,
2540
+ * for the same reason it reads `observedScore` for its proxy: it measures the
2541
+ * DIVERGENCE between the judge signal and the deterministic one, and a
2542
+ * deterministic reward that another gate already forced to 0 manufactures
2543
+ * exactly that divergence on exactly the gamed population. The detector would
2544
+ * then be re-reporting a verdict it was supposed to reach independently.
2545
+ */
2546
+ applyRealnessGate?: boolean;
2553
2547
  }
2554
2548
 
2555
2549
  /**
@@ -2835,8 +2829,6 @@ interface JudgeInput {
2835
2829
  costPhase?: string;
2836
2830
  costTags?: Record<string, string>;
2837
2831
  signal?: AbortSignal;
2838
- /** Exact maximum provider attempts configured on the supplied TCloud client. */
2839
- tcloudMaximumAttempts?: number;
2840
2832
  }
2841
2833
 
2842
2834
  interface PairedBootstrapResult {
@@ -4747,13 +4739,8 @@ interface AnalystRunSummary {
4747
4739
  reason?: string;
4748
4740
  findings_count: number;
4749
4741
  latency_ms: number;
4750
- cost_usd: number;
4751
- /**
4752
- * Additive receipt for model usage. Registry-produced summaries populate it
4753
- * even when the analyst emits no findings. `cost_usd` remains the legacy
4754
- * numeric field; inspect `usage.cost` before treating zero as observed.
4755
- */
4756
- usage?: AnalystUsageReceipt;
4742
+ /** Additive model usage and cost provenance for this analyst. */
4743
+ usage: AnalystUsageReceipt;
4757
4744
  /** When `status='failed'`: the error class + message, never the full stack. */
4758
4745
  error?: {
4759
4746
  class: string;
@@ -4815,12 +4802,9 @@ type AnalystRunEvent = {
4815
4802
  /**
4816
4803
  * Typed Ax output for analyst findings.
4817
4804
  *
4818
- * Replaces the legacy `findings:string[]` pattern (where every bullet
4819
- * became a flat-severity `AnalystFinding`) with a structured object
4820
- * array. Ax binds the field as `findings:json[]` so the provider emits
4821
- * native structured output; at the kind-factory boundary we Zod-validate
4822
- * each emitted finding so malformed rows fail loud instead of being
4823
- * silently lifted with default severity.
4805
+ * Ax binds the field as `findings:json[]` so the provider emits native
4806
+ * structured output. At the kind-factory boundary every row is validated
4807
+ * before it becomes an `AnalystFinding`.
4824
4808
  *
4825
4809
  * Why not `f.object().array()` directly in the signature? The Ax
4826
4810
  * signature string `question:string -> findings:json[]` already lets
@@ -4829,31 +4813,7 @@ type AnalystRunEvent = {
4829
4813
  * validation surface independent of which Ax version is installed.
4830
4814
  */
4831
4815
 
4832
- /** Original public schema retained for stored rows and callback contracts. */
4833
4816
  declare const RawAnalystFindingSchema: z.ZodObject<{
4834
- evidence_uri: z.ZodString;
4835
- evidence_excerpt: z.ZodOptional<z.ZodString>;
4836
- severity: z.ZodEnum<{
4837
- low: "low";
4838
- high: "high";
4839
- medium: "medium";
4840
- critical: "critical";
4841
- info: "info";
4842
- }>;
4843
- claim: z.ZodString;
4844
- subject: z.ZodOptional<z.ZodString>;
4845
- confidence: z.ZodNumber;
4846
- rationale: z.ZodOptional<z.ZodString>;
4847
- recommended_action: z.ZodOptional<z.ZodString>;
4848
- }, z.core.$strict>;
4849
- type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
4850
- /**
4851
- * Canonical plural-evidence contract. The preprocessor accepts the original
4852
- * `evidence_uri` / `evidence_excerpt` pair and normalizes it into one evidence
4853
- * item so persisted rows and older model fixtures remain readable. New output
4854
- * always receives the plural shape.
4855
- */
4856
- declare const CanonicalRawAnalystFindingSchema: z.ZodPreprocess<z.ZodObject<{
4857
4817
  evidence: z.ZodArray<z.ZodObject<{
4858
4818
  uri: z.ZodString;
4859
4819
  excerpt: z.ZodOptional<z.ZodString>;
@@ -4870,8 +4830,8 @@ declare const CanonicalRawAnalystFindingSchema: z.ZodPreprocess<z.ZodObject<{
4870
4830
  confidence: z.ZodNumber;
4871
4831
  rationale: z.ZodOptional<z.ZodString>;
4872
4832
  recommended_action: z.ZodOptional<z.ZodString>;
4873
- }, z.core.$strict>>;
4874
- type CanonicalRawAnalystFinding = z.infer<typeof CanonicalRawAnalystFindingSchema>;
4833
+ }, z.core.$strict>;
4834
+ type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
4875
4835
 
4876
4836
  /**
4877
4837
  * Analyst-kind factory — the typed way to define trace analysts.
@@ -4880,7 +4840,7 @@ type CanonicalRawAnalystFinding = z.infer<typeof CanonicalRawAnalystFindingSchem
4880
4840
  * and bounded Ax subqueries target one failure-mode lens (failure-mode
4881
4841
  * classification, knowledge gap discovery, knowledge poisoning,
4882
4842
  * self-improvement, ...). Kinds emit findings in the typed
4883
- * `CanonicalRawAnalystFinding` shape via a JSON-array Ax output; the factory
4843
+ * `RawAnalystFinding` shape via a JSON-array Ax output; the factory
4884
4844
  * validates each row with Zod and lifts it into `AnalystFinding[]`.
4885
4845
  *
4886
4846
  * Composition rules:
@@ -4943,7 +4903,7 @@ interface TraceAnalystKindSpec {
4943
4903
  */
4944
4904
  interface TraceAnalystGolden {
4945
4905
  question: string;
4946
- expected: ReadonlyArray<Omit<CanonicalRawAnalystFinding, 'confidence'>>;
4906
+ expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
4947
4907
  }
4948
4908
 
4949
4909
  /**
@@ -5806,4 +5766,4 @@ interface FromOtelSpansOptions {
5806
5766
  }
5807
5767
  declare function fromOtelSpans(opts: FromOtelSpansOptions): RunRecord[];
5808
5768
 
5809
- export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, type AnalystFinding, type AnalyzeRunsOptions, type AuthoringProvenance, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignStorage, type CampaignTraceWriter, type CandidateExperimentExecutionInput, type ChatClient, type CodeAgentSessionAction, type CodeAgentSessionActionKind, type CodeAgentSessionActionStatus, type CodeAgentSessionActionSurface, type CodeAgentSessionDiagnostic, type CodeAgentSessionExecutionReceipt, type CodeAgentSessionIntakeOptions, type CodeAgentSessionIntakeResult, type CodeAgentSessionMetrics, type CodeAgentSessionObservation, type CodeAgentSessionSource, type CodeAgentSessionTerminalStatus, type CodeSurface, type CompareCandidateExperimentOptions, type CompareOptimizationMethodsOptions, type ComparisonCost, type CostLedgerHandle, type CostProvenanceSummary, type CreateChatClientOpts, type DefaultAnalystRegistryOptions, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DefineAgentEvalOptions, type DefinedAgentEval, type DeploymentOutcome, type DispatchFn as Dispatch, type DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type EvaluatePairedMeasurementsOptions, type EvidenceVector, type ExecutionErrorOutcomeCell, type ExecutionInsight, type ExecutionReport, type ExternalOptimizationExample, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, type FailureClassTally, type FailureClusterInsight, type FeedbackTableMeta, type FeedbackTableRow, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type HeldOutGateOptions, type HostedTenant, InMemoryOutcomeStore, type InsightReport, type InterRaterInsight, type JudgeConfig, type JudgeDimension, type JudgeInsight, type JudgeScore, type LiftInsight, type LlmJudgeDimension, type LlmJudgeOptions, type MutableSurface, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationPackageSource, type OptimizationProposer, type OptimizationTokenUsage, type OptimizerConfig, type OptimizerModelBudget, type OutcomeCorrelationInsight, type OutcomeStore, type PairedMeasurement, type PairedMeasurementAdapter, type PairedMeasurementEvaluation, type ParetoSignificanceGateOptions, type ParsedCodeAgentJsonl, type PartitionByAuthoringModelResult, type PromotionObjective, type PromotionPolicy, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, type Recommendation, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReleaseSummary, type RunCampaignOptions, type RunCandidateExperimentOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunRecordRejection, type ScalarDistribution, type Scenario$1 as Scenario, type SealCandidateBenchmarkSuiteOptions, type SelfImproveBudget, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, type SessionScript, type SkillOptOptimizationMethodConfig, type SkillOptRunnerCommand, type SkillOptTrainerConfig, type SummarizeExecutionOptions, type SurfaceProposer, type TokenUsageInsight, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, summarizeExecution, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
5769
+ export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, type AnalystFinding, type AnalyzeRunsOptions, type AuthoringProvenance, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignStorage, type CampaignTraceWriter, type CandidateExperimentExecutionInput, type ChatClient, type CodeAgentSessionAction, type CodeAgentSessionActionKind, type CodeAgentSessionActionStatus, type CodeAgentSessionActionSurface, type CodeAgentSessionDiagnostic, type CodeAgentSessionExecutionReceipt, type CodeAgentSessionIntakeOptions, type CodeAgentSessionIntakeResult, type CodeAgentSessionMetrics, type CodeAgentSessionObservation, type CodeAgentSessionSource, type CodeAgentSessionTerminalStatus, type CodeSurface, type CompareCandidateExperimentOptions, type CompareOptimizationMethodsOptions, type ComparisonCost, type CostLedgerHandle, type CostProvenanceSummary, type CreateChatClientOpts, type DefaultAnalystRegistryOptions, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DefineAgentEvalOptions, type DefinedAgentEval, type DeploymentOutcome, type DispatchFn as Dispatch, type DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type EvaluatePairedMeasurementsOptions, type EvidenceVector, type ExecutionErrorOutcomeCell, type ExecutionInsight, type ExecutionReport, type ExternalOptimizationExample, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, type FailureClassTally, type FailureClusterInsight, type FeedbackTableMeta, type FeedbackTableRow, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type HeldOutGateOptions, type HostedTenant, InMemoryOutcomeStore, type InsightReport, type InterRaterInsight, type JudgeConfig, type JudgeDimension, type JudgeInsight, type JudgeScore, type LiftInsight, type LlmJudgeDimension, type LlmJudgeOptions, type MutableSurface, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationPackageSource, type OptimizationTokenUsage, type OptimizerConfig, type OptimizerModelBudget, type OutcomeCorrelationInsight, type OutcomeStore, type PairedMeasurement, type PairedMeasurementAdapter, type PairedMeasurementEvaluation, type ParetoSignificanceGateOptions, type ParsedCodeAgentJsonl, type PartitionByAuthoringModelResult, type PromotionObjective, type PromotionPolicy, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, type Recommendation, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReleaseSummary, type RunCampaignOptions, type RunCandidateExperimentOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunRecordRejection, type ScalarDistribution, type Scenario$1 as Scenario, type SealCandidateBenchmarkSuiteOptions, type SelfImproveBudget, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, type SessionScript, type SkillOptOptimizationMethodConfig, type SkillOptRunnerCommand, type SkillOptTrainerConfig, type SummarizeExecutionOptions, type SurfaceProposer, type TokenUsageInsight, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, summarizeExecution, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
@@ -14,7 +14,7 @@ import {
14
14
  import {
15
15
  analyzeRuns,
16
16
  summarizeExecution
17
- } from "../chunk-NACAGYSY.js";
17
+ } from "../chunk-M4YBQKIJ.js";
18
18
  import {
19
19
  REFERENCE_EQUIVALENCE_INPUT_LIMITS,
20
20
  REFERENCE_EQUIVALENCE_JUDGE_VERSION,
@@ -40,7 +40,7 @@ import {
40
40
  skillOptOptimizationMethod,
41
41
  surfaceContentHash,
42
42
  surfaceHash
43
- } from "../chunk-NKAGIDE2.js";
43
+ } from "../chunk-2QU3YOPR.js";
44
44
  import {
45
45
  campaignSplitDigest,
46
46
  createRunCostLedger,
@@ -48,31 +48,31 @@ import {
48
48
  inMemoryCampaignStorage,
49
49
  resolveRunDir,
50
50
  runCampaign
51
- } from "../chunk-EZJEIH2R.js";
51
+ } from "../chunk-C6LXANRU.js";
52
52
  import {
53
53
  buildDefaultAnalystRegistry,
54
54
  createChatClient
55
- } from "../chunk-DJKY2TSY.js";
55
+ } from "../chunk-BSO5JDQH.js";
56
56
  import "../chunk-HHWE3POT.js";
57
57
  import "../chunk-WGXIEX7P.js";
58
58
  import {
59
59
  FileSystemOutcomeStore,
60
60
  InMemoryOutcomeStore
61
61
  } from "../chunk-3RF76KTD.js";
62
- import "../chunk-NYLOYM6N.js";
62
+ import "../chunk-EG66UGL4.js";
63
63
  import {
64
64
  campaignCellExecutionEvidence,
65
65
  campaignCellJudgeDimensions,
66
66
  campaignCellTaskScore,
67
67
  campaignCellToRunRecord
68
- } from "../chunk-2MKQIFS4.js";
69
- import "../chunk-PBE2LOSS.js";
70
- import "../chunk-VLOATJQ2.js";
71
- import "../chunk-DPUHNQLN.js";
68
+ } from "../chunk-E7QXT7SX.js";
69
+ import "../chunk-SFLLL76A.js";
70
+ import "../chunk-TJVT4QFF.js";
71
+ import "../chunk-7FO3TNPI.js";
72
72
  import {
73
73
  pairedBootstrap
74
- } from "../chunk-MHELPNRP.js";
75
- import "../chunk-WS3NZZQQ.js";
74
+ } from "../chunk-ZHTZ4EYI.js";
75
+ import "../chunk-VCZ5FQYW.js";
76
76
  import "../chunk-VI2UW6B6.js";
77
77
  import {
78
78
  readTaskFailureLabels,
@@ -91,8 +91,9 @@ import "../chunk-PC4UYEBM.js";
91
91
  import {
92
92
  modelHasSnapshot,
93
93
  parseRunRecordSafe
94
- } from "../chunk-2JX3CFMB.js";
94
+ } from "../chunk-56TAVBOK.js";
95
95
  import "../chunk-MA6HLL3S.js";
96
+ import "../chunk-OIUOT4QD.js";
96
97
  import {
97
98
  ValidationError
98
99
  } from "../chunk-ONWEPEDO.js";
@@ -512,7 +513,7 @@ async function shipEvalRunToHosted(tenant, opts, summary, raw, runDir) {
512
513
  surface,
513
514
  cells,
514
515
  compositeMean,
515
- costUsd: campaign.aggregates.totalCostUsd,
516
+ costUsd: campaign.aggregates.cost.totalCostUsd,
516
517
  durationMs
517
518
  };
518
519
  }