@tangle-network/agent-eval 0.128.2 → 0.129.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. package/CHANGELOG.md +265 -0
  2. package/README.md +18 -0
  3. package/dist/analyst/index.d.ts +107 -165
  4. package/dist/analyst/index.js +5 -9
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/belief-state/index.d.ts +2 -19
  7. package/dist/belief-state/index.js +30 -31
  8. package/dist/belief-state/index.js.map +1 -1
  9. package/dist/benchmarks/index.d.ts +5 -8
  10. package/dist/benchmarks/index.js +12 -11
  11. package/dist/builder-eval/index.js +1 -1
  12. package/dist/campaign/index.d.ts +30 -39
  13. package/dist/campaign/index.js +11 -10
  14. package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
  15. package/dist/chunk-2QU3YOPR.js.map +1 -0
  16. package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
  17. package/dist/chunk-3OCR4R5I.js.map +1 -0
  18. package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
  19. package/dist/chunk-56TAVBOK.js.map +1 -0
  20. package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
  21. package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
  22. package/dist/chunk-BSO5JDQH.js.map +1 -0
  23. package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
  24. package/dist/chunk-C6LXANRU.js.map +1 -0
  25. package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
  26. package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
  27. package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
  28. package/dist/chunk-EG66UGL4.js.map +1 -0
  29. package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
  30. package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
  31. package/dist/chunk-G7MGMCZD.js.map +1 -0
  32. package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
  33. package/dist/chunk-H23X7XKK.js.map +1 -0
  34. package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
  35. package/dist/chunk-HPWUNB47.js.map +1 -0
  36. package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
  37. package/dist/chunk-IYCLP2N2.js.map +1 -0
  38. package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
  39. package/dist/chunk-JQSF5DQT.js.map +1 -0
  40. package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
  41. package/dist/chunk-M4YBQKIJ.js.map +1 -0
  42. package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
  43. package/dist/chunk-OIUOT4QD.js +44 -0
  44. package/dist/chunk-OIUOT4QD.js.map +1 -0
  45. package/dist/chunk-OWN5NPMC.js +152 -0
  46. package/dist/chunk-OWN5NPMC.js.map +1 -0
  47. package/dist/chunk-PC5DOSM7.js +579 -0
  48. package/dist/chunk-PC5DOSM7.js.map +1 -0
  49. package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
  50. package/dist/chunk-QB6BDBP2.js.map +1 -0
  51. package/dist/chunk-RXHCETDZ.js +536 -0
  52. package/dist/chunk-RXHCETDZ.js.map +1 -0
  53. package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
  54. package/dist/chunk-SFLLL76A.js.map +1 -0
  55. package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
  56. package/dist/chunk-T6RLYGAD.js.map +1 -0
  57. package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
  58. package/dist/chunk-TJVT4QFF.js.map +1 -0
  59. package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
  60. package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
  61. package/dist/chunk-U4L7JRPZ.js.map +1 -0
  62. package/dist/chunk-U4PHLT2N.js +419 -0
  63. package/dist/chunk-U4PHLT2N.js.map +1 -0
  64. package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
  65. package/dist/chunk-VCZ5FQYW.js.map +1 -0
  66. package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
  67. package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
  68. package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
  69. package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
  70. package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
  71. package/dist/chunk-ZHTZ4EYI.js.map +1 -0
  72. package/dist/cli.js +6 -5
  73. package/dist/cli.js.map +1 -1
  74. package/dist/contract/index.d.ts +47 -87
  75. package/dist/contract/index.js +14 -13
  76. package/dist/contract/index.js.map +1 -1
  77. package/dist/control.js +3 -2
  78. package/dist/fuzz.js +3 -2
  79. package/dist/fuzz.js.map +1 -1
  80. package/dist/index.d.ts +659 -203
  81. package/dist/index.js +145 -117
  82. package/dist/index.js.map +1 -1
  83. package/dist/meta-eval/index.js +2 -2
  84. package/dist/multishot/index.d.ts +3 -4
  85. package/dist/multishot/index.js.map +1 -1
  86. package/dist/openapi.json +1 -1
  87. package/dist/pipelines/index.js +5 -5
  88. package/dist/reporting.d.ts +14 -0
  89. package/dist/reporting.js +7 -6
  90. package/dist/rl.d.ts +652 -82
  91. package/dist/rl.js +415 -171
  92. package/dist/rl.js.map +1 -1
  93. package/dist/rollout/index.d.ts +1071 -32
  94. package/dist/rollout/index.js +68 -10
  95. package/dist/run-campaign-OJJ7CZF4.js +18 -0
  96. package/dist/supervisor-run/index.d.ts +114 -4
  97. package/dist/supervisor-run/index.js +4 -3
  98. package/dist/traces.d.ts +1 -1
  99. package/dist/traces.js +6 -5
  100. package/dist/wire/index.d.ts +10 -11
  101. package/dist/wire/index.js +3 -3
  102. package/docs/feature-guide.md +1 -1
  103. package/docs/rollout.md +116 -2
  104. package/package.json +4 -4
  105. package/dist/chunk-2JX3CFMB.js.map +0 -1
  106. package/dist/chunk-DJKY2TSY.js.map +0 -1
  107. package/dist/chunk-EJGRPCO3.js.map +0 -1
  108. package/dist/chunk-EOSZT7PL.js.map +0 -1
  109. package/dist/chunk-EZJEIH2R.js.map +0 -1
  110. package/dist/chunk-IHQDPH7D.js.map +0 -1
  111. package/dist/chunk-MHELPNRP.js.map +0 -1
  112. package/dist/chunk-NACAGYSY.js.map +0 -1
  113. package/dist/chunk-NKAGIDE2.js.map +0 -1
  114. package/dist/chunk-NYLOYM6N.js.map +0 -1
  115. package/dist/chunk-PBE2LOSS.js.map +0 -1
  116. package/dist/chunk-TT4KNT67.js +0 -124
  117. package/dist/chunk-TT4KNT67.js.map +0 -1
  118. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  119. package/dist/chunk-UWZZKKU7.js +0 -237
  120. package/dist/chunk-UWZZKKU7.js.map +0 -1
  121. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  122. package/dist/chunk-VGRCHJON.js.map +0 -1
  123. package/dist/chunk-VLOATJQ2.js.map +0 -1
  124. package/dist/chunk-VZSRQ272.js.map +0 -1
  125. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  126. package/dist/chunk-XDWDC2MP.js.map +0 -1
  127. package/dist/chunk-XPRT64IE.js.map +0 -1
  128. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  129. /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
  130. /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
  131. /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
  132. /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
  133. /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
  134. /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
  135. /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
  136. /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
  137. /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
@@ -137,9 +137,8 @@ interface CostLedgerSummary {
137
137
  * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
138
138
  * )
139
139
  *
140
- * This is THE llm-calling seam for agent-eval primitives that need structured
141
- * output (semantic concept judge, reviewer directives, critic scores). Primitives
142
- * that need free-form text use `callLlm` and parse output themselves.
140
+ * `createChatClient` wraps this implementation for provider-neutral package
141
+ * entry points. Direct callers can use `callLlm` or `callLlmJson`.
143
142
  */
144
143
 
145
144
  interface LlmUsage {
@@ -290,7 +289,7 @@ interface DispatchContext {
290
289
  /** The canonical judge verdict shape — one declaration, shared by campaign
291
290
  * judges and the multishot judge runner (which re-exports this type).
292
291
  *
293
- * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the legacy
292
+ * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
294
293
  * multishot runner emits 0-10. Cross-scale comparison must go through
295
294
  * `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
296
295
  * promotion-policy) — never renormalize a producer's values in place, as
@@ -470,8 +469,6 @@ interface CampaignAggregates {
470
469
  byScenario: Record<string, ScenarioAggregate>;
471
470
  /** Canonical campaign accounting, including worker and judge calls. */
472
471
  cost: CostLedgerSummary;
473
- /** Compatibility alias of `cost.totalCostUsd`. */
474
- totalCostUsd: number;
475
472
  /** Cells whose dispatch completed, including cells whose later judge failed. */
476
473
  cellsExecuted: number;
477
474
  cellsSkipped: number;
@@ -482,7 +479,7 @@ interface CampaignAggregates {
482
479
  cellsDispatchFailed?: number;
483
480
  /** Present on results that record failure stages. */
484
481
  cellsJudgeFailed?: number;
485
- /** Legacy failures whose stage was not recorded. */
482
+ /** Failures whose stage could not be classified. */
486
483
  cellsUnclassifiedFailed?: number;
487
484
  }
488
485
  interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
@@ -734,7 +731,7 @@ interface CampaignStorage {
734
731
  write(path: string, content: string | Uint8Array): void;
735
732
  /** Append only when the current UTF-8 byte length matches `expectedBytes`.
736
733
  * Returns the new length, or undefined when another writer won. */
737
- append?(path: string, content: string, expectedBytes: number): number | undefined;
734
+ append(path: string, content: string, expectedBytes: number): number | undefined;
738
735
  }
739
736
 
740
737
  interface BenchmarkRunOptions<TPayload = unknown, TArtifact = string> {
@@ -16,24 +16,25 @@ import {
16
16
  routing_exports,
17
17
  runBenchmarkAdapter,
18
18
  summarizeBenchmarkCampaign
19
- } from "../chunk-XPRT64IE.js";
20
- import "../chunk-UB2LOJ6Q.js";
21
- import "../chunk-NKAGIDE2.js";
22
- import "../chunk-EZJEIH2R.js";
19
+ } from "../chunk-IYCLP2N2.js";
20
+ import "../chunk-QB6BDBP2.js";
21
+ import "../chunk-2QU3YOPR.js";
22
+ import "../chunk-C6LXANRU.js";
23
23
  import "../chunk-WGXIEX7P.js";
24
- import "../chunk-NYLOYM6N.js";
25
- import "../chunk-2MKQIFS4.js";
26
- import "../chunk-PBE2LOSS.js";
27
- import "../chunk-DPUHNQLN.js";
28
- import "../chunk-MHELPNRP.js";
29
- import "../chunk-WS3NZZQQ.js";
24
+ import "../chunk-EG66UGL4.js";
25
+ import "../chunk-E7QXT7SX.js";
26
+ import "../chunk-SFLLL76A.js";
27
+ import "../chunk-7FO3TNPI.js";
28
+ import "../chunk-ZHTZ4EYI.js";
29
+ import "../chunk-VCZ5FQYW.js";
30
30
  import "../chunk-VI2UW6B6.js";
31
31
  import "../chunk-5DTSBUL2.js";
32
32
  import "../chunk-GGE4NNQT.js";
33
33
  import "../chunk-P6FYH6K4.js";
34
34
  import "../chunk-PC4UYEBM.js";
35
- import "../chunk-2JX3CFMB.js";
35
+ import "../chunk-56TAVBOK.js";
36
36
  import "../chunk-MA6HLL3S.js";
37
+ import "../chunk-OIUOT4QD.js";
37
38
  import "../chunk-ONWEPEDO.js";
38
39
  import "../chunk-K4DBDHLK.js";
39
40
  import "../chunk-PZ5AY32C.js";
@@ -5,7 +5,7 @@ import {
5
5
  import {
6
6
  pearsonR,
7
7
  spearmanR
8
- } from "../chunk-MHELPNRP.js";
8
+ } from "../chunk-ZHTZ4EYI.js";
9
9
  import {
10
10
  judgeSpans
11
11
  } from "../chunk-ZET2UAYW.js";
@@ -249,9 +249,8 @@ type CostLedgerHandle = Pick<CostLedger, Exclude<keyof CostLedger, 'listPending'
249
249
  * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
250
250
  * )
251
251
  *
252
- * This is THE llm-calling seam for agent-eval primitives that need structured
253
- * output (semantic concept judge, reviewer directives, critic scores). Primitives
254
- * that need free-form text use `callLlm` and parse output themselves.
252
+ * `createChatClient` wraps this implementation for provider-neutral package
253
+ * entry points. Direct callers can use `callLlm` or `callLlmJson`.
255
254
  */
256
255
 
257
256
  interface LlmMessage {
@@ -335,42 +334,27 @@ interface LlmCallResult {
335
334
  type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
336
335
 
337
336
  /**
338
- * ChatClient the single LLM abstraction analysts call.
339
- *
340
- * agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
341
- * graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
342
- * mixed patterns force every analyst author to pick a transport, which
343
- * couples analyst code to runtime concerns (cli-bridge vs router vs
344
- * sandbox-sdk) it shouldn't know about.
345
- *
346
- * `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
347
- * The operator decides at the registry boundary which transport binds
348
- * to it. Analyst code stays transport-agnostic; swapping production
349
- * (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
350
- * line factory call.
351
- *
352
- * Designed to coexist: existing `LlmClient` callers and existing
353
- * `TCloud`-based judges keep working untouched. New analyst code uses
354
- * `ChatClient`. When old call sites migrate, they pick up budgeting,
355
- * cancellation, and unified telemetry for free.
337
+ * Provider-neutral chat contract for every model call made by agent-eval.
338
+ *
339
+ * Callers choose the transport at the package boundary with `createChatClient`.
340
+ * Evaluation code receives canonical requests and results without importing a
341
+ * provider SDK.
356
342
  */
357
343
 
358
344
  /**
359
- * Unified chat interface. Mirrors LlmCallRequest/Result so the OpenAI-
360
- * compatible mental model stays. Two methods: a one-shot `chat()` and
361
- * an `streamChat()` for future agentic loops (not yet exposed).
345
+ * Unified chat interface using the package's canonical LLM request and result.
362
346
  */
363
347
  interface ChatClient {
364
- /** Display name of the bound transport included in telemetry. */
348
+ /** Display name of the bound transport, included in telemetry. */
365
349
  readonly transport: ChatTransport;
366
- /** Default model when caller omits — operators bind this per environment. */
350
+ /** Default model when the caller omits one. */
367
351
  readonly defaultModel?: string;
368
352
  /** Total provider attempts this transport can make for one chat call. */
369
353
  readonly maximumAttempts?: number;
370
354
  /** Implementations must enforce `req.maxTokens` when it is present. */
371
355
  chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
372
356
  }
373
- type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
357
+ type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'custom' | 'mock';
374
358
  interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
375
359
  /** Optional — falls back to ChatClient.defaultModel. */
376
360
  model?: string;
@@ -738,7 +722,7 @@ interface JudgeConfig<TArtifact, TScenario extends Scenario = Scenario> {
738
722
  /** The canonical judge verdict shape — one declaration, shared by campaign
739
723
  * judges and the multishot judge runner (which re-exports this type).
740
724
  *
741
- * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the legacy
725
+ * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
742
726
  * multishot runner emits 0-10. Cross-scale comparison must go through
743
727
  * `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
744
728
  * promotion-policy) — never renormalize a producer's values in place, as
@@ -961,9 +945,6 @@ interface SurfaceProposer<TFindings = unknown> {
961
945
  reason?: string;
962
946
  };
963
947
  }
964
- /** Optional vocabulary alias. The loop is the optimizer; this object is the
965
- * proposer inside that loop. */
966
- type OptimizationProposer<TFindings = unknown> = SurfaceProposer<TFindings>;
967
948
  interface OptimizerConfigBase {
968
949
  populationSize: number;
969
950
  maxGenerations: number;
@@ -1243,8 +1224,6 @@ interface CampaignAggregates {
1243
1224
  byScenario: Record<string, ScenarioAggregate>;
1244
1225
  /** Canonical campaign accounting, including worker and judge calls. */
1245
1226
  cost: CostLedgerSummary;
1246
- /** Compatibility alias of `cost.totalCostUsd`. */
1247
- totalCostUsd: number;
1248
1227
  /** Cells whose dispatch completed, including cells whose later judge failed. */
1249
1228
  cellsExecuted: number;
1250
1229
  cellsSkipped: number;
@@ -1255,7 +1234,7 @@ interface CampaignAggregates {
1255
1234
  cellsDispatchFailed?: number;
1256
1235
  /** Present on results that record failure stages. */
1257
1236
  cellsJudgeFailed?: number;
1258
- /** Legacy failures whose stage was not recorded. */
1237
+ /** Failures whose stage could not be classified. */
1259
1238
  cellsUnclassifiedFailed?: number;
1260
1239
  }
1261
1240
  interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
@@ -2313,7 +2292,7 @@ interface CampaignStorage {
2313
2292
  write(path: string, content: string | Uint8Array): void;
2314
2293
  /** Append only when the current UTF-8 byte length matches `expectedBytes`.
2315
2294
  * Returns the new length, or undefined when another writer won. */
2316
- append?(path: string, content: string, expectedBytes: number): number | undefined;
2295
+ append(path: string, content: string, expectedBytes: number): number | undefined;
2317
2296
  }
2318
2297
  /** Node-filesystem storage — the default. Lazily requires `node:fs` so the
2319
2298
  * module imports cleanly in non-Node runtimes (where the caller passes
@@ -2455,9 +2434,9 @@ interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
2455
2434
  }) => string | undefined;
2456
2435
  }
2457
2436
  /** Durable `<cell>/failure-receipt.json` written before a failed cell can
2458
- * trigger campaign-wide cancellation. The cell keeps its dispatch-only usage
2459
- * fields for compatibility; `cost` covers every settled agent and judge call
2460
- * attributed to this exact run attempt. */
2437
+ * trigger campaign-wide cancellation. The cell records dispatch measurements;
2438
+ * `cost` covers every settled agent and judge call attributed to this exact run
2439
+ * attempt. */
2461
2440
  interface CampaignCellFailureReceipt<TArtifact = unknown> {
2462
2441
  schemaVersion: 1;
2463
2442
  runAttemptId: string;
@@ -3110,6 +3089,18 @@ interface VerifiableRewardExtractionOptions {
3110
3089
  * doesn't report one. Default `0.7`.
3111
3090
  */
3112
3091
  judgeConfidenceFloor?: number;
3092
+ /**
3093
+ * Whether the anti-Goodhart realness gate applies. Default `true`, and the
3094
+ * default is the one every training path must keep.
3095
+ *
3096
+ * Set `false` ONLY for detection and analysis. `rl/reward-hacking.ts` does,
3097
+ * for the same reason it reads `observedScore` for its proxy: it measures the
3098
+ * DIVERGENCE between the judge signal and the deterministic one, and a
3099
+ * deterministic reward that another gate already forced to 0 manufactures
3100
+ * exactly that divergence on exactly the gamed population. The detector would
3101
+ * then be re-reporting a verdict it was supposed to reach independently.
3102
+ */
3103
+ applyRealnessGate?: boolean;
3113
3104
  }
3114
3105
 
3115
3106
  /**
@@ -6387,4 +6378,4 @@ declare function verifyCodeSurface(surface: CodeSurface, worktreeDir?: string):
6387
6378
  * identity against the checkout at `worktreeRef`. */
6388
6379
  declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
6389
6380
 
6390
- export { type AnalystArtifact, type AnalystScenario, type AnalyzeCrossSurfaceInteractionsInput, type AxisEvidence, type AxisVerdict, type BuildAnalystSurfaceDispatchOptions, type BuildEvidenceVectorOptions, type BuildLoopProvenanceArgs, type CampaignAggregates, type CampaignArtifactWriter, type CampaignBreakdown, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignRunPlan, type CampaignRunPlanCell, type CampaignScenarioIdentity, type CampaignStorage, type CampaignTokenUsage, type CampaignTraceWriter, type CodeSurface, type CodeSurfaceVerification, type CompareOptimizationMethodsOptions, type ComparisonCost, type ComponentSurface, type CostLedgerHandle, type CrossSurfaceAdditionDecision, type CrossSurfaceAdditionRejectionReason, type CrossSurfaceAttemptCompleteness, type CrossSurfaceBestSingleSelection, type CrossSurfaceBootstrapPolicy, type CrossSurfaceCandidate, type CrossSurfaceCandidateComparison, type CrossSurfaceCandidateEvidence, type CrossSurfaceCandidateOutcome, type CrossSurfaceCandidateSummary, type CrossSurfaceComponent, type CrossSurfaceComponentEvidence, type CrossSurfaceCompositionStep, type CrossSurfaceDistribution, type CrossSurfaceEligibility, type CrossSurfaceEvidenceBreakdown, type CrossSurfaceIneligibilityReason, type CrossSurfaceInteractionAwareSelection, type CrossSurfaceInteractionEffect, type CrossSurfaceInteractionPath, type CrossSurfaceInteractionReport, type CrossSurfaceInteractionTask, type CrossSurfaceNaiveStackSelection, type CrossSurfacePairCompatibility, type CrossSurfacePairEvidence, type CrossSurfacePairIncompatibilityReason, type CrossSurfacePairwiseEntry, type CrossSurfaceRankedSingle, type CrossSurfaceRelativeCost, type CrossSurfaceSelectionPolicy, type CrossSurfaceSelections, type CrossSurfaceTaskRow, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DimensionRegression, type DiscriminationScore, type DispatchContext, type DispatchFn, type EmitLoopProvenanceArgs, type EmitLoopProvenanceResult, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type EvidenceVector, type ExternalOptimizationExample, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, type FailureModeRecallJudgeOptions, FileSearchLedger, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type GitWorktreeAdapterOptions, type HeldOutGateOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type JudgeAggregate, type JudgeConfig, type JudgeDimension, type JudgeScore, type LabelTrust, type LabeledScenarioRecord, type LabeledScenarioSampleArgs, type LabeledScenarioSource, type LabeledScenarioStore, LabeledScenarioStoreError, type LabeledScenarioWrite, type LlmJudgeDimension, type LlmJudgeOptions, type LoadEvalFixtureScenariosOptions, type LoopProvenanceArgsFromResult, type LoopProvenanceBackend, type LoopProvenanceCandidate, type LoopProvenanceEvidence, type LoopProvenanceOptimizationMethod, type LoopProvenanceRecord, type MutableSurface, type NeutralizationGateOptions, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OpenAutoPrOptions, type OpenAutoPrResult, type OpenSearchLedgerOptions, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodPairwise, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationMethodRunOptions, type OptimizationMethodScore, type OptimizationPackageSource, type OptimizationProposer, type OptimizationTokenUsage, type OptimizerConfig, type OptimizerModelBudget, type PairedHoldout, type ParetoParent, type ParetoSignificanceGateOptions, type PendingCostCallView, type PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PowerPreflight, type PowerPreflightOptions, type PremeasuredOptimizationBaseline, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, type PromotionObjective, type PromotionPolicy, type ProposalTrackContext, type ProposeContext, type ProposedCandidate, type RedactionStatus, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceScenario, type RolloutArgumentDiff, type RolloutArgumentDiffOptions, type RolloutCall, type RunCampaignOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunOptimizationOptions, type RunOptimizationResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, SEARCH_LEDGER_SCHEMA, type Scenario, type ScenarioAggregate, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type ScoredRollout, type ScoredSurfaceOutcome, type SearchAccountingAudit, type SearchArtifactRef, type SearchAttemptAccounting, type SearchCandidateDecidedEvent, type SearchCandidateLineage, type SearchCandidateRegisteredEvent, type SearchCandidateSlot, type SearchCandidateSlotClosedEvent, type SearchCandidateSurface, type SearchCompletedEvent, type SearchCostAccounting, type SearchFailureReason, type SearchLedger, type SearchLedgerAppendResult, SearchLedgerConflictError, type SearchLedgerEntry, SearchLedgerError, type SearchLedgerEvent, type SearchLedgerHash, SearchLedgerIntegrityError, type SearchLedgerReplay, type SearchModelIdentity, type SearchOperationKind, type SearchOperationRecordedEvent, type SearchPlan, type SearchPlannedEvent, type SearchPlannedOperation, type SearchPlannedTask, type SearchSourceRef, type SearchSurfaceEffect, type SearchSurfaceEvidence, type SearchSurfaceKind, type SearchTaskAttemptedEvent, type SearchTaskOutcome, type SearchTokenAccounting, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SessionScript, type SingleRunLock, type SingleRunLockOptions, type SkillOptOptimizationMethodConfig, type SkillOptRunnerCommand, type SkillOptTrainerConfig, type SurfaceProposer, type TraceSpan, type TransientFailureOptions, type UngroundedLiteralReport, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, acquireSingleRunLock, analyzeCrossSurfaceInteractions, assertCampaignDesign, assertCampaignSplitIdentity, assertCodeSurfaceIdentity, assertComponentSurface, buildAnalystSurfaceDispatch, buildEvidenceVector, buildLoopProvenanceRecord, campaignBreakdown, campaignMeanComposite, campaignMeasurementDigest, campaignScenarioIdentity, campaignSplitDigest, campaignSplitDigestFromIdentities, canonicalDigest, classifyUngroundedLiterals, codeSurfaceIdentityMaterial, compareOptimizationMethods, compareRankKeys, componentSurfaceIdentityMaterial, composeGate, costFromLedgerSummary, createReferenceEquivalenceJudge, createRunCostLedger, defaultProductionGate, detectScale, dimensionRegressions, discoverEvalFixtures, emitLoopProvenance, externalTextOptimizationMethod, failureModeRecallJudge, fsCampaignStorage, gepaOptimizationMethod, gitWorktreeAdapter, heldOutGate, heldoutSignificance, inMemoryCampaignStorage, isProposedCandidate, isTransientTransportFailure, labelTrustRank, llmJudge, loadEvalFixture, loadEvalFixtureScenarios, loopProvenanceArgsFromResult, loopProvenanceSpans, makePlaybackDispatch, neutralizationGate, neutralizeText, openAutoPr, openSearchLedger, optimizationTokenUsageFromSummary, pairHoldout, paretoPolicy, paretoSignificanceGate, planCampaignRun, planEvalFixtureRun, powerPreflight, provenanceRecordPath, provenanceSpansPath, renderScoreboardMarkdown, renderSurfaceDiff, resolveRunDir, resolveWorktreePath, rolloutArgumentDiff, runCampaign, runEval, runImprovementLoop, runOptimization, runProfileMatrix, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptOptimizationMethod, surfaceContentHash, surfaceHash, tangleTracesRoot, userStoryScoreboard, validateSearchLedgerEvent, verifyCodeSurface, verifyLoopProvenanceRecord };
6381
+ export { type AnalystArtifact, type AnalystScenario, type AnalyzeCrossSurfaceInteractionsInput, type AxisEvidence, type AxisVerdict, type BuildAnalystSurfaceDispatchOptions, type BuildEvidenceVectorOptions, type BuildLoopProvenanceArgs, type CampaignAggregates, type CampaignArtifactWriter, type CampaignBreakdown, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignRunPlan, type CampaignRunPlanCell, type CampaignScenarioIdentity, type CampaignStorage, type CampaignTokenUsage, type CampaignTraceWriter, type CodeSurface, type CodeSurfaceVerification, type CompareOptimizationMethodsOptions, type ComparisonCost, type ComponentSurface, type CostLedgerHandle, type CrossSurfaceAdditionDecision, type CrossSurfaceAdditionRejectionReason, type CrossSurfaceAttemptCompleteness, type CrossSurfaceBestSingleSelection, type CrossSurfaceBootstrapPolicy, type CrossSurfaceCandidate, type CrossSurfaceCandidateComparison, type CrossSurfaceCandidateEvidence, type CrossSurfaceCandidateOutcome, type CrossSurfaceCandidateSummary, type CrossSurfaceComponent, type CrossSurfaceComponentEvidence, type CrossSurfaceCompositionStep, type CrossSurfaceDistribution, type CrossSurfaceEligibility, type CrossSurfaceEvidenceBreakdown, type CrossSurfaceIneligibilityReason, type CrossSurfaceInteractionAwareSelection, type CrossSurfaceInteractionEffect, type CrossSurfaceInteractionPath, type CrossSurfaceInteractionReport, type CrossSurfaceInteractionTask, type CrossSurfaceNaiveStackSelection, type CrossSurfacePairCompatibility, type CrossSurfacePairEvidence, type CrossSurfacePairIncompatibilityReason, type CrossSurfacePairwiseEntry, type CrossSurfaceRankedSingle, type CrossSurfaceRelativeCost, type CrossSurfaceSelectionPolicy, type CrossSurfaceSelections, type CrossSurfaceTaskRow, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DimensionRegression, type DiscriminationScore, type DispatchContext, type DispatchFn, type EmitLoopProvenanceArgs, type EmitLoopProvenanceResult, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type EvidenceVector, type ExternalOptimizationExample, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, type FailureModeRecallJudgeOptions, FileSearchLedger, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type GitWorktreeAdapterOptions, type HeldOutGateOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type JudgeAggregate, type JudgeConfig, type JudgeDimension, type JudgeScore, type LabelTrust, type LabeledScenarioRecord, type LabeledScenarioSampleArgs, type LabeledScenarioSource, type LabeledScenarioStore, LabeledScenarioStoreError, type LabeledScenarioWrite, type LlmJudgeDimension, type LlmJudgeOptions, type LoadEvalFixtureScenariosOptions, type LoopProvenanceArgsFromResult, type LoopProvenanceBackend, type LoopProvenanceCandidate, type LoopProvenanceEvidence, type LoopProvenanceOptimizationMethod, type LoopProvenanceRecord, type MutableSurface, type NeutralizationGateOptions, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OpenAutoPrOptions, type OpenAutoPrResult, type OpenSearchLedgerOptions, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodPairwise, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationMethodRunOptions, type OptimizationMethodScore, type OptimizationPackageSource, type OptimizationTokenUsage, type OptimizerConfig, type OptimizerModelBudget, type PairedHoldout, type ParetoParent, type ParetoSignificanceGateOptions, type PendingCostCallView, type PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PowerPreflight, type PowerPreflightOptions, type PremeasuredOptimizationBaseline, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, type PromotionObjective, type PromotionPolicy, type ProposalTrackContext, type ProposeContext, type ProposedCandidate, type RedactionStatus, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceScenario, type RolloutArgumentDiff, type RolloutArgumentDiffOptions, type RolloutCall, type RunCampaignOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunOptimizationOptions, type RunOptimizationResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, SEARCH_LEDGER_SCHEMA, type Scenario, type ScenarioAggregate, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type ScoredRollout, type ScoredSurfaceOutcome, type SearchAccountingAudit, type SearchArtifactRef, type SearchAttemptAccounting, type SearchCandidateDecidedEvent, type SearchCandidateLineage, type SearchCandidateRegisteredEvent, type SearchCandidateSlot, type SearchCandidateSlotClosedEvent, type SearchCandidateSurface, type SearchCompletedEvent, type SearchCostAccounting, type SearchFailureReason, type SearchLedger, type SearchLedgerAppendResult, SearchLedgerConflictError, type SearchLedgerEntry, SearchLedgerError, type SearchLedgerEvent, type SearchLedgerHash, SearchLedgerIntegrityError, type SearchLedgerReplay, type SearchModelIdentity, type SearchOperationKind, type SearchOperationRecordedEvent, type SearchPlan, type SearchPlannedEvent, type SearchPlannedOperation, type SearchPlannedTask, type SearchSourceRef, type SearchSurfaceEffect, type SearchSurfaceEvidence, type SearchSurfaceKind, type SearchTaskAttemptedEvent, type SearchTaskOutcome, type SearchTokenAccounting, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SessionScript, type SingleRunLock, type SingleRunLockOptions, type SkillOptOptimizationMethodConfig, type SkillOptRunnerCommand, type SkillOptTrainerConfig, type SurfaceProposer, type TraceSpan, type TransientFailureOptions, type UngroundedLiteralReport, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, acquireSingleRunLock, analyzeCrossSurfaceInteractions, assertCampaignDesign, assertCampaignSplitIdentity, assertCodeSurfaceIdentity, assertComponentSurface, buildAnalystSurfaceDispatch, buildEvidenceVector, buildLoopProvenanceRecord, campaignBreakdown, campaignMeanComposite, campaignMeasurementDigest, campaignScenarioIdentity, campaignSplitDigest, campaignSplitDigestFromIdentities, canonicalDigest, classifyUngroundedLiterals, codeSurfaceIdentityMaterial, compareOptimizationMethods, compareRankKeys, componentSurfaceIdentityMaterial, composeGate, costFromLedgerSummary, createReferenceEquivalenceJudge, createRunCostLedger, defaultProductionGate, detectScale, dimensionRegressions, discoverEvalFixtures, emitLoopProvenance, externalTextOptimizationMethod, failureModeRecallJudge, fsCampaignStorage, gepaOptimizationMethod, gitWorktreeAdapter, heldOutGate, heldoutSignificance, inMemoryCampaignStorage, isProposedCandidate, isTransientTransportFailure, labelTrustRank, llmJudge, loadEvalFixture, loadEvalFixtureScenarios, loopProvenanceArgsFromResult, loopProvenanceSpans, makePlaybackDispatch, neutralizationGate, neutralizeText, openAutoPr, openSearchLedger, optimizationTokenUsageFromSummary, pairHoldout, paretoPolicy, paretoSignificanceGate, planCampaignRun, planEvalFixtureRun, powerPreflight, provenanceRecordPath, provenanceSpansPath, renderScoreboardMarkdown, renderSurfaceDiff, resolveRunDir, resolveWorktreePath, rolloutArgumentDiff, runCampaign, runEval, runImprovementLoop, runOptimization, runProfileMatrix, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptOptimizationMethod, surfaceContentHash, surfaceHash, tangleTracesRoot, userStoryScoreboard, validateSearchLedgerEvent, verifyCodeSurface, verifyLoopProvenanceRecord };
@@ -32,7 +32,7 @@ import {
32
32
  userStoryScoreboard,
33
33
  validateSearchLedgerEvent,
34
34
  verifyCodeSurface
35
- } from "../chunk-UB2LOJ6Q.js";
35
+ } from "../chunk-QB6BDBP2.js";
36
36
  import {
37
37
  acquireSingleRunLock,
38
38
  assertCodeSurfaceIdentity,
@@ -79,7 +79,7 @@ import {
79
79
  surfaceContentHash,
80
80
  surfaceHash,
81
81
  verifyLoopProvenanceRecord
82
- } from "../chunk-NKAGIDE2.js";
82
+ } from "../chunk-2QU3YOPR.js";
83
83
  import {
84
84
  SearchLedgerConflictError,
85
85
  SearchLedgerError,
@@ -96,21 +96,22 @@ import {
96
96
  resolveRunDir,
97
97
  runCampaign,
98
98
  tangleTracesRoot
99
- } from "../chunk-EZJEIH2R.js";
99
+ } from "../chunk-C6LXANRU.js";
100
100
  import "../chunk-WGXIEX7P.js";
101
- import "../chunk-NYLOYM6N.js";
102
- import "../chunk-2MKQIFS4.js";
103
- import "../chunk-PBE2LOSS.js";
104
- import "../chunk-DPUHNQLN.js";
105
- import "../chunk-MHELPNRP.js";
106
- import "../chunk-WS3NZZQQ.js";
101
+ import "../chunk-EG66UGL4.js";
102
+ import "../chunk-E7QXT7SX.js";
103
+ import "../chunk-SFLLL76A.js";
104
+ import "../chunk-7FO3TNPI.js";
105
+ import "../chunk-ZHTZ4EYI.js";
106
+ import "../chunk-VCZ5FQYW.js";
107
107
  import "../chunk-VI2UW6B6.js";
108
108
  import "../chunk-5DTSBUL2.js";
109
109
  import "../chunk-GGE4NNQT.js";
110
110
  import "../chunk-P6FYH6K4.js";
111
111
  import "../chunk-PC4UYEBM.js";
112
- import "../chunk-2JX3CFMB.js";
112
+ import "../chunk-56TAVBOK.js";
113
113
  import "../chunk-MA6HLL3S.js";
114
+ import "../chunk-OIUOT4QD.js";
114
115
  import "../chunk-ONWEPEDO.js";
115
116
  import "../chunk-K4DBDHLK.js";
116
117
  import "../chunk-PZ5AY32C.js";