@tangle-network/agent-eval 0.94.0 → 0.95.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/CHANGELOG.md +32 -0
  2. package/README.md +44 -30
  3. package/dist/adapters/http.d.ts +8 -7
  4. package/dist/adapters/http.js.map +1 -1
  5. package/dist/adapters/langchain.d.ts +3 -2
  6. package/dist/adapters/otel.d.ts +5 -4
  7. package/dist/analyst/index.d.ts +11 -31
  8. package/dist/analyst/index.js +5 -65
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/{analyze-runs-B6Ljo_dI.d.ts → analyze-runs-DtT6F_6T.d.ts} +3 -3
  11. package/dist/belief-state/index.d.ts +4 -3
  12. package/dist/benchmarks/index.d.ts +3 -2
  13. package/dist/campaign/index.d.ts +727 -616
  14. package/dist/campaign/index.js +1863 -1316
  15. package/dist/campaign/index.js.map +1 -1
  16. package/dist/{chunk-2K6UUZ7P.js → chunk-2T4EZACH.js} +1 -1
  17. package/dist/chunk-2T4EZACH.js.map +1 -0
  18. package/dist/{chunk-CTBHKLEU.js → chunk-77T4STFI.js} +59 -86
  19. package/dist/chunk-77T4STFI.js.map +1 -0
  20. package/dist/{chunk-EGPMSBEZ.js → chunk-7QTQKIDD.js} +178 -177
  21. package/dist/chunk-7QTQKIDD.js.map +1 -0
  22. package/dist/{chunk-MIFZUPEK.js → chunk-AQ5WQAIV.js} +21 -6
  23. package/dist/chunk-AQ5WQAIV.js.map +1 -0
  24. package/dist/chunk-DJWX3GVS.js +81 -0
  25. package/dist/chunk-DJWX3GVS.js.map +1 -0
  26. package/dist/{chunk-S6OZEZQK.js → chunk-HMA63UEO.js} +37 -9
  27. package/dist/{chunk-S6OZEZQK.js.map → chunk-HMA63UEO.js.map} +1 -1
  28. package/dist/{chunk-TBDR6PAI.js → chunk-IZCEK2HR.js} +2 -2
  29. package/dist/{chunk-SD2YFWQQ.js → chunk-KKWJD5E6.js} +20 -20
  30. package/dist/chunk-KKWJD5E6.js.map +1 -0
  31. package/dist/{chunk-KWRRMR3J.js → chunk-LO6IOIJ2.js} +10 -10
  32. package/dist/chunk-LO6IOIJ2.js.map +1 -0
  33. package/dist/{chunk-E4GH6USR.js → chunk-NZEQVRH5.js} +2 -2
  34. package/dist/chunk-NZEQVRH5.js.map +1 -0
  35. package/dist/{chunk-MPQWFX6Y.js → chunk-PSWWQXHF.js} +13 -88
  36. package/dist/chunk-PSWWQXHF.js.map +1 -0
  37. package/dist/{chunk-Q5LIB7BC.js → chunk-S4SYLDFX.js} +2 -2
  38. package/dist/chunk-S4SYLDFX.js.map +1 -0
  39. package/dist/{chunk-KW53MSA5.js → chunk-X74V6ESX.js} +2 -2
  40. package/dist/{chunk-QMUEXQJS.js → chunk-YBIGNSCZ.js} +81 -4
  41. package/dist/chunk-YBIGNSCZ.js.map +1 -0
  42. package/dist/{chunk-2KNZHH3P.js → chunk-Z6L6YSU6.js} +2 -2
  43. package/dist/{code-agent-session-BO8nCnv3.d.ts → code-agent-session-CPHRCb4-.d.ts} +1 -1
  44. package/dist/contract/index.d.ts +91 -43
  45. package/dist/contract/index.js +127 -17
  46. package/dist/contract/index.js.map +1 -1
  47. package/dist/{control-D6qwHXIR.d.ts → control-Doncu-B_.d.ts} +2 -2
  48. package/dist/control.d.ts +3 -2
  49. package/dist/control.js +2 -2
  50. package/dist/{corpus-B8A4BDR3.d.ts → corpus-D4YW9UoJ.d.ts} +1 -1
  51. package/dist/{default-registry-6dhErQbs.d.ts → default-registry-GyE8X5SP.d.ts} +3 -3
  52. package/dist/diagnose.d.ts +4 -3
  53. package/dist/diagnose.js +1 -1
  54. package/dist/{run-improvement-loop-DBahB8Ax.d.ts → gepa-C1NCIZ9o.d.ts} +117 -130
  55. package/dist/hosted/index.d.ts +5 -4
  56. package/dist/{index-Bx3gZ8xl.d.ts → index-_Y4oNOOb.d.ts} +1 -1
  57. package/dist/index.d.ts +76 -81
  58. package/dist/index.js +66 -31
  59. package/dist/index.js.map +1 -1
  60. package/dist/{insight-report-DWl3z9tl.d.ts → insight-report-BnRjTibG.d.ts} +1 -1
  61. package/dist/{kind-factory-0BhLSI27.d.ts → kind-factory-X3eDYbKn.d.ts} +2 -3
  62. package/dist/matrix/index.d.ts +1 -1
  63. package/dist/meta-eval/index.d.ts +3 -2
  64. package/dist/multishot/index.d.ts +4 -4
  65. package/dist/multishot/index.js.map +1 -1
  66. package/dist/openapi.json +1 -1
  67. package/dist/{pre-registration-mAnCugl9.d.ts → pre-registration-nfUdc9EQ.d.ts} +2 -42
  68. package/dist/{provenance-P-bCL2Fo.d.ts → provenance-CncDq9qE.d.ts} +26 -41
  69. package/dist/{release-report-BEbWmVYj.d.ts → release-report-pidWUMZ2.d.ts} +2 -2
  70. package/dist/reporting.d.ts +5 -4
  71. package/dist/{researcher-B0C2_fVO.d.ts → researcher-Jr8ME1dZ.d.ts} +2 -2
  72. package/dist/rl.d.ts +516 -515
  73. package/dist/rl.js +612 -612
  74. package/dist/rl.js.map +1 -1
  75. package/dist/{rubric-predictive-validity-Cy_W-hWZ.d.ts → rubric-predictive-validity-C2hDKM8Z.d.ts} +1 -1
  76. package/dist/{run-campaign-7WNXMDSN.js → run-campaign-WXY7KI67.js} +2 -2
  77. package/dist/{run-record-e7vj1uZQ.d.ts → run-record-CP2ObebC.d.ts} +14 -18
  78. package/dist/{runtime-trajectory-BDgfGZSr.d.ts → runtime-trajectory-BOUUjI0y.d.ts} +1 -1
  79. package/dist/{semantic-concept-judge-B9MgmBnM.d.ts → semantic-concept-judge-DSBB2Cfp.d.ts} +2 -2
  80. package/dist/{summary-report-BDOFevaT.d.ts → summary-report-CInXwsza.d.ts} +1 -1
  81. package/dist/testing-C21CHsq2.d.ts +20 -0
  82. package/dist/testing.d.ts +1 -0
  83. package/dist/testing.js +8 -0
  84. package/dist/testing.js.map +1 -0
  85. package/dist/traces.d.ts +26 -10
  86. package/dist/traces.js +41 -11
  87. package/dist/{types-Ce17tDlG.d.ts → types-B5x54y6n.d.ts} +1 -1
  88. package/dist/{types-mn5Aqk7x.d.ts → types-BUxNaJ8c.d.ts} +2 -4
  89. package/dist/{types-BU-7W85F.d.ts → types-DQRY8ZT-.d.ts} +60 -58
  90. package/dist/workflow/index.d.ts +5 -4
  91. package/dist/workflow/index.js +1 -1
  92. package/docs/campaign-proposers.md +170 -0
  93. package/docs/concepts.md +8 -4
  94. package/docs/customer-journeys.md +15 -13
  95. package/docs/design/loop-taxonomy.md +34 -66
  96. package/docs/distributed-driver.md +14 -14
  97. package/docs/feature-guide.md +1 -1
  98. package/docs/hosted-ingest-spec.md +2 -3
  99. package/docs/multi-shot-optimization.md +8 -8
  100. package/docs/product-eval-adoption.md +1 -1
  101. package/docs/self-improvement-map.md +33 -29
  102. package/package.json +8 -14
  103. package/dist/chunk-2K6UUZ7P.js.map +0 -1
  104. package/dist/chunk-CTBHKLEU.js.map +0 -1
  105. package/dist/chunk-E4GH6USR.js.map +0 -1
  106. package/dist/chunk-EGPMSBEZ.js.map +0 -1
  107. package/dist/chunk-KWRRMR3J.js.map +0 -1
  108. package/dist/chunk-MIFZUPEK.js.map +0 -1
  109. package/dist/chunk-MPQWFX6Y.js.map +0 -1
  110. package/dist/chunk-Q5LIB7BC.js.map +0 -1
  111. package/dist/chunk-QMUEXQJS.js.map +0 -1
  112. package/dist/chunk-SD2YFWQQ.js.map +0 -1
  113. package/docs/design/external-agent-wedge.md +0 -89
  114. package/docs/design/phase-d-rfc.md +0 -125
  115. package/docs/design/phase4-consumer-migration.md +0 -70
  116. package/docs/design/primitives-integration-spec.md +0 -393
  117. package/docs/design/product-self-improvement-loop.md +0 -146
  118. package/docs/design/self-improvement-engine.md +0 -140
  119. package/docs/design/self-improvement-protocol.md +0 -223
  120. package/docs/design/self-improvement-roadmap.md +0 -106
  121. package/docs/design/substrate-gaps.md +0 -118
  122. package/docs/phase-b-pairing-kit.md +0 -188
  123. package/docs/phase-b-runbook.md +0 -176
  124. package/docs/pilot/README.md +0 -62
  125. package/docs/pilot/customer-checklist.md +0 -90
  126. package/docs/pilot/integration-foreign-stack.md +0 -296
  127. package/docs/pilot/integration-tangle-stack.md +0 -248
  128. package/docs/pilot/one-pager.md +0 -161
  129. package/docs/pilot/sample-insight-report.json +0 -172
  130. package/docs/quickstart-external.md +0 -229
  131. package/docs/research/belief-state-agent-eval-roadmap.md +0 -593
  132. package/docs/research/research-roadmap.md +0 -205
  133. package/docs/specs/driver-honest-spec.md +0 -251
  134. package/docs/specs/hermes-self-improvement-audit.md +0 -93
  135. package/docs/specs/profile-versioning.md +0 -291
  136. package/docs/three-package-architecture.md +0 -168
  137. /package/dist/{chunk-TBDR6PAI.js.map → chunk-IZCEK2HR.js.map} +0 -0
  138. /package/dist/{chunk-KW53MSA5.js.map → chunk-X74V6ESX.js.map} +0 -0
  139. /package/dist/{chunk-2KNZHH3P.js.map → chunk-Z6L6YSU6.js.map} +0 -0
  140. /package/dist/{run-campaign-7WNXMDSN.js.map → run-campaign-WXY7KI67.js.map} +0 -0
@@ -1,8 +1,6 @@
1
- import { b as RunTokenUsage } from './run-record-e7vj1uZQ.js';
1
+ import { b as RunTokenUsage } from './run-record-CP2ObebC.js';
2
2
 
3
3
  /**
4
- * @experimental
5
- *
6
4
  * Pass A substrate types — `runCampaign` is the one primitive every
7
5
  * eval flow composes from. Three contracts in this file:
8
6
  *
@@ -20,14 +18,14 @@ import { b as RunTokenUsage } from './run-record-e7vj1uZQ.js';
20
18
  * can build dashboards / CI gates / regression diffs against a stable schema.
21
19
  */
22
20
 
23
- /** @experimental Stable identifier + kind tag for any scenario. Consumers
21
+ /** Stable identifier + kind tag for any scenario. Consumers
24
22
  * extend with their per-domain payload (persona, task, requirement, ...). */
25
23
  interface Scenario {
26
24
  id: string;
27
25
  kind: string;
28
26
  tags?: string[];
29
27
  }
30
- /** @experimental Context handed to every dispatch invocation. Scoped — every
28
+ /** Context handed to every dispatch invocation. Scoped — every
31
29
  * trace/span carries the cellId, every artifact write lands under the cell's
32
30
  * artifact root, the cost meter accumulates per cell. */
33
31
  interface DispatchContext {
@@ -52,10 +50,10 @@ interface DispatchContext {
52
50
  */
53
51
  placement?: string;
54
52
  }
55
- /** @experimental One function: scenario + ctx → artifact. Dispatcher chooses
53
+ /** One function: scenario + ctx → artifact. Dispatcher chooses
56
54
  * whether to call `runMultishot`, `runLoop`, raw `streamPrompt`, anything. */
57
55
  type DispatchFn<TScenario extends Scenario, TArtifact> = (scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
58
- /** @experimental One session within a multi-session journey. Dispatch is
56
+ /** One session within a multi-session journey. Dispatch is
59
57
  * invoked once per session in order; state from prior session's artifact
60
58
  * is exposed via `ctx.priorSessionArtifact`. */
61
59
  interface SessionScript<TScenario, TArtifact> {
@@ -74,7 +72,7 @@ interface JudgeDimension {
74
72
  /** Description shown in the judge's user prompt. */
75
73
  description: string;
76
74
  }
77
- /** @experimental Pluggable dimensional scorer. `score` is the contract:
75
+ /** Pluggable dimensional scorer. `score` is the contract:
78
76
  * given an artifact + scenario, return a `JudgeScore`. This is deliberately a
79
77
  * function, not a fixed LLM-prompt shape — real consumers judge with
80
78
  * ensembles, deterministic checks, or a single LLM call, and the substrate
@@ -118,7 +116,7 @@ interface JudgeScore {
118
116
  /** Ensemble extras: each surviving judge's per-dimension scores. */
119
117
  perJudge?: Record<string, Record<string, number>>;
120
118
  }
121
- /** @experimental A tier-4 code surface — a candidate change to the agent's
119
+ /** A tier-4 code surface — a candidate change to the agent's
122
120
  * IMPLEMENTATION, not its prompt. Produced by autoresearch (reads codebase +
123
121
  * trace findings → opens a worktree). Measured by checking out `worktreeRef`
124
122
  * and running the worker against the changed code. See the improvement-tier
@@ -133,7 +131,7 @@ interface CodeSurface {
133
131
  /** Human summary of what changed — rendered into the auto-PR body. */
134
132
  summary?: string;
135
133
  }
136
- /** @experimental The mutable surface a driver proposes. Tiers (see
134
+ /** The mutable surface a proposer changes. Tiers (see
137
135
  * `docs/design/loop-taxonomy.md`):
138
136
  * - `string` — tiers 1-2: system-prompt addendum / serialized tool
139
137
  * config. Cheap, reversible, text-diffable.
@@ -141,12 +139,12 @@ interface CodeSurface {
141
139
  * Tier 3 (knowledge) is owned by agent-knowledge and rides its own adapter,
142
140
  * not this type. */
143
141
  type MutableSurface = string | CodeSurface;
144
- /** @experimental A driver proposal carrying the surface AND the WHY behind
145
- * it. Reflective drivers (`gepaDriver`) parse a `{label, rationale, payload}`
142
+ /** A proposer output carrying the surface AND the WHY behind
143
+ * it. Reflective proposers (`gepaProposer`) parse a `{label, rationale, payload}`
146
144
  * from the model; without this wrapper the loop keeps only `payload` and the
147
145
  * rationale that motivated the change is lost — the candidate becomes
148
146
  * unattributable. `propose()` may return either bare `MutableSurface`s (cheap
149
- * blind mutators) or these (reflective drivers); the loop normalizes both. */
147
+ * blind mutators) or these (reflective proposers); the loop normalizes both. */
150
148
  interface ProposedCandidate {
151
149
  surface: MutableSurface;
152
150
  /** Short human label for the change (≤ 40 chars typical). */
@@ -156,16 +154,16 @@ interface ProposedCandidate {
156
154
  * emitted provenance record. */
157
155
  rationale: string;
158
156
  }
159
- /** @experimental Type guard: a proposal carrying its rationale vs a bare
157
+ /** Type guard: a proposal carrying its rationale vs a bare
160
158
  * surface. The loop branches on this to populate `GenerationCandidate`. */
161
159
  declare function isProposedCandidate(value: MutableSurface | ProposedCandidate): value is ProposedCandidate;
162
- /** @experimental A non-dominated parent on the GEPA Pareto frontier — a
160
+ /** A non-dominated parent on the GEPA Pareto frontier — a
163
161
  * surface that, across the per-scenario objective vectors, no other tried
164
162
  * surface beats on every scenario. A candidate worse on the mean composite
165
163
  * but uniquely best on one hard scenario is non-dominated and survives here;
166
164
  * the composite-best ranking would discard the lesson it carries. The loop
167
- * computes the frontier across ALL generations and hands it to the driver so
168
- * a reflective driver can combine complementary lessons (GEPA, Agrawal et
165
+ * computes the frontier across ALL generations and hands it to the proposer so
166
+ * a reflective proposer can combine complementary lessons (GEPA, Agrawal et
169
167
  * al., arXiv:2507.19457). See `pareto.ts` (`paretoFrontier`). */
170
168
  interface ParetoParent {
171
169
  surface: MutableSurface;
@@ -181,10 +179,10 @@ interface ParetoParent {
181
179
  label?: string;
182
180
  rationale?: string;
183
181
  }
184
- /** @experimental Stateless surface mutation — given findings + current
182
+ /** Stateless surface mutation — given findings + current
185
183
  * surface, return N candidate surfaces. Pure transform, no generation
186
184
  * awareness. Reflective-mutation and `AxGEPA` mutators conform. Wrapped by
187
- * `evolutionaryDriver` to become an `ImprovementDriver`. */
185
+ * `evolutionaryProposer` to become a `SurfaceProposer`. */
188
186
  interface Mutator<TFindings = unknown> {
189
187
  kind: string;
190
188
  mutate(args: {
@@ -194,12 +192,12 @@ interface Mutator<TFindings = unknown> {
194
192
  signal: AbortSignal;
195
193
  }): Promise<Array<MutableSurface | ProposedCandidate>>;
196
194
  }
197
- /** @experimental Everything a driver's `propose()` may read to plan the next
195
+ /** Everything a proposer may read to plan the next
198
196
  * batch of candidates. The first six fields are always present; the rest are
199
- * optional context the loop supplies when available, so cheap drivers
200
- * (`evolutionaryDriver`) can ignore them while a code-tier agentic generator
201
- * consumes the research report + dataset to drive a coding harness.
202
- * See `docs/design/self-improvement-engine.md`. */
197
+ * optional context the loop supplies when available, so cheap proposers
198
+ * (`evolutionaryProposer`) can ignore them while a code-tier agentic generator
199
+ * consumes the report + dataset to drive a coding harness.
200
+ * See `docs/campaign-proposers.md`. */
203
201
  interface ProposeContext<TFindings = unknown> {
204
202
  currentSurface: MutableSurface;
205
203
  history: GenerationRecord[];
@@ -208,11 +206,10 @@ interface ProposeContext<TFindings = unknown> {
208
206
  populationSize: number;
209
207
  generation: number;
210
208
  signal: AbortSignal;
211
- /** The Phase-2 research report (analyst findings + diff), produced AFTER the
212
- * trace analysts run. Opaque to the substrate — the driver that consumes it
213
- * types it. See the phase diagram in self-improvement-engine.md. */
209
+ /** Optional analysis report produced before proposal. Opaque to the substrate:
210
+ * the proposer that consumes it owns the shape. */
214
211
  report?: unknown;
215
- /** Handle to all captured data — the driver samples traces / artifacts /
212
+ /** Handle to all captured data — the proposer samples traces / artifacts /
216
213
  * rewards here to ground its proposals. */
217
214
  dataset?: LabeledScenarioStore;
218
215
  /** DEPTH: max iterations the agentic generator may take per candidate.
@@ -221,38 +218,38 @@ interface ProposeContext<TFindings = unknown> {
221
218
  maxImprovementShots?: number;
222
219
  /** GEPA Pareto frontier across ALL generations so far — the non-dominated
223
220
  * surfaces by per-scenario objective vector. Empty/absent on generation 0
224
- * (only the baseline is scored). A reflective driver combines the
221
+ * (only the baseline is scored). A reflective proposer combines the
225
222
  * complementary lessons of these parents (each excels on different
226
- * scenarios) into a merged candidate. Drivers doing pure single-parent
223
+ * scenarios) into a merged candidate. Proposers doing pure single-parent
227
224
  * reflection may ignore it. See {@link ParetoParent}. */
228
225
  paretoParents?: ParetoParent[];
229
226
  /** FIREWALL (non-negotiable): the held-out judge is write-only — its verdicts
230
227
  * score the chosen output and gate promotion, and are NEVER an input to
231
228
  * proposal/steering (else the optimizer games the acceptance axis = an
232
229
  * oracle). This `never`-typed field makes that a compile-time tripwire: a
233
- * driver that tries to thread judge verdicts into the proposal will not type.
230
+ * proposer that tries to thread judge verdicts into the proposal will not type.
234
231
  * Steering may consume TRACE-OBSERVABLE signals (what the agent did) via
235
232
  * `findings`/`report`; it may NOT consume the judge's held-out verdict. */
236
233
  judgeScores?: never;
237
234
  }
238
- /** @experimental A surface-improvement strategy the DRIVER of the
239
- * improvement loop. Given the current best surface, the history of what's
240
- * been tried + scored, and any external findings, propose the next batch of
241
- * candidate surfaces to measure. Optionally decide to stop early.
235
+ /** A surface-improvement strategy. Given the current best
236
+ * surface, the history of what's been tried + scored, and any external
237
+ * findings, propose the next batch of candidate surfaces to measure.
238
+ * Optionally decide to stop early.
242
239
  *
243
- * The evolutionary mutator (`evolutionaryDriver`, here) and agent-runtime's
244
- * `improvementDriver` (with reflective / agentic generators) both conform
245
- * drivers of the SAME loop, not separate loops. The loop body
246
- * (`runOptimization`) and the gated promotion shell (`runImprovementLoop`)
247
- * are driver-agnostic. */
248
- interface ImprovementDriver<TFindings = unknown> {
240
+ * The evolutionary mutator (`evolutionaryProposer`, here) and agent-runtime's
241
+ * reflective / agentic generators both conform. They are proposers for the
242
+ * SAME loop, not separate loops. The loop body (`runOptimization`) and the
243
+ * gated promotion shell (`runImprovementLoop`) are proposer-agnostic.
244
+ */
245
+ interface SurfaceProposer<TFindings = unknown> {
249
246
  kind: string;
250
- /** Plan: propose N candidate surfaces for the next generation. A driver
247
+ /** Plan: propose N candidate surfaces for the next generation. A proposer
251
248
  * may return bare `MutableSurface`s or `ProposedCandidate`s that carry the
252
249
  * `{label, rationale}` motivating the change — the loop threads the
253
250
  * rationale into `GenerationCandidate` and the emitted provenance. */
254
251
  propose(ctx: ProposeContext<TFindings>): Promise<Array<MutableSurface | ProposedCandidate>>;
255
- /** Decide: stop early when the driver judges the search converged or
252
+ /** Decide: stop early when the proposer judges the search converged or
256
253
  * exhausted. Default (omitted) runs all `maxGenerations`. */
257
254
  decide?(args: {
258
255
  history: GenerationRecord[];
@@ -261,13 +258,18 @@ interface ImprovementDriver<TFindings = unknown> {
261
258
  reason?: string;
262
259
  };
263
260
  }
264
- interface OptimizerConfig {
265
- driver: ImprovementDriver;
261
+ /** Optional vocabulary alias. The loop is the optimizer; this object is the
262
+ * proposer inside that loop. */
263
+ type OptimizationProposer<TFindings = unknown> = SurfaceProposer<TFindings>;
264
+ interface OptimizerConfigBase {
266
265
  populationSize: number;
267
266
  maxGenerations: number;
268
267
  surfaceExtractor: (profile: unknown) => MutableSurface;
269
268
  }
270
- /** @experimental Five-valued verdict taxonomy (MOSS-paper alignment). */
269
+ interface OptimizerConfig extends OptimizerConfigBase {
270
+ proposer: SurfaceProposer;
271
+ }
272
+ /** Five-valued verdict taxonomy (MOSS-paper alignment). */
271
273
  type GateDecision = 'ship' | 'hold' | 'need_more_work' | 'model_ceiling' | 'arch_ceiling';
272
274
  interface GateContext<TArtifact, TScenario extends Scenario> {
273
275
  candidateArtifacts: Map<string, TArtifact>;
@@ -296,12 +298,12 @@ interface GateResult {
296
298
  }>;
297
299
  delta?: number;
298
300
  }
299
- /** @experimental Composable promotion gate. */
301
+ /** Composable promotion gate. */
300
302
  interface Gate<TArtifact = unknown, TScenario extends Scenario = Scenario> {
301
303
  name: string;
302
304
  decide(ctx: GateContext<TArtifact, TScenario>): Promise<GateResult>;
303
305
  }
304
- /** @experimental Scoped trace writer handed to each dispatch — every span
306
+ /** Scoped trace writer handed to each dispatch — every span
305
307
  * auto-tagged with the cellId so traces filter cleanly. */
306
308
  interface CampaignTraceWriter {
307
309
  span(name: string, attributes?: Record<string, unknown>): TraceSpan;
@@ -311,7 +313,7 @@ interface TraceSpan {
311
313
  end(attributes?: Record<string, unknown>): void;
312
314
  setAttribute(key: string, value: unknown): void;
313
315
  }
314
- /** @experimental Scoped artifact writer — `write(path, content)` lands under
316
+ /** Scoped artifact writer — `write(path, content)` lands under
315
317
  * `<runDir>/<cellId>/<path>`. */
316
318
  interface CampaignArtifactWriter {
317
319
  write(path: string, content: string | Uint8Array): Promise<string>;
@@ -322,7 +324,7 @@ interface CampaignArtifactWriter {
322
324
  * backend-integrity guard with ONE source of truth — a field added to
323
325
  * `RunTokenUsage` is a compile error here, not a silent drift. */
324
326
  type CampaignTokenUsage = RunTokenUsage;
325
- /** @experimental Cell-scoped cost meter. NOTHING is captured automatically —
327
+ /** Cell-scoped cost meter. NOTHING is captured automatically —
326
328
  * the substrate does not intercept the LLM call, so it cannot see cost or
327
329
  * tokens unless the dispatch reports them. Every LLM cost MUST be reported via
328
330
  * `observe` and every token count via `observeTokens`; a dispatch that reports
@@ -341,7 +343,7 @@ interface CampaignCostMeter {
341
343
  /** Accumulated token usage for this cell (zeros if never observed). */
342
344
  tokens(): CampaignTokenUsage;
343
345
  }
344
- /** @experimental Source tag — required on every store write. Used by the
346
+ /** Source tag — required on every store write. Used by the
345
347
  * default training-source filter (production-trace samples NOT used as
346
348
  * training scenarios unless explicitly opted in). */
347
349
  type LabeledScenarioSource = 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
@@ -363,7 +365,7 @@ type RedactionStatus = 'raw' | 'redacted-pii' | 'redacted-secrets' | 'fully-reda
363
365
  type LabelTrust = 'unverified' | 'verified-signal' | 'human-rated';
364
366
  /** Ordinal rank for a label-trust tier; absent ⇒ `unverified` (rank 0). */
365
367
  declare function labelTrustRank(trust: LabelTrust | undefined): number;
366
- /** @experimental Required-provenance write. The store rejects writes that
368
+ /** Required-provenance write. The store rejects writes that
367
369
  * lack provenance — a default-on flywheel without provenance is the
368
370
  * data-poisoning vector flagged in the alignment review. */
369
371
  interface LabeledScenarioWrite<TScenario extends Scenario = Scenario, TArtifact = unknown> {
@@ -455,7 +457,7 @@ interface GenerationRecord {
455
457
  promoted: string[];
456
458
  }
457
459
  /** One scored candidate surface in a generation. `dimensions` + `scenarios`
458
- * let a reflective `ImprovementDriver` ground its next proposal on WHICH
460
+ * let a reflective proposer ground its next proposal on WHICH
459
461
  * dimensions the candidate is weakest on and WHICH scenarios it best/worst
460
462
  * handled — the evidence a blind `Mutator` cannot see. */
461
463
  interface GenerationCandidate {
@@ -467,7 +469,7 @@ interface GenerationCandidate {
467
469
  dimensions: Record<string, number>;
468
470
  /** Per-scenario composite (mean over reps + judges), plus the judge's
469
471
  * free-form `notes` for that scenario — the "why it scored low" evidence a
470
- * reflective driver grounds its next edit on. Keep `notes` GENERALIZABLE
472
+ * reflective proposer grounds its next edit on. Keep `notes` GENERALIZABLE
471
473
  * (which checks/lines/dimensions failed and how), NOT case-specific ground
472
474
  * truth: leaking expected answers into the prompt is memorization, and the
473
475
  * held-out gate would reject it anyway. */
@@ -476,12 +478,12 @@ interface GenerationCandidate {
476
478
  composite: number;
477
479
  notes?: string;
478
480
  }>;
479
- /** Driver-supplied short label for the change. Present when the driver
481
+ /** Proposer-supplied short label for the change. Present when the proposer
480
482
  * returned a `ProposedCandidate`; absent for bare-surface mutators. */
481
483
  label?: string;
482
- /** Driver-supplied rationale — WHY this candidate was proposed. The
484
+ /** Proposer-supplied rationale — WHY this candidate was proposed. The
483
485
  * "because rationale Z" the audit requires to survive to the result.
484
- * Present when the driver returned a `ProposedCandidate`. */
486
+ * Present when the proposer returned a `ProposedCandidate`. */
485
487
  rationale?: string;
486
488
  }
487
489
  interface CampaignAggregates {
@@ -517,4 +519,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
517
519
  scenarios: Array<Pick<TScenario, 'id' | 'kind'>>;
518
520
  }
519
521
 
520
- export { isProposedCandidate as A, labelTrustRank as B, type CampaignAggregates as C, type DispatchFn as D, type Gate as G, type ImprovementDriver as I, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizerConfig as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeConfig as a, type DispatchContext as b, type GateDecision as c, type CampaignArtifactWriter as d, type CampaignCellResult as e, type CampaignCostMeter as f, type CampaignResult as g, type CampaignTraceWriter as h, type CodeSurface as i, type GateContext as j, type GateResult as k, type GenerationCandidate as l, type GenerationRecord as m, type JudgeDimension as n, type Mutator as o, type SessionScript as p, type ProposeContext as q, type LabeledScenarioWrite as r, type LabeledScenarioSampleArgs as s, type LabeledScenarioRecord as t, type LabelTrust as u, type LabeledScenarioSource as v, type CampaignTokenUsage as w, type JudgeAggregate as x, type ProposedCandidate as y, type ScenarioAggregate as z };
522
+ export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchFn as D, isProposedCandidate as E, labelTrustRank as F, type Gate as G, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizationProposer as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeConfig as a, type DispatchContext as b, type SurfaceProposer as c, type GateDecision as d, type CampaignAggregates as e, type CampaignArtifactWriter as f, type CampaignCellResult as g, type CampaignCostMeter as h, type CampaignTraceWriter as i, type CodeSurface as j, type GateContext as k, type GateResult as l, type GenerationCandidate as m, type GenerationRecord as n, type JudgeDimension as o, type Mutator as p, type OptimizerConfig as q, type SessionScript as r, type LabeledScenarioWrite as s, type LabeledScenarioSampleArgs as t, type LabeledScenarioRecord as u, type LabelTrust as v, type ProposedCandidate as w, type ProposeContext as x, type LabeledScenarioSource as y, type CampaignTokenUsage as z };
@@ -1,7 +1,7 @@
1
1
  import { W as WorkflowTopology } from '../harness-optimizer-mOl9XX_O.js';
2
- import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from '../run-record-e7vj1uZQ.js';
3
- import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-Ce17tDlG.js';
4
- import { F as FailureClusterInsight } from '../insight-report-DWl3z9tl.js';
2
+ import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from '../run-record-CP2ObebC.js';
3
+ import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-B5x54y6n.js';
4
+ import { F as FailureClusterInsight } from '../insight-report-BnRjTibG.js';
5
5
  import { a as VerificationReport, L as LayerResult } from '../multi-layer-verifier-DUZXrPDA.js';
6
6
  import { F as FailureClusterReport } from '../failure-cluster-DH9Flgcf.js';
7
7
  import { R as RedactionRule, a as RedactionReport } from '../redact-B40YG2M_.js';
@@ -12,13 +12,14 @@ import '../pareto-E-pembql.js';
12
12
  import '../run-critic-CmMf05uV.js';
13
13
  import '../schema-m0gsnbt3.js';
14
14
  import '../store-BcFXE6LG.js';
15
+ import '@tangle-network/agent-interface';
15
16
  import '../errors-CzMUYo7b.js';
16
17
  import '../store-C1YxJDEK.js';
17
18
  import '../types-C7DGg5ex.js';
18
19
  import '@tangle-network/tcloud';
19
20
  import '../llm-client-Bj7g0rqu.js';
20
21
  import '../raw-provider-sink-C46HDghv.js';
21
- import '../summary-report-BDOFevaT.js';
22
+ import '../summary-report-CInXwsza.js';
22
23
  import '../judge-calibration-0p2QcWNE.js';
23
24
  import '../verdict-C9MlYujm.js';
24
25
  import '../control-runtime-Acf9CGhw.js';
@@ -7,7 +7,7 @@ import {
7
7
  } from "../chunk-GGE4NNQT.js";
8
8
  import {
9
9
  validateRunRecord
10
- } from "../chunk-KWRRMR3J.js";
10
+ } from "../chunk-LO6IOIJ2.js";
11
11
  import "../chunk-VSMTAMNK.js";
12
12
  import {
13
13
  ValidationError
@@ -0,0 +1,170 @@
1
+ # Campaign Proposers, ELI5
2
+
3
+ A campaign proposer is the part of an improvement loop that says: "try this
4
+ candidate next."
5
+
6
+ It does not run your agent. It does not score anything. It only proposes a new
7
+ surface to measure. A surface is the thing you are changing: a prompt string, a
8
+ serialized config string, or a code/worktree surface.
9
+
10
+ Use **proposer** for this role. Older optimizer APIs used "driver"; that word is
11
+ now reserved for execution, sandbox, and router agents that actually drive
12
+ workers.
13
+
14
+ ## The Loop
15
+
16
+ ```text
17
+ current surface
18
+ -> proposer suggests candidate surfaces
19
+ -> runCampaign runs each candidate on scenarios
20
+ -> judges score the artifacts
21
+ -> runOptimization picks the best candidate
22
+ -> runImprovementLoop re-scores on holdout and gates release
23
+ ```
24
+
25
+ ## Proposer Input
26
+
27
+ Every `SurfaceProposer.propose(ctx)` receives:
28
+
29
+ | Field | Plain meaning |
30
+ |---|---|
31
+ | `currentSurface` | The prompt/config/code surface currently being improved. |
32
+ | `history` | What candidates were tried before and how they scored. |
33
+ | `findings` | Failure analysis or analyst findings from traces/eval runs. |
34
+ | `populationSize` | How many candidates the loop asks for this generation. |
35
+ | `generation` | Which generation number this is. |
36
+ | `signal` | Abort signal for cancellation. |
37
+ | `report` | Optional larger analysis report. |
38
+ | `dataset` | Optional labeled scenario store. |
39
+ | `paretoParents` | Optional non-dominated surfaces from prior generations. |
40
+
41
+ ## Proposer Output
42
+
43
+ A proposer returns candidates:
44
+
45
+ ```ts
46
+ {
47
+ surface: 'the full new prompt or config',
48
+ label: 'short name',
49
+ rationale: 'why this change should help'
50
+ }
51
+ ```
52
+
53
+ Bare surfaces are still accepted, but `label` and `rationale` make results
54
+ auditable, so new proposers should return `ProposedCandidate`.
55
+
56
+ ## Which Proposer To Use
57
+
58
+ | Proposer factory | Best when | Output surface |
59
+ |---|---|---|
60
+ | `gepaProposer` | You want a strong prompt rewrite driven by prior scores and findings. | prompt string |
61
+ | `skillOptProposer` | You are editing a structured skill/runbook and want anchored small patches. | prompt/skill string |
62
+ | `aceProposer` | You want append-only lessons from findings, preserving every distinct lesson. | prompt/playbook string |
63
+ | `memoryCurationProposer` | You want compact deduped lessons from findings. | prompt/playbook string |
64
+ | `parameterSweepProposer` | You want FAPO-style config/parameter edits from a JSON config surface. | JSON string |
65
+ | `fapoProposer` | You want the FAPO policy: prompt first, then parameter, then structural only when evidence supports escalation. | whatever its level proposer returns |
66
+
67
+ ## FAPO Proposer
68
+
69
+ FAPO is not "another prompt mutator." The paper describes a reviewed escalation
70
+ policy:
71
+
72
+ 1. evaluate the current workflow,
73
+ 2. attribute failures to prompt, parameter/config, or structure,
74
+ 3. propose one scoped change,
75
+ 4. review the change for scope/leakage/compatibility,
76
+ 5. measure it,
77
+ 6. keep moving or escalate only when the cheaper level is exhausted.
78
+
79
+ The simplest useful setup is prompt plus JSON config. Structural/code edits are
80
+ optional and should be injected by the app or runtime layer.
81
+
82
+ ```ts
83
+ import {
84
+ fapoProposer,
85
+ gepaProposer,
86
+ parameterSweepProposer,
87
+ runImprovementLoop,
88
+ } from '@tangle-network/agent-eval/campaign'
89
+
90
+ const proposer = fapoProposer({
91
+ scope: { allowedLevels: ['prompt', 'parameter'] },
92
+ promptProposer: gepaProposer({ llm, model, target: 'agent prompt' }),
93
+ parameterProposer: parameterSweepProposer({
94
+ candidates: [
95
+ {
96
+ label: 'raise-retrieval-k',
97
+ rationale: 'retrieval misses indicate the search budget may be too low',
98
+ changes: [{ path: 'retrieval.k', value: 10 }],
99
+ },
100
+ ],
101
+ }),
102
+ })
103
+
104
+ await runImprovementLoop({
105
+ scenarios: trainScenarios,
106
+ holdoutScenarios,
107
+ baselineSurface: JSON.stringify(currentConfig),
108
+ dispatchWithSurface,
109
+ judges,
110
+ proposer,
111
+ gate,
112
+ autoOnPromote: 'none',
113
+ runDir,
114
+ populationSize: 1,
115
+ maxGenerations: 10,
116
+ })
117
+ ```
118
+
119
+ If you do have a real code/worktree proposer, pass it as `structuralProposer`.
120
+ `agent-eval` intentionally does not provide that proposer because this package
121
+ measures candidates; the runtime or app owns code generation.
122
+
123
+ For side-by-side experiments with existing proposers, use the compare entry:
124
+
125
+ ```ts
126
+ import {
127
+ compareProposers,
128
+ fapoEscalationEntry,
129
+ gepaParetoEntry,
130
+ } from '@tangle-network/agent-eval/campaign'
131
+
132
+ await compareProposers({
133
+ proposers: [
134
+ gepaParetoEntry(config),
135
+ fapoEscalationEntry({
136
+ ...config,
137
+ parameterCandidates: [
138
+ {
139
+ label: 'raise-retrieval-k',
140
+ rationale: 'retrieval misses indicate the search budget may be too low',
141
+ changes: [{ path: 'retrieval.k', value: 10 }],
142
+ },
143
+ ],
144
+ }),
145
+ ],
146
+ baselineSurface,
147
+ holdoutScenarios,
148
+ dispatchWithSurface,
149
+ judges,
150
+ runDir,
151
+ })
152
+ ```
153
+
154
+ ## Common Mistakes
155
+
156
+ - Do not put eval logic inside a proposer. Put it in `dispatch` and `judges`.
157
+ - Do not let a proposer read held-out judge scores. `ProposeContext` makes this
158
+ a type-level firewall.
159
+ - Do not call FAPO a prompt-only optimizer. Its main value is evidence-based
160
+ escalation beyond prompt edits.
161
+ - Do not put Claude Code or sandbox-specific code in `agent-eval`. Structural
162
+ code generation should be supplied as an injected `SurfaceProposer` from the
163
+ runtime/app layer.
164
+
165
+ ## Simpler Mental Model
166
+
167
+ Use this sentence when wiring a loop:
168
+
169
+ > The proposer chooses candidates; the campaign measures them; the gate decides
170
+ > whether the measured winner is safe to promote.
package/docs/concepts.md CHANGED
@@ -9,17 +9,21 @@ connected, or the answer lacks required sources. The package gives products a
9
9
  shared way to record runs, check outcomes, classify failures, compare variants,
10
10
  and make release decisions.
11
11
 
12
- ## The three top-level functions
12
+ ## The top-level functions
13
13
 
14
- Everything funnels through `/contract`. Three entries, one shape coming back:
14
+ Everything funnels through `/contract`. Start with `defineAgentEval()` when you
15
+ can; drop to the raw functions when you need lower-level control.
15
16
 
16
17
  | Function | When to call it | What you give it | What you get back |
17
18
  |---|---|---|---|
19
+ | **`defineAgentEval()`** | You have scenarios, an agent, a judge, and a baseline surface, and you want one object you can score or improve. | scenarios, agent, judge, baseline surface | `{ evaluate(), improve() }` where `evaluate()` returns a campaign result and `improve()` returns a decision packet |
18
20
  | **`selfImprove()`** | You have a closed loop — scenarios, judge, agent in hand, and you want the substrate to propose better candidates + gate them. | scenarios, agent, judge, baseline surface | `SelfImproveResult.insight: InsightReport` + ship/hold verdict + winner surface |
19
21
  | **`analyzeRuns()`** | You have observed runs (production traces, an approve/reject corpus, a CSV gold set) and want the same rigor packet without invoking an agent. | `RunRecord[]` + optional flags | `InsightReport` |
20
22
  | **Intake adapters** (`fromFeedbackTable`, `fromOtelSpans`) | Your data isn't already in `RunRecord` shape — it's in Obsidian, Sheets, an OTel collector, etc. | source-specific input | `RunRecord[]` ready to pipe into `analyzeRuns()` |
21
23
 
22
- The three customer maturity stages — logs only → ratings → closed loop — map exactly to the three functions. See [`customer-journeys.md`](./customer-journeys.md) for the runnable walkthroughs.
24
+ The customer maturity stages — logs only → ratings → closed loop — map to these
25
+ entry points. See [`customer-journeys.md`](./customer-journeys.md) for the
26
+ runnable walkthroughs.
23
27
 
24
28
  The shape of the answer — `InsightReport` — is identical across all three paths. Distributional summary, paired-bootstrap lift CI, judge stats, inter-rater agreement, cost-quality Pareto, failure clusters, contamination check, outcome correlation, release axes, and a ranked recommendations array. Walked through section-by-section in [`insight-report.md`](./insight-report.md).
25
29
 
@@ -182,7 +186,7 @@ release decision.
182
186
 
183
187
  ## Where to go next
184
188
 
185
- - **Confused by "GEPA / HALO / trace analysis / drivers everywhere"?** → [self-improvement-map.md](./self-improvement-map.md) — one loop, four roles, the seven-driver catalog (production vs bench-only), and why `gepa-refine` is the same loop on a test bench.
189
+ - **Confused by "GEPA / HALO / trace analysis / proposers everywhere"?** → [self-improvement-map.md](./self-improvement-map.md) — one loop, four roles, the proposer catalog (production vs bench-only), and why `gepa-refine` is the same loop on a test bench.
186
190
  - **Which `run*` primitive do I use, and how do I grade produced state?** → [eval-surface-map.md](./eval-surface-map.md) — the campaign/matrix/optimization/gate primitives as a pick-by-"use-when" table, plus the produced-state grading composition (verifyCompletion-as-judge — there is no persona-dispatch wrapper) and the in-band body contract.
187
191
  - **Need the layman feature map?** → [feature-guide.md](./feature-guide.md) — what each primitive does, when to use it, integration patterns, and guardrails.
188
192
  - **Just want to score a string against a rubric?** → [wire-protocol.md](./wire-protocol.md) — HTTP/RPC interface, pluggable from any language.
@@ -58,7 +58,7 @@ Cost mean: $0.103 (p95: $0.131)
58
58
 
59
59
  1. Wire an `AnalystRegistry` to cluster the 6 failures by root cause via LLM analysis.
60
60
  2. Add `outcomeSignal` once they have downstream conversion / engagement / post-engagement data, and the report fits a reward model showing whether their score predicts the customer outcome.
61
- 3. Once they identify a step worth optimizing (translation, say), graduate to journey #3 — wrap that step in a `Dispatch` and call `selfImprove()`.
61
+ 3. Once they identify a step worth optimizing (translation, say), graduate to journey #3 — wrap that step as an `agent(surface, scenario)` and call `defineAgentEval()`.
62
62
 
63
63
  **Runnable:** [`examples/customer-otel-traces/`](../examples/customer-otel-traces/)
64
64
 
@@ -130,7 +130,7 @@ Mean κ: 0.43
130
130
  1. **Triage meeting on the disagreement cases.** Mean κ=0.43 means the rubric is ambiguous; clarify it on the cases that split.
131
131
  2. **Calibrate one LLM judge per reviewer.** Each reviewer's history is the gold signal — substrate primitive `calibrateJudge` against `raterScores` filtered to that reviewer.
132
132
  3. **Add engagement as `outcomeSignal`** once the content downstream is instrumented. The `outcomeCorrelation` section tells the team whether their taste predicts the founder's token-max goal — and if not, the linear reward model says how to retarget.
133
- 4. **Graduate to journey #3** — wrap the research-generation Claude-P call as a `Dispatch`, use the calibrated judges, run `selfImprove()` nightly. Open a PR against the GitHub Action when the holdout approval rate beats baseline.
133
+ 4. **Graduate to journey #3** — wrap the research-generation Claude-P call as an `agent(surface, scenario)`, use the calibrated judges, run `evalKit.improve()` nightly. Open a PR against the GitHub Action when the holdout approval rate beats baseline.
134
134
 
135
135
  **Runnable:** [`examples/customer-feedback-loop/`](../examples/customer-feedback-loop/)
136
136
 
@@ -142,14 +142,14 @@ Mean κ: 0.43
142
142
 
143
143
  **The frustration:** "We can run an A/B by hand but we don't know if the improvement is real. We don't have time to run paired bootstrap by hand. We want a function that decides."
144
144
 
145
- **What they need from agent-eval:** the closed loop in one function — propose, score, gate, ship — with the full rigor packet on the way out.
145
+ **What they need from agent-eval:** one reusable eval definition — propose, score, gate, ship — with the full rigor packet on the way out.
146
146
 
147
147
  ### The code
148
148
 
149
149
  ```ts
150
- import { selfImprove } from '@tangle-network/agent-eval/contract'
150
+ import { defineAgentEval } from '@tangle-network/agent-eval/contract'
151
151
 
152
- const result = await selfImprove({
152
+ const evalKit = defineAgentEval({
153
153
  scenarios,
154
154
  agent: async (surface, scenario) =>
155
155
  await myAgent.run({ systemPrompt: (surface as { systemPrompt: string }).systemPrompt, scenario }),
@@ -162,6 +162,8 @@ const result = await selfImprove({
162
162
  budget: { generations: 3, populationSize: 2 },
163
163
  })
164
164
 
165
+ const result = await evalKit.improve()
166
+
165
167
  result.gateDecision // 'ship' | 'hold' | ...
166
168
  result.insight // full decision packet
167
169
  ```
@@ -172,18 +174,18 @@ result.insight // full decision packet
172
174
  ═══ selfImprove() decision packet ═══
173
175
 
174
176
  Gate decision: ship
175
- Raw lift: +0.194
177
+ Raw lift: +0.361
176
178
 
177
179
  ── Statistical lift (paired bootstrap) ──
178
- delta: +0.254
179
- CI95: [0.254, 0.254]
180
- pValue: 1.0000
181
- Cohen's d: 0.00
182
- MDE @ 80% power: 2.802
183
- required n at observed effect: 244
180
+ delta: +0.359
181
+ CI95: [0.311, 0.408]
182
+ pValue: 0.0013
183
+ Cohen's d: 8.58
184
+ MDE @ 80% power: 1.401
185
+ required n at observed effect: 122
184
186
 
185
187
  ── Recommendations ──
186
- [critical] ship — Ship — lift 0.254 (95% CI 0.254..0.254)
188
+ [critical] ship — Ship — lift 0.359 (95% CI 0.311..0.408)
187
189
  ```
188
190
 
189
191
  ### Next steps for this customer