@tangle-network/agent-eval 0.94.0 → 0.95.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -0
- package/README.md +44 -30
- package/dist/adapters/http.d.ts +8 -7
- package/dist/adapters/http.js.map +1 -1
- package/dist/adapters/langchain.d.ts +3 -2
- package/dist/adapters/otel.d.ts +5 -4
- package/dist/analyst/index.d.ts +11 -31
- package/dist/analyst/index.js +5 -65
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-B6Ljo_dI.d.ts → analyze-runs-DtT6F_6T.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +4 -3
- package/dist/benchmarks/index.d.ts +3 -2
- package/dist/campaign/index.d.ts +727 -616
- package/dist/campaign/index.js +1863 -1316
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-2K6UUZ7P.js → chunk-2T4EZACH.js} +1 -1
- package/dist/chunk-2T4EZACH.js.map +1 -0
- package/dist/{chunk-CTBHKLEU.js → chunk-77T4STFI.js} +59 -86
- package/dist/chunk-77T4STFI.js.map +1 -0
- package/dist/{chunk-EGPMSBEZ.js → chunk-7QTQKIDD.js} +178 -177
- package/dist/chunk-7QTQKIDD.js.map +1 -0
- package/dist/{chunk-MIFZUPEK.js → chunk-AQ5WQAIV.js} +21 -6
- package/dist/chunk-AQ5WQAIV.js.map +1 -0
- package/dist/chunk-DJWX3GVS.js +81 -0
- package/dist/chunk-DJWX3GVS.js.map +1 -0
- package/dist/{chunk-S6OZEZQK.js → chunk-HMA63UEO.js} +37 -9
- package/dist/{chunk-S6OZEZQK.js.map → chunk-HMA63UEO.js.map} +1 -1
- package/dist/{chunk-TBDR6PAI.js → chunk-IZCEK2HR.js} +2 -2
- package/dist/{chunk-SD2YFWQQ.js → chunk-KKWJD5E6.js} +20 -20
- package/dist/chunk-KKWJD5E6.js.map +1 -0
- package/dist/{chunk-KWRRMR3J.js → chunk-LO6IOIJ2.js} +10 -10
- package/dist/chunk-LO6IOIJ2.js.map +1 -0
- package/dist/{chunk-E4GH6USR.js → chunk-NZEQVRH5.js} +2 -2
- package/dist/chunk-NZEQVRH5.js.map +1 -0
- package/dist/{chunk-MPQWFX6Y.js → chunk-PSWWQXHF.js} +13 -88
- package/dist/chunk-PSWWQXHF.js.map +1 -0
- package/dist/{chunk-Q5LIB7BC.js → chunk-S4SYLDFX.js} +2 -2
- package/dist/chunk-S4SYLDFX.js.map +1 -0
- package/dist/{chunk-KW53MSA5.js → chunk-X74V6ESX.js} +2 -2
- package/dist/{chunk-QMUEXQJS.js → chunk-YBIGNSCZ.js} +81 -4
- package/dist/chunk-YBIGNSCZ.js.map +1 -0
- package/dist/{chunk-2KNZHH3P.js → chunk-Z6L6YSU6.js} +2 -2
- package/dist/{code-agent-session-BO8nCnv3.d.ts → code-agent-session-CPHRCb4-.d.ts} +1 -1
- package/dist/contract/index.d.ts +91 -43
- package/dist/contract/index.js +127 -17
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-D6qwHXIR.d.ts → control-Doncu-B_.d.ts} +2 -2
- package/dist/control.d.ts +3 -2
- package/dist/control.js +2 -2
- package/dist/{corpus-B8A4BDR3.d.ts → corpus-D4YW9UoJ.d.ts} +1 -1
- package/dist/{default-registry-6dhErQbs.d.ts → default-registry-GyE8X5SP.d.ts} +3 -3
- package/dist/diagnose.d.ts +4 -3
- package/dist/diagnose.js +1 -1
- package/dist/{run-improvement-loop-DBahB8Ax.d.ts → gepa-C1NCIZ9o.d.ts} +117 -130
- package/dist/hosted/index.d.ts +5 -4
- package/dist/{index-Bx3gZ8xl.d.ts → index-_Y4oNOOb.d.ts} +1 -1
- package/dist/index.d.ts +76 -81
- package/dist/index.js +66 -31
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DWl3z9tl.d.ts → insight-report-BnRjTibG.d.ts} +1 -1
- package/dist/{kind-factory-0BhLSI27.d.ts → kind-factory-X3eDYbKn.d.ts} +2 -3
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +3 -2
- package/dist/multishot/index.d.ts +4 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{pre-registration-mAnCugl9.d.ts → pre-registration-nfUdc9EQ.d.ts} +2 -42
- package/dist/{provenance-P-bCL2Fo.d.ts → provenance-CncDq9qE.d.ts} +26 -41
- package/dist/{release-report-BEbWmVYj.d.ts → release-report-pidWUMZ2.d.ts} +2 -2
- package/dist/reporting.d.ts +5 -4
- package/dist/{researcher-B0C2_fVO.d.ts → researcher-Jr8ME1dZ.d.ts} +2 -2
- package/dist/rl.d.ts +516 -515
- package/dist/rl.js +612 -612
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-Cy_W-hWZ.d.ts → rubric-predictive-validity-C2hDKM8Z.d.ts} +1 -1
- package/dist/{run-campaign-7WNXMDSN.js → run-campaign-WXY7KI67.js} +2 -2
- package/dist/{run-record-e7vj1uZQ.d.ts → run-record-CP2ObebC.d.ts} +14 -18
- package/dist/{runtime-trajectory-BDgfGZSr.d.ts → runtime-trajectory-BOUUjI0y.d.ts} +1 -1
- package/dist/{semantic-concept-judge-B9MgmBnM.d.ts → semantic-concept-judge-DSBB2Cfp.d.ts} +2 -2
- package/dist/{summary-report-BDOFevaT.d.ts → summary-report-CInXwsza.d.ts} +1 -1
- package/dist/testing-C21CHsq2.d.ts +20 -0
- package/dist/testing.d.ts +1 -0
- package/dist/testing.js +8 -0
- package/dist/testing.js.map +1 -0
- package/dist/traces.d.ts +26 -10
- package/dist/traces.js +41 -11
- package/dist/{types-Ce17tDlG.d.ts → types-B5x54y6n.d.ts} +1 -1
- package/dist/{types-mn5Aqk7x.d.ts → types-BUxNaJ8c.d.ts} +2 -4
- package/dist/{types-BU-7W85F.d.ts → types-DQRY8ZT-.d.ts} +60 -58
- package/dist/workflow/index.d.ts +5 -4
- package/dist/workflow/index.js +1 -1
- package/docs/campaign-proposers.md +170 -0
- package/docs/concepts.md +8 -4
- package/docs/customer-journeys.md +15 -13
- package/docs/design/loop-taxonomy.md +34 -66
- package/docs/distributed-driver.md +14 -14
- package/docs/feature-guide.md +1 -1
- package/docs/hosted-ingest-spec.md +2 -3
- package/docs/multi-shot-optimization.md +8 -8
- package/docs/product-eval-adoption.md +1 -1
- package/docs/self-improvement-map.md +33 -29
- package/package.json +8 -14
- package/dist/chunk-2K6UUZ7P.js.map +0 -1
- package/dist/chunk-CTBHKLEU.js.map +0 -1
- package/dist/chunk-E4GH6USR.js.map +0 -1
- package/dist/chunk-EGPMSBEZ.js.map +0 -1
- package/dist/chunk-KWRRMR3J.js.map +0 -1
- package/dist/chunk-MIFZUPEK.js.map +0 -1
- package/dist/chunk-MPQWFX6Y.js.map +0 -1
- package/dist/chunk-Q5LIB7BC.js.map +0 -1
- package/dist/chunk-QMUEXQJS.js.map +0 -1
- package/dist/chunk-SD2YFWQQ.js.map +0 -1
- package/docs/design/external-agent-wedge.md +0 -89
- package/docs/design/phase-d-rfc.md +0 -125
- package/docs/design/phase4-consumer-migration.md +0 -70
- package/docs/design/primitives-integration-spec.md +0 -393
- package/docs/design/product-self-improvement-loop.md +0 -146
- package/docs/design/self-improvement-engine.md +0 -140
- package/docs/design/self-improvement-protocol.md +0 -223
- package/docs/design/self-improvement-roadmap.md +0 -106
- package/docs/design/substrate-gaps.md +0 -118
- package/docs/phase-b-pairing-kit.md +0 -188
- package/docs/phase-b-runbook.md +0 -176
- package/docs/pilot/README.md +0 -62
- package/docs/pilot/customer-checklist.md +0 -90
- package/docs/pilot/integration-foreign-stack.md +0 -296
- package/docs/pilot/integration-tangle-stack.md +0 -248
- package/docs/pilot/one-pager.md +0 -161
- package/docs/pilot/sample-insight-report.json +0 -172
- package/docs/quickstart-external.md +0 -229
- package/docs/research/belief-state-agent-eval-roadmap.md +0 -593
- package/docs/research/research-roadmap.md +0 -205
- package/docs/specs/driver-honest-spec.md +0 -251
- package/docs/specs/hermes-self-improvement-audit.md +0 -93
- package/docs/specs/profile-versioning.md +0 -291
- package/docs/three-package-architecture.md +0 -168
- /package/dist/{chunk-TBDR6PAI.js.map → chunk-IZCEK2HR.js.map} +0 -0
- /package/dist/{chunk-KW53MSA5.js.map → chunk-X74V6ESX.js.map} +0 -0
- /package/dist/{chunk-2KNZHH3P.js.map → chunk-Z6L6YSU6.js.map} +0 -0
- /package/dist/{run-campaign-7WNXMDSN.js.map → run-campaign-WXY7KI67.js.map} +0 -0
|
@@ -1,8 +1,6 @@
|
|
|
1
|
-
import { b as RunTokenUsage } from './run-record-
|
|
1
|
+
import { b as RunTokenUsage } from './run-record-CP2ObebC.js';
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
|
-
* @experimental
|
|
5
|
-
*
|
|
6
4
|
* Pass A substrate types — `runCampaign` is the one primitive every
|
|
7
5
|
* eval flow composes from. Three contracts in this file:
|
|
8
6
|
*
|
|
@@ -20,14 +18,14 @@ import { b as RunTokenUsage } from './run-record-e7vj1uZQ.js';
|
|
|
20
18
|
* can build dashboards / CI gates / regression diffs against a stable schema.
|
|
21
19
|
*/
|
|
22
20
|
|
|
23
|
-
/**
|
|
21
|
+
/** Stable identifier + kind tag for any scenario. Consumers
|
|
24
22
|
* extend with their per-domain payload (persona, task, requirement, ...). */
|
|
25
23
|
interface Scenario {
|
|
26
24
|
id: string;
|
|
27
25
|
kind: string;
|
|
28
26
|
tags?: string[];
|
|
29
27
|
}
|
|
30
|
-
/**
|
|
28
|
+
/** Context handed to every dispatch invocation. Scoped — every
|
|
31
29
|
* trace/span carries the cellId, every artifact write lands under the cell's
|
|
32
30
|
* artifact root, the cost meter accumulates per cell. */
|
|
33
31
|
interface DispatchContext {
|
|
@@ -52,10 +50,10 @@ interface DispatchContext {
|
|
|
52
50
|
*/
|
|
53
51
|
placement?: string;
|
|
54
52
|
}
|
|
55
|
-
/**
|
|
53
|
+
/** One function: scenario + ctx → artifact. Dispatcher chooses
|
|
56
54
|
* whether to call `runMultishot`, `runLoop`, raw `streamPrompt`, anything. */
|
|
57
55
|
type DispatchFn<TScenario extends Scenario, TArtifact> = (scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
58
|
-
/**
|
|
56
|
+
/** One session within a multi-session journey. Dispatch is
|
|
59
57
|
* invoked once per session in order; state from prior session's artifact
|
|
60
58
|
* is exposed via `ctx.priorSessionArtifact`. */
|
|
61
59
|
interface SessionScript<TScenario, TArtifact> {
|
|
@@ -74,7 +72,7 @@ interface JudgeDimension {
|
|
|
74
72
|
/** Description shown in the judge's user prompt. */
|
|
75
73
|
description: string;
|
|
76
74
|
}
|
|
77
|
-
/**
|
|
75
|
+
/** Pluggable dimensional scorer. `score` is the contract:
|
|
78
76
|
* given an artifact + scenario, return a `JudgeScore`. This is deliberately a
|
|
79
77
|
* function, not a fixed LLM-prompt shape — real consumers judge with
|
|
80
78
|
* ensembles, deterministic checks, or a single LLM call, and the substrate
|
|
@@ -118,7 +116,7 @@ interface JudgeScore {
|
|
|
118
116
|
/** Ensemble extras: each surviving judge's per-dimension scores. */
|
|
119
117
|
perJudge?: Record<string, Record<string, number>>;
|
|
120
118
|
}
|
|
121
|
-
/**
|
|
119
|
+
/** A tier-4 code surface — a candidate change to the agent's
|
|
122
120
|
* IMPLEMENTATION, not its prompt. Produced by autoresearch (reads codebase +
|
|
123
121
|
* trace findings → opens a worktree). Measured by checking out `worktreeRef`
|
|
124
122
|
* and running the worker against the changed code. See the improvement-tier
|
|
@@ -133,7 +131,7 @@ interface CodeSurface {
|
|
|
133
131
|
/** Human summary of what changed — rendered into the auto-PR body. */
|
|
134
132
|
summary?: string;
|
|
135
133
|
}
|
|
136
|
-
/**
|
|
134
|
+
/** The mutable surface a proposer changes. Tiers (see
|
|
137
135
|
* `docs/design/loop-taxonomy.md`):
|
|
138
136
|
* - `string` — tiers 1-2: system-prompt addendum / serialized tool
|
|
139
137
|
* config. Cheap, reversible, text-diffable.
|
|
@@ -141,12 +139,12 @@ interface CodeSurface {
|
|
|
141
139
|
* Tier 3 (knowledge) is owned by agent-knowledge and rides its own adapter,
|
|
142
140
|
* not this type. */
|
|
143
141
|
type MutableSurface = string | CodeSurface;
|
|
144
|
-
/**
|
|
145
|
-
* it. Reflective
|
|
142
|
+
/** A proposer output carrying the surface AND the WHY behind
|
|
143
|
+
* it. Reflective proposers (`gepaProposer`) parse a `{label, rationale, payload}`
|
|
146
144
|
* from the model; without this wrapper the loop keeps only `payload` and the
|
|
147
145
|
* rationale that motivated the change is lost — the candidate becomes
|
|
148
146
|
* unattributable. `propose()` may return either bare `MutableSurface`s (cheap
|
|
149
|
-
* blind mutators) or these (reflective
|
|
147
|
+
* blind mutators) or these (reflective proposers); the loop normalizes both. */
|
|
150
148
|
interface ProposedCandidate {
|
|
151
149
|
surface: MutableSurface;
|
|
152
150
|
/** Short human label for the change (≤ 40 chars typical). */
|
|
@@ -156,16 +154,16 @@ interface ProposedCandidate {
|
|
|
156
154
|
* emitted provenance record. */
|
|
157
155
|
rationale: string;
|
|
158
156
|
}
|
|
159
|
-
/**
|
|
157
|
+
/** Type guard: a proposal carrying its rationale vs a bare
|
|
160
158
|
* surface. The loop branches on this to populate `GenerationCandidate`. */
|
|
161
159
|
declare function isProposedCandidate(value: MutableSurface | ProposedCandidate): value is ProposedCandidate;
|
|
162
|
-
/**
|
|
160
|
+
/** A non-dominated parent on the GEPA Pareto frontier — a
|
|
163
161
|
* surface that, across the per-scenario objective vectors, no other tried
|
|
164
162
|
* surface beats on every scenario. A candidate worse on the mean composite
|
|
165
163
|
* but uniquely best on one hard scenario is non-dominated and survives here;
|
|
166
164
|
* the composite-best ranking would discard the lesson it carries. The loop
|
|
167
|
-
* computes the frontier across ALL generations and hands it to the
|
|
168
|
-
* a reflective
|
|
165
|
+
* computes the frontier across ALL generations and hands it to the proposer so
|
|
166
|
+
* a reflective proposer can combine complementary lessons (GEPA, Agrawal et
|
|
169
167
|
* al., arXiv:2507.19457). See `pareto.ts` (`paretoFrontier`). */
|
|
170
168
|
interface ParetoParent {
|
|
171
169
|
surface: MutableSurface;
|
|
@@ -181,10 +179,10 @@ interface ParetoParent {
|
|
|
181
179
|
label?: string;
|
|
182
180
|
rationale?: string;
|
|
183
181
|
}
|
|
184
|
-
/**
|
|
182
|
+
/** Stateless surface mutation — given findings + current
|
|
185
183
|
* surface, return N candidate surfaces. Pure transform, no generation
|
|
186
184
|
* awareness. Reflective-mutation and `AxGEPA` mutators conform. Wrapped by
|
|
187
|
-
* `
|
|
185
|
+
* `evolutionaryProposer` to become a `SurfaceProposer`. */
|
|
188
186
|
interface Mutator<TFindings = unknown> {
|
|
189
187
|
kind: string;
|
|
190
188
|
mutate(args: {
|
|
@@ -194,12 +192,12 @@ interface Mutator<TFindings = unknown> {
|
|
|
194
192
|
signal: AbortSignal;
|
|
195
193
|
}): Promise<Array<MutableSurface | ProposedCandidate>>;
|
|
196
194
|
}
|
|
197
|
-
/**
|
|
195
|
+
/** Everything a proposer may read to plan the next
|
|
198
196
|
* batch of candidates. The first six fields are always present; the rest are
|
|
199
|
-
* optional context the loop supplies when available, so cheap
|
|
200
|
-
* (`
|
|
201
|
-
* consumes the
|
|
202
|
-
* See `docs/
|
|
197
|
+
* optional context the loop supplies when available, so cheap proposers
|
|
198
|
+
* (`evolutionaryProposer`) can ignore them while a code-tier agentic generator
|
|
199
|
+
* consumes the report + dataset to drive a coding harness.
|
|
200
|
+
* See `docs/campaign-proposers.md`. */
|
|
203
201
|
interface ProposeContext<TFindings = unknown> {
|
|
204
202
|
currentSurface: MutableSurface;
|
|
205
203
|
history: GenerationRecord[];
|
|
@@ -208,11 +206,10 @@ interface ProposeContext<TFindings = unknown> {
|
|
|
208
206
|
populationSize: number;
|
|
209
207
|
generation: number;
|
|
210
208
|
signal: AbortSignal;
|
|
211
|
-
/**
|
|
212
|
-
*
|
|
213
|
-
* types it. See the phase diagram in self-improvement-engine.md. */
|
|
209
|
+
/** Optional analysis report produced before proposal. Opaque to the substrate:
|
|
210
|
+
* the proposer that consumes it owns the shape. */
|
|
214
211
|
report?: unknown;
|
|
215
|
-
/** Handle to all captured data — the
|
|
212
|
+
/** Handle to all captured data — the proposer samples traces / artifacts /
|
|
216
213
|
* rewards here to ground its proposals. */
|
|
217
214
|
dataset?: LabeledScenarioStore;
|
|
218
215
|
/** DEPTH: max iterations the agentic generator may take per candidate.
|
|
@@ -221,38 +218,38 @@ interface ProposeContext<TFindings = unknown> {
|
|
|
221
218
|
maxImprovementShots?: number;
|
|
222
219
|
/** GEPA Pareto frontier across ALL generations so far — the non-dominated
|
|
223
220
|
* surfaces by per-scenario objective vector. Empty/absent on generation 0
|
|
224
|
-
* (only the baseline is scored). A reflective
|
|
221
|
+
* (only the baseline is scored). A reflective proposer combines the
|
|
225
222
|
* complementary lessons of these parents (each excels on different
|
|
226
|
-
* scenarios) into a merged candidate.
|
|
223
|
+
* scenarios) into a merged candidate. Proposers doing pure single-parent
|
|
227
224
|
* reflection may ignore it. See {@link ParetoParent}. */
|
|
228
225
|
paretoParents?: ParetoParent[];
|
|
229
226
|
/** FIREWALL (non-negotiable): the held-out judge is write-only — its verdicts
|
|
230
227
|
* score the chosen output and gate promotion, and are NEVER an input to
|
|
231
228
|
* proposal/steering (else the optimizer games the acceptance axis = an
|
|
232
229
|
* oracle). This `never`-typed field makes that a compile-time tripwire: a
|
|
233
|
-
*
|
|
230
|
+
* proposer that tries to thread judge verdicts into the proposal will not type.
|
|
234
231
|
* Steering may consume TRACE-OBSERVABLE signals (what the agent did) via
|
|
235
232
|
* `findings`/`report`; it may NOT consume the judge's held-out verdict. */
|
|
236
233
|
judgeScores?: never;
|
|
237
234
|
}
|
|
238
|
-
/**
|
|
239
|
-
*
|
|
240
|
-
*
|
|
241
|
-
*
|
|
235
|
+
/** A surface-improvement strategy. Given the current best
|
|
236
|
+
* surface, the history of what's been tried + scored, and any external
|
|
237
|
+
* findings, propose the next batch of candidate surfaces to measure.
|
|
238
|
+
* Optionally decide to stop early.
|
|
242
239
|
*
|
|
243
|
-
* The evolutionary mutator (`
|
|
244
|
-
*
|
|
245
|
-
*
|
|
246
|
-
*
|
|
247
|
-
|
|
248
|
-
interface
|
|
240
|
+
* The evolutionary mutator (`evolutionaryProposer`, here) and agent-runtime's
|
|
241
|
+
* reflective / agentic generators both conform. They are proposers for the
|
|
242
|
+
* SAME loop, not separate loops. The loop body (`runOptimization`) and the
|
|
243
|
+
* gated promotion shell (`runImprovementLoop`) are proposer-agnostic.
|
|
244
|
+
*/
|
|
245
|
+
interface SurfaceProposer<TFindings = unknown> {
|
|
249
246
|
kind: string;
|
|
250
|
-
/** Plan: propose N candidate surfaces for the next generation. A
|
|
247
|
+
/** Plan: propose N candidate surfaces for the next generation. A proposer
|
|
251
248
|
* may return bare `MutableSurface`s or `ProposedCandidate`s that carry the
|
|
252
249
|
* `{label, rationale}` motivating the change — the loop threads the
|
|
253
250
|
* rationale into `GenerationCandidate` and the emitted provenance. */
|
|
254
251
|
propose(ctx: ProposeContext<TFindings>): Promise<Array<MutableSurface | ProposedCandidate>>;
|
|
255
|
-
/** Decide: stop early when the
|
|
252
|
+
/** Decide: stop early when the proposer judges the search converged or
|
|
256
253
|
* exhausted. Default (omitted) runs all `maxGenerations`. */
|
|
257
254
|
decide?(args: {
|
|
258
255
|
history: GenerationRecord[];
|
|
@@ -261,13 +258,18 @@ interface ImprovementDriver<TFindings = unknown> {
|
|
|
261
258
|
reason?: string;
|
|
262
259
|
};
|
|
263
260
|
}
|
|
264
|
-
|
|
265
|
-
|
|
261
|
+
/** Optional vocabulary alias. The loop is the optimizer; this object is the
|
|
262
|
+
* proposer inside that loop. */
|
|
263
|
+
type OptimizationProposer<TFindings = unknown> = SurfaceProposer<TFindings>;
|
|
264
|
+
interface OptimizerConfigBase {
|
|
266
265
|
populationSize: number;
|
|
267
266
|
maxGenerations: number;
|
|
268
267
|
surfaceExtractor: (profile: unknown) => MutableSurface;
|
|
269
268
|
}
|
|
270
|
-
|
|
269
|
+
interface OptimizerConfig extends OptimizerConfigBase {
|
|
270
|
+
proposer: SurfaceProposer;
|
|
271
|
+
}
|
|
272
|
+
/** Five-valued verdict taxonomy (MOSS-paper alignment). */
|
|
271
273
|
type GateDecision = 'ship' | 'hold' | 'need_more_work' | 'model_ceiling' | 'arch_ceiling';
|
|
272
274
|
interface GateContext<TArtifact, TScenario extends Scenario> {
|
|
273
275
|
candidateArtifacts: Map<string, TArtifact>;
|
|
@@ -296,12 +298,12 @@ interface GateResult {
|
|
|
296
298
|
}>;
|
|
297
299
|
delta?: number;
|
|
298
300
|
}
|
|
299
|
-
/**
|
|
301
|
+
/** Composable promotion gate. */
|
|
300
302
|
interface Gate<TArtifact = unknown, TScenario extends Scenario = Scenario> {
|
|
301
303
|
name: string;
|
|
302
304
|
decide(ctx: GateContext<TArtifact, TScenario>): Promise<GateResult>;
|
|
303
305
|
}
|
|
304
|
-
/**
|
|
306
|
+
/** Scoped trace writer handed to each dispatch — every span
|
|
305
307
|
* auto-tagged with the cellId so traces filter cleanly. */
|
|
306
308
|
interface CampaignTraceWriter {
|
|
307
309
|
span(name: string, attributes?: Record<string, unknown>): TraceSpan;
|
|
@@ -311,7 +313,7 @@ interface TraceSpan {
|
|
|
311
313
|
end(attributes?: Record<string, unknown>): void;
|
|
312
314
|
setAttribute(key: string, value: unknown): void;
|
|
313
315
|
}
|
|
314
|
-
/**
|
|
316
|
+
/** Scoped artifact writer — `write(path, content)` lands under
|
|
315
317
|
* `<runDir>/<cellId>/<path>`. */
|
|
316
318
|
interface CampaignArtifactWriter {
|
|
317
319
|
write(path: string, content: string | Uint8Array): Promise<string>;
|
|
@@ -322,7 +324,7 @@ interface CampaignArtifactWriter {
|
|
|
322
324
|
* backend-integrity guard with ONE source of truth — a field added to
|
|
323
325
|
* `RunTokenUsage` is a compile error here, not a silent drift. */
|
|
324
326
|
type CampaignTokenUsage = RunTokenUsage;
|
|
325
|
-
/**
|
|
327
|
+
/** Cell-scoped cost meter. NOTHING is captured automatically —
|
|
326
328
|
* the substrate does not intercept the LLM call, so it cannot see cost or
|
|
327
329
|
* tokens unless the dispatch reports them. Every LLM cost MUST be reported via
|
|
328
330
|
* `observe` and every token count via `observeTokens`; a dispatch that reports
|
|
@@ -341,7 +343,7 @@ interface CampaignCostMeter {
|
|
|
341
343
|
/** Accumulated token usage for this cell (zeros if never observed). */
|
|
342
344
|
tokens(): CampaignTokenUsage;
|
|
343
345
|
}
|
|
344
|
-
/**
|
|
346
|
+
/** Source tag — required on every store write. Used by the
|
|
345
347
|
* default training-source filter (production-trace samples NOT used as
|
|
346
348
|
* training scenarios unless explicitly opted in). */
|
|
347
349
|
type LabeledScenarioSource = 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
|
|
@@ -363,7 +365,7 @@ type RedactionStatus = 'raw' | 'redacted-pii' | 'redacted-secrets' | 'fully-reda
|
|
|
363
365
|
type LabelTrust = 'unverified' | 'verified-signal' | 'human-rated';
|
|
364
366
|
/** Ordinal rank for a label-trust tier; absent ⇒ `unverified` (rank 0). */
|
|
365
367
|
declare function labelTrustRank(trust: LabelTrust | undefined): number;
|
|
366
|
-
/**
|
|
368
|
+
/** Required-provenance write. The store rejects writes that
|
|
367
369
|
* lack provenance — a default-on flywheel without provenance is the
|
|
368
370
|
* data-poisoning vector flagged in the alignment review. */
|
|
369
371
|
interface LabeledScenarioWrite<TScenario extends Scenario = Scenario, TArtifact = unknown> {
|
|
@@ -455,7 +457,7 @@ interface GenerationRecord {
|
|
|
455
457
|
promoted: string[];
|
|
456
458
|
}
|
|
457
459
|
/** One scored candidate surface in a generation. `dimensions` + `scenarios`
|
|
458
|
-
* let a reflective
|
|
460
|
+
* let a reflective proposer ground its next proposal on WHICH
|
|
459
461
|
* dimensions the candidate is weakest on and WHICH scenarios it best/worst
|
|
460
462
|
* handled — the evidence a blind `Mutator` cannot see. */
|
|
461
463
|
interface GenerationCandidate {
|
|
@@ -467,7 +469,7 @@ interface GenerationCandidate {
|
|
|
467
469
|
dimensions: Record<string, number>;
|
|
468
470
|
/** Per-scenario composite (mean over reps + judges), plus the judge's
|
|
469
471
|
* free-form `notes` for that scenario — the "why it scored low" evidence a
|
|
470
|
-
* reflective
|
|
472
|
+
* reflective proposer grounds its next edit on. Keep `notes` GENERALIZABLE
|
|
471
473
|
* (which checks/lines/dimensions failed and how), NOT case-specific ground
|
|
472
474
|
* truth: leaking expected answers into the prompt is memorization, and the
|
|
473
475
|
* held-out gate would reject it anyway. */
|
|
@@ -476,12 +478,12 @@ interface GenerationCandidate {
|
|
|
476
478
|
composite: number;
|
|
477
479
|
notes?: string;
|
|
478
480
|
}>;
|
|
479
|
-
/**
|
|
481
|
+
/** Proposer-supplied short label for the change. Present when the proposer
|
|
480
482
|
* returned a `ProposedCandidate`; absent for bare-surface mutators. */
|
|
481
483
|
label?: string;
|
|
482
|
-
/**
|
|
484
|
+
/** Proposer-supplied rationale — WHY this candidate was proposed. The
|
|
483
485
|
* "because rationale Z" the audit requires to survive to the result.
|
|
484
|
-
* Present when the
|
|
486
|
+
* Present when the proposer returned a `ProposedCandidate`. */
|
|
485
487
|
rationale?: string;
|
|
486
488
|
}
|
|
487
489
|
interface CampaignAggregates {
|
|
@@ -517,4 +519,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
|
|
|
517
519
|
scenarios: Array<Pick<TScenario, 'id' | 'kind'>>;
|
|
518
520
|
}
|
|
519
521
|
|
|
520
|
-
export {
|
|
522
|
+
export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchFn as D, isProposedCandidate as E, labelTrustRank as F, type Gate as G, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizationProposer as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeConfig as a, type DispatchContext as b, type SurfaceProposer as c, type GateDecision as d, type CampaignAggregates as e, type CampaignArtifactWriter as f, type CampaignCellResult as g, type CampaignCostMeter as h, type CampaignTraceWriter as i, type CodeSurface as j, type GateContext as k, type GateResult as l, type GenerationCandidate as m, type GenerationRecord as n, type JudgeDimension as o, type Mutator as p, type OptimizerConfig as q, type SessionScript as r, type LabeledScenarioWrite as s, type LabeledScenarioSampleArgs as t, type LabeledScenarioRecord as u, type LabelTrust as v, type ProposedCandidate as w, type ProposeContext as x, type LabeledScenarioSource as y, type CampaignTokenUsage as z };
|
package/dist/workflow/index.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { W as WorkflowTopology } from '../harness-optimizer-mOl9XX_O.js';
|
|
2
|
-
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from '../run-record-
|
|
3
|
-
import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-
|
|
4
|
-
import { F as FailureClusterInsight } from '../insight-report-
|
|
2
|
+
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from '../run-record-CP2ObebC.js';
|
|
3
|
+
import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-B5x54y6n.js';
|
|
4
|
+
import { F as FailureClusterInsight } from '../insight-report-BnRjTibG.js';
|
|
5
5
|
import { a as VerificationReport, L as LayerResult } from '../multi-layer-verifier-DUZXrPDA.js';
|
|
6
6
|
import { F as FailureClusterReport } from '../failure-cluster-DH9Flgcf.js';
|
|
7
7
|
import { R as RedactionRule, a as RedactionReport } from '../redact-B40YG2M_.js';
|
|
@@ -12,13 +12,14 @@ import '../pareto-E-pembql.js';
|
|
|
12
12
|
import '../run-critic-CmMf05uV.js';
|
|
13
13
|
import '../schema-m0gsnbt3.js';
|
|
14
14
|
import '../store-BcFXE6LG.js';
|
|
15
|
+
import '@tangle-network/agent-interface';
|
|
15
16
|
import '../errors-CzMUYo7b.js';
|
|
16
17
|
import '../store-C1YxJDEK.js';
|
|
17
18
|
import '../types-C7DGg5ex.js';
|
|
18
19
|
import '@tangle-network/tcloud';
|
|
19
20
|
import '../llm-client-Bj7g0rqu.js';
|
|
20
21
|
import '../raw-provider-sink-C46HDghv.js';
|
|
21
|
-
import '../summary-report-
|
|
22
|
+
import '../summary-report-CInXwsza.js';
|
|
22
23
|
import '../judge-calibration-0p2QcWNE.js';
|
|
23
24
|
import '../verdict-C9MlYujm.js';
|
|
24
25
|
import '../control-runtime-Acf9CGhw.js';
|
package/dist/workflow/index.js
CHANGED
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
# Campaign Proposers, ELI5
|
|
2
|
+
|
|
3
|
+
A campaign proposer is the part of an improvement loop that says: "try this
|
|
4
|
+
candidate next."
|
|
5
|
+
|
|
6
|
+
It does not run your agent. It does not score anything. It only proposes a new
|
|
7
|
+
surface to measure. A surface is the thing you are changing: a prompt string, a
|
|
8
|
+
serialized config string, or a code/worktree surface.
|
|
9
|
+
|
|
10
|
+
Use **proposer** for this role. Older optimizer APIs used "driver"; that word is
|
|
11
|
+
now reserved for execution, sandbox, and router agents that actually drive
|
|
12
|
+
workers.
|
|
13
|
+
|
|
14
|
+
## The Loop
|
|
15
|
+
|
|
16
|
+
```text
|
|
17
|
+
current surface
|
|
18
|
+
-> proposer suggests candidate surfaces
|
|
19
|
+
-> runCampaign runs each candidate on scenarios
|
|
20
|
+
-> judges score the artifacts
|
|
21
|
+
-> runOptimization picks the best candidate
|
|
22
|
+
-> runImprovementLoop re-scores on holdout and gates release
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## Proposer Input
|
|
26
|
+
|
|
27
|
+
Every `SurfaceProposer.propose(ctx)` receives:
|
|
28
|
+
|
|
29
|
+
| Field | Plain meaning |
|
|
30
|
+
|---|---|
|
|
31
|
+
| `currentSurface` | The prompt/config/code surface currently being improved. |
|
|
32
|
+
| `history` | What candidates were tried before and how they scored. |
|
|
33
|
+
| `findings` | Failure analysis or analyst findings from traces/eval runs. |
|
|
34
|
+
| `populationSize` | How many candidates the loop asks for this generation. |
|
|
35
|
+
| `generation` | Which generation number this is. |
|
|
36
|
+
| `signal` | Abort signal for cancellation. |
|
|
37
|
+
| `report` | Optional larger analysis report. |
|
|
38
|
+
| `dataset` | Optional labeled scenario store. |
|
|
39
|
+
| `paretoParents` | Optional non-dominated surfaces from prior generations. |
|
|
40
|
+
|
|
41
|
+
## Proposer Output
|
|
42
|
+
|
|
43
|
+
A proposer returns candidates:
|
|
44
|
+
|
|
45
|
+
```ts
|
|
46
|
+
{
|
|
47
|
+
surface: 'the full new prompt or config',
|
|
48
|
+
label: 'short name',
|
|
49
|
+
rationale: 'why this change should help'
|
|
50
|
+
}
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
Bare surfaces are still accepted, but `label` and `rationale` make results
|
|
54
|
+
auditable, so new proposers should return `ProposedCandidate`.
|
|
55
|
+
|
|
56
|
+
## Which Proposer To Use
|
|
57
|
+
|
|
58
|
+
| Proposer factory | Best when | Output surface |
|
|
59
|
+
|---|---|---|
|
|
60
|
+
| `gepaProposer` | You want a strong prompt rewrite driven by prior scores and findings. | prompt string |
|
|
61
|
+
| `skillOptProposer` | You are editing a structured skill/runbook and want anchored small patches. | prompt/skill string |
|
|
62
|
+
| `aceProposer` | You want append-only lessons from findings, preserving every distinct lesson. | prompt/playbook string |
|
|
63
|
+
| `memoryCurationProposer` | You want compact deduped lessons from findings. | prompt/playbook string |
|
|
64
|
+
| `parameterSweepProposer` | You want FAPO-style config/parameter edits from a JSON config surface. | JSON string |
|
|
65
|
+
| `fapoProposer` | You want the FAPO policy: prompt first, then parameter, then structural only when evidence supports escalation. | whatever its level proposer returns |
|
|
66
|
+
|
|
67
|
+
## FAPO Proposer
|
|
68
|
+
|
|
69
|
+
FAPO is not "another prompt mutator." The paper describes a reviewed escalation
|
|
70
|
+
policy:
|
|
71
|
+
|
|
72
|
+
1. evaluate the current workflow,
|
|
73
|
+
2. attribute failures to prompt, parameter/config, or structure,
|
|
74
|
+
3. propose one scoped change,
|
|
75
|
+
4. review the change for scope/leakage/compatibility,
|
|
76
|
+
5. measure it,
|
|
77
|
+
6. keep moving or escalate only when the cheaper level is exhausted.
|
|
78
|
+
|
|
79
|
+
The simplest useful setup is prompt plus JSON config. Structural/code edits are
|
|
80
|
+
optional and should be injected by the app or runtime layer.
|
|
81
|
+
|
|
82
|
+
```ts
|
|
83
|
+
import {
|
|
84
|
+
fapoProposer,
|
|
85
|
+
gepaProposer,
|
|
86
|
+
parameterSweepProposer,
|
|
87
|
+
runImprovementLoop,
|
|
88
|
+
} from '@tangle-network/agent-eval/campaign'
|
|
89
|
+
|
|
90
|
+
const proposer = fapoProposer({
|
|
91
|
+
scope: { allowedLevels: ['prompt', 'parameter'] },
|
|
92
|
+
promptProposer: gepaProposer({ llm, model, target: 'agent prompt' }),
|
|
93
|
+
parameterProposer: parameterSweepProposer({
|
|
94
|
+
candidates: [
|
|
95
|
+
{
|
|
96
|
+
label: 'raise-retrieval-k',
|
|
97
|
+
rationale: 'retrieval misses indicate the search budget may be too low',
|
|
98
|
+
changes: [{ path: 'retrieval.k', value: 10 }],
|
|
99
|
+
},
|
|
100
|
+
],
|
|
101
|
+
}),
|
|
102
|
+
})
|
|
103
|
+
|
|
104
|
+
await runImprovementLoop({
|
|
105
|
+
scenarios: trainScenarios,
|
|
106
|
+
holdoutScenarios,
|
|
107
|
+
baselineSurface: JSON.stringify(currentConfig),
|
|
108
|
+
dispatchWithSurface,
|
|
109
|
+
judges,
|
|
110
|
+
proposer,
|
|
111
|
+
gate,
|
|
112
|
+
autoOnPromote: 'none',
|
|
113
|
+
runDir,
|
|
114
|
+
populationSize: 1,
|
|
115
|
+
maxGenerations: 10,
|
|
116
|
+
})
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
If you do have a real code/worktree proposer, pass it as `structuralProposer`.
|
|
120
|
+
`agent-eval` intentionally does not provide that proposer because this package
|
|
121
|
+
measures candidates; the runtime or app owns code generation.
|
|
122
|
+
|
|
123
|
+
For side-by-side experiments with existing proposers, use the compare entry:
|
|
124
|
+
|
|
125
|
+
```ts
|
|
126
|
+
import {
|
|
127
|
+
compareProposers,
|
|
128
|
+
fapoEscalationEntry,
|
|
129
|
+
gepaParetoEntry,
|
|
130
|
+
} from '@tangle-network/agent-eval/campaign'
|
|
131
|
+
|
|
132
|
+
await compareProposers({
|
|
133
|
+
proposers: [
|
|
134
|
+
gepaParetoEntry(config),
|
|
135
|
+
fapoEscalationEntry({
|
|
136
|
+
...config,
|
|
137
|
+
parameterCandidates: [
|
|
138
|
+
{
|
|
139
|
+
label: 'raise-retrieval-k',
|
|
140
|
+
rationale: 'retrieval misses indicate the search budget may be too low',
|
|
141
|
+
changes: [{ path: 'retrieval.k', value: 10 }],
|
|
142
|
+
},
|
|
143
|
+
],
|
|
144
|
+
}),
|
|
145
|
+
],
|
|
146
|
+
baselineSurface,
|
|
147
|
+
holdoutScenarios,
|
|
148
|
+
dispatchWithSurface,
|
|
149
|
+
judges,
|
|
150
|
+
runDir,
|
|
151
|
+
})
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
## Common Mistakes
|
|
155
|
+
|
|
156
|
+
- Do not put eval logic inside a proposer. Put it in `dispatch` and `judges`.
|
|
157
|
+
- Do not let a proposer read held-out judge scores. `ProposeContext` makes this
|
|
158
|
+
a type-level firewall.
|
|
159
|
+
- Do not call FAPO a prompt-only optimizer. Its main value is evidence-based
|
|
160
|
+
escalation beyond prompt edits.
|
|
161
|
+
- Do not put Claude Code or sandbox-specific code in `agent-eval`. Structural
|
|
162
|
+
code generation should be supplied as an injected `SurfaceProposer` from the
|
|
163
|
+
runtime/app layer.
|
|
164
|
+
|
|
165
|
+
## Simpler Mental Model
|
|
166
|
+
|
|
167
|
+
Use this sentence when wiring a loop:
|
|
168
|
+
|
|
169
|
+
> The proposer chooses candidates; the campaign measures them; the gate decides
|
|
170
|
+
> whether the measured winner is safe to promote.
|
package/docs/concepts.md
CHANGED
|
@@ -9,17 +9,21 @@ connected, or the answer lacks required sources. The package gives products a
|
|
|
9
9
|
shared way to record runs, check outcomes, classify failures, compare variants,
|
|
10
10
|
and make release decisions.
|
|
11
11
|
|
|
12
|
-
## The
|
|
12
|
+
## The top-level functions
|
|
13
13
|
|
|
14
|
-
Everything funnels through `/contract`.
|
|
14
|
+
Everything funnels through `/contract`. Start with `defineAgentEval()` when you
|
|
15
|
+
can; drop to the raw functions when you need lower-level control.
|
|
15
16
|
|
|
16
17
|
| Function | When to call it | What you give it | What you get back |
|
|
17
18
|
|---|---|---|---|
|
|
19
|
+
| **`defineAgentEval()`** | You have scenarios, an agent, a judge, and a baseline surface, and you want one object you can score or improve. | scenarios, agent, judge, baseline surface | `{ evaluate(), improve() }` where `evaluate()` returns a campaign result and `improve()` returns a decision packet |
|
|
18
20
|
| **`selfImprove()`** | You have a closed loop — scenarios, judge, agent in hand, and you want the substrate to propose better candidates + gate them. | scenarios, agent, judge, baseline surface | `SelfImproveResult.insight: InsightReport` + ship/hold verdict + winner surface |
|
|
19
21
|
| **`analyzeRuns()`** | You have observed runs (production traces, an approve/reject corpus, a CSV gold set) and want the same rigor packet without invoking an agent. | `RunRecord[]` + optional flags | `InsightReport` |
|
|
20
22
|
| **Intake adapters** (`fromFeedbackTable`, `fromOtelSpans`) | Your data isn't already in `RunRecord` shape — it's in Obsidian, Sheets, an OTel collector, etc. | source-specific input | `RunRecord[]` ready to pipe into `analyzeRuns()` |
|
|
21
23
|
|
|
22
|
-
The
|
|
24
|
+
The customer maturity stages — logs only → ratings → closed loop — map to these
|
|
25
|
+
entry points. See [`customer-journeys.md`](./customer-journeys.md) for the
|
|
26
|
+
runnable walkthroughs.
|
|
23
27
|
|
|
24
28
|
The shape of the answer — `InsightReport` — is identical across all three paths. Distributional summary, paired-bootstrap lift CI, judge stats, inter-rater agreement, cost-quality Pareto, failure clusters, contamination check, outcome correlation, release axes, and a ranked recommendations array. Walked through section-by-section in [`insight-report.md`](./insight-report.md).
|
|
25
29
|
|
|
@@ -182,7 +186,7 @@ release decision.
|
|
|
182
186
|
|
|
183
187
|
## Where to go next
|
|
184
188
|
|
|
185
|
-
- **Confused by "GEPA / HALO / trace analysis /
|
|
189
|
+
- **Confused by "GEPA / HALO / trace analysis / proposers everywhere"?** → [self-improvement-map.md](./self-improvement-map.md) — one loop, four roles, the proposer catalog (production vs bench-only), and why `gepa-refine` is the same loop on a test bench.
|
|
186
190
|
- **Which `run*` primitive do I use, and how do I grade produced state?** → [eval-surface-map.md](./eval-surface-map.md) — the campaign/matrix/optimization/gate primitives as a pick-by-"use-when" table, plus the produced-state grading composition (verifyCompletion-as-judge — there is no persona-dispatch wrapper) and the in-band body contract.
|
|
187
191
|
- **Need the layman feature map?** → [feature-guide.md](./feature-guide.md) — what each primitive does, when to use it, integration patterns, and guardrails.
|
|
188
192
|
- **Just want to score a string against a rubric?** → [wire-protocol.md](./wire-protocol.md) — HTTP/RPC interface, pluggable from any language.
|
|
@@ -58,7 +58,7 @@ Cost mean: $0.103 (p95: $0.131)
|
|
|
58
58
|
|
|
59
59
|
1. Wire an `AnalystRegistry` to cluster the 6 failures by root cause via LLM analysis.
|
|
60
60
|
2. Add `outcomeSignal` once they have downstream conversion / engagement / post-engagement data, and the report fits a reward model showing whether their score predicts the customer outcome.
|
|
61
|
-
3. Once they identify a step worth optimizing (translation, say), graduate to journey #3 — wrap that step
|
|
61
|
+
3. Once they identify a step worth optimizing (translation, say), graduate to journey #3 — wrap that step as an `agent(surface, scenario)` and call `defineAgentEval()`.
|
|
62
62
|
|
|
63
63
|
**Runnable:** [`examples/customer-otel-traces/`](../examples/customer-otel-traces/)
|
|
64
64
|
|
|
@@ -130,7 +130,7 @@ Mean κ: 0.43
|
|
|
130
130
|
1. **Triage meeting on the disagreement cases.** Mean κ=0.43 means the rubric is ambiguous; clarify it on the cases that split.
|
|
131
131
|
2. **Calibrate one LLM judge per reviewer.** Each reviewer's history is the gold signal — substrate primitive `calibrateJudge` against `raterScores` filtered to that reviewer.
|
|
132
132
|
3. **Add engagement as `outcomeSignal`** once the content downstream is instrumented. The `outcomeCorrelation` section tells the team whether their taste predicts the founder's token-max goal — and if not, the linear reward model says how to retarget.
|
|
133
|
-
4. **Graduate to journey #3** — wrap the research-generation Claude-P call as
|
|
133
|
+
4. **Graduate to journey #3** — wrap the research-generation Claude-P call as an `agent(surface, scenario)`, use the calibrated judges, run `evalKit.improve()` nightly. Open a PR against the GitHub Action when the holdout approval rate beats baseline.
|
|
134
134
|
|
|
135
135
|
**Runnable:** [`examples/customer-feedback-loop/`](../examples/customer-feedback-loop/)
|
|
136
136
|
|
|
@@ -142,14 +142,14 @@ Mean κ: 0.43
|
|
|
142
142
|
|
|
143
143
|
**The frustration:** "We can run an A/B by hand but we don't know if the improvement is real. We don't have time to run paired bootstrap by hand. We want a function that decides."
|
|
144
144
|
|
|
145
|
-
**What they need from agent-eval:**
|
|
145
|
+
**What they need from agent-eval:** one reusable eval definition — propose, score, gate, ship — with the full rigor packet on the way out.
|
|
146
146
|
|
|
147
147
|
### The code
|
|
148
148
|
|
|
149
149
|
```ts
|
|
150
|
-
import {
|
|
150
|
+
import { defineAgentEval } from '@tangle-network/agent-eval/contract'
|
|
151
151
|
|
|
152
|
-
const
|
|
152
|
+
const evalKit = defineAgentEval({
|
|
153
153
|
scenarios,
|
|
154
154
|
agent: async (surface, scenario) =>
|
|
155
155
|
await myAgent.run({ systemPrompt: (surface as { systemPrompt: string }).systemPrompt, scenario }),
|
|
@@ -162,6 +162,8 @@ const result = await selfImprove({
|
|
|
162
162
|
budget: { generations: 3, populationSize: 2 },
|
|
163
163
|
})
|
|
164
164
|
|
|
165
|
+
const result = await evalKit.improve()
|
|
166
|
+
|
|
165
167
|
result.gateDecision // 'ship' | 'hold' | ...
|
|
166
168
|
result.insight // full decision packet
|
|
167
169
|
```
|
|
@@ -172,18 +174,18 @@ result.insight // full decision packet
|
|
|
172
174
|
═══ selfImprove() decision packet ═══
|
|
173
175
|
|
|
174
176
|
Gate decision: ship
|
|
175
|
-
Raw lift: +0.
|
|
177
|
+
Raw lift: +0.361
|
|
176
178
|
|
|
177
179
|
── Statistical lift (paired bootstrap) ──
|
|
178
|
-
delta: +0.
|
|
179
|
-
CI95: [0.
|
|
180
|
-
pValue:
|
|
181
|
-
Cohen's d:
|
|
182
|
-
MDE @ 80% power:
|
|
183
|
-
required n at observed effect:
|
|
180
|
+
delta: +0.359
|
|
181
|
+
CI95: [0.311, 0.408]
|
|
182
|
+
pValue: 0.0013
|
|
183
|
+
Cohen's d: 8.58
|
|
184
|
+
MDE @ 80% power: 1.401
|
|
185
|
+
required n at observed effect: 122
|
|
184
186
|
|
|
185
187
|
── Recommendations ──
|
|
186
|
-
[critical] ship — Ship — lift 0.
|
|
188
|
+
[critical] ship — Ship — lift 0.359 (95% CI 0.311..0.408)
|
|
187
189
|
```
|
|
188
190
|
|
|
189
191
|
### Next steps for this customer
|