@tangle-network/agent-eval 0.94.0 → 0.95.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -0
- package/README.md +44 -30
- package/dist/adapters/http.d.ts +8 -7
- package/dist/adapters/http.js.map +1 -1
- package/dist/adapters/langchain.d.ts +3 -2
- package/dist/adapters/otel.d.ts +5 -4
- package/dist/analyst/index.d.ts +11 -31
- package/dist/analyst/index.js +5 -65
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-B6Ljo_dI.d.ts → analyze-runs-DtT6F_6T.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +4 -3
- package/dist/benchmarks/index.d.ts +3 -2
- package/dist/campaign/index.d.ts +727 -616
- package/dist/campaign/index.js +1863 -1316
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-2K6UUZ7P.js → chunk-2T4EZACH.js} +1 -1
- package/dist/chunk-2T4EZACH.js.map +1 -0
- package/dist/{chunk-CTBHKLEU.js → chunk-77T4STFI.js} +59 -86
- package/dist/chunk-77T4STFI.js.map +1 -0
- package/dist/{chunk-EGPMSBEZ.js → chunk-7QTQKIDD.js} +178 -177
- package/dist/chunk-7QTQKIDD.js.map +1 -0
- package/dist/{chunk-MIFZUPEK.js → chunk-AQ5WQAIV.js} +21 -6
- package/dist/chunk-AQ5WQAIV.js.map +1 -0
- package/dist/chunk-DJWX3GVS.js +81 -0
- package/dist/chunk-DJWX3GVS.js.map +1 -0
- package/dist/{chunk-S6OZEZQK.js → chunk-HMA63UEO.js} +37 -9
- package/dist/{chunk-S6OZEZQK.js.map → chunk-HMA63UEO.js.map} +1 -1
- package/dist/{chunk-TBDR6PAI.js → chunk-IZCEK2HR.js} +2 -2
- package/dist/{chunk-SD2YFWQQ.js → chunk-KKWJD5E6.js} +20 -20
- package/dist/chunk-KKWJD5E6.js.map +1 -0
- package/dist/{chunk-KWRRMR3J.js → chunk-LO6IOIJ2.js} +10 -10
- package/dist/chunk-LO6IOIJ2.js.map +1 -0
- package/dist/{chunk-E4GH6USR.js → chunk-NZEQVRH5.js} +2 -2
- package/dist/chunk-NZEQVRH5.js.map +1 -0
- package/dist/{chunk-MPQWFX6Y.js → chunk-PSWWQXHF.js} +13 -88
- package/dist/chunk-PSWWQXHF.js.map +1 -0
- package/dist/{chunk-Q5LIB7BC.js → chunk-S4SYLDFX.js} +2 -2
- package/dist/chunk-S4SYLDFX.js.map +1 -0
- package/dist/{chunk-KW53MSA5.js → chunk-X74V6ESX.js} +2 -2
- package/dist/{chunk-QMUEXQJS.js → chunk-YBIGNSCZ.js} +81 -4
- package/dist/chunk-YBIGNSCZ.js.map +1 -0
- package/dist/{chunk-2KNZHH3P.js → chunk-Z6L6YSU6.js} +2 -2
- package/dist/{code-agent-session-BO8nCnv3.d.ts → code-agent-session-CPHRCb4-.d.ts} +1 -1
- package/dist/contract/index.d.ts +91 -43
- package/dist/contract/index.js +127 -17
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-D6qwHXIR.d.ts → control-Doncu-B_.d.ts} +2 -2
- package/dist/control.d.ts +3 -2
- package/dist/control.js +2 -2
- package/dist/{corpus-B8A4BDR3.d.ts → corpus-D4YW9UoJ.d.ts} +1 -1
- package/dist/{default-registry-6dhErQbs.d.ts → default-registry-GyE8X5SP.d.ts} +3 -3
- package/dist/diagnose.d.ts +4 -3
- package/dist/diagnose.js +1 -1
- package/dist/{run-improvement-loop-DBahB8Ax.d.ts → gepa-C1NCIZ9o.d.ts} +117 -130
- package/dist/hosted/index.d.ts +5 -4
- package/dist/{index-Bx3gZ8xl.d.ts → index-_Y4oNOOb.d.ts} +1 -1
- package/dist/index.d.ts +76 -81
- package/dist/index.js +66 -31
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DWl3z9tl.d.ts → insight-report-BnRjTibG.d.ts} +1 -1
- package/dist/{kind-factory-0BhLSI27.d.ts → kind-factory-X3eDYbKn.d.ts} +2 -3
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +3 -2
- package/dist/multishot/index.d.ts +4 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{pre-registration-mAnCugl9.d.ts → pre-registration-nfUdc9EQ.d.ts} +2 -42
- package/dist/{provenance-P-bCL2Fo.d.ts → provenance-CncDq9qE.d.ts} +26 -41
- package/dist/{release-report-BEbWmVYj.d.ts → release-report-pidWUMZ2.d.ts} +2 -2
- package/dist/reporting.d.ts +5 -4
- package/dist/{researcher-B0C2_fVO.d.ts → researcher-Jr8ME1dZ.d.ts} +2 -2
- package/dist/rl.d.ts +516 -515
- package/dist/rl.js +612 -612
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-Cy_W-hWZ.d.ts → rubric-predictive-validity-C2hDKM8Z.d.ts} +1 -1
- package/dist/{run-campaign-7WNXMDSN.js → run-campaign-WXY7KI67.js} +2 -2
- package/dist/{run-record-e7vj1uZQ.d.ts → run-record-CP2ObebC.d.ts} +14 -18
- package/dist/{runtime-trajectory-BDgfGZSr.d.ts → runtime-trajectory-BOUUjI0y.d.ts} +1 -1
- package/dist/{semantic-concept-judge-B9MgmBnM.d.ts → semantic-concept-judge-DSBB2Cfp.d.ts} +2 -2
- package/dist/{summary-report-BDOFevaT.d.ts → summary-report-CInXwsza.d.ts} +1 -1
- package/dist/testing-C21CHsq2.d.ts +20 -0
- package/dist/testing.d.ts +1 -0
- package/dist/testing.js +8 -0
- package/dist/testing.js.map +1 -0
- package/dist/traces.d.ts +26 -10
- package/dist/traces.js +41 -11
- package/dist/{types-Ce17tDlG.d.ts → types-B5x54y6n.d.ts} +1 -1
- package/dist/{types-mn5Aqk7x.d.ts → types-BUxNaJ8c.d.ts} +2 -4
- package/dist/{types-BU-7W85F.d.ts → types-DQRY8ZT-.d.ts} +60 -58
- package/dist/workflow/index.d.ts +5 -4
- package/dist/workflow/index.js +1 -1
- package/docs/campaign-proposers.md +170 -0
- package/docs/concepts.md +8 -4
- package/docs/customer-journeys.md +15 -13
- package/docs/design/loop-taxonomy.md +34 -66
- package/docs/distributed-driver.md +14 -14
- package/docs/feature-guide.md +1 -1
- package/docs/hosted-ingest-spec.md +2 -3
- package/docs/multi-shot-optimization.md +8 -8
- package/docs/product-eval-adoption.md +1 -1
- package/docs/self-improvement-map.md +33 -29
- package/package.json +8 -14
- package/dist/chunk-2K6UUZ7P.js.map +0 -1
- package/dist/chunk-CTBHKLEU.js.map +0 -1
- package/dist/chunk-E4GH6USR.js.map +0 -1
- package/dist/chunk-EGPMSBEZ.js.map +0 -1
- package/dist/chunk-KWRRMR3J.js.map +0 -1
- package/dist/chunk-MIFZUPEK.js.map +0 -1
- package/dist/chunk-MPQWFX6Y.js.map +0 -1
- package/dist/chunk-Q5LIB7BC.js.map +0 -1
- package/dist/chunk-QMUEXQJS.js.map +0 -1
- package/dist/chunk-SD2YFWQQ.js.map +0 -1
- package/docs/design/external-agent-wedge.md +0 -89
- package/docs/design/phase-d-rfc.md +0 -125
- package/docs/design/phase4-consumer-migration.md +0 -70
- package/docs/design/primitives-integration-spec.md +0 -393
- package/docs/design/product-self-improvement-loop.md +0 -146
- package/docs/design/self-improvement-engine.md +0 -140
- package/docs/design/self-improvement-protocol.md +0 -223
- package/docs/design/self-improvement-roadmap.md +0 -106
- package/docs/design/substrate-gaps.md +0 -118
- package/docs/phase-b-pairing-kit.md +0 -188
- package/docs/phase-b-runbook.md +0 -176
- package/docs/pilot/README.md +0 -62
- package/docs/pilot/customer-checklist.md +0 -90
- package/docs/pilot/integration-foreign-stack.md +0 -296
- package/docs/pilot/integration-tangle-stack.md +0 -248
- package/docs/pilot/one-pager.md +0 -161
- package/docs/pilot/sample-insight-report.json +0 -172
- package/docs/quickstart-external.md +0 -229
- package/docs/research/belief-state-agent-eval-roadmap.md +0 -593
- package/docs/research/research-roadmap.md +0 -205
- package/docs/specs/driver-honest-spec.md +0 -251
- package/docs/specs/hermes-self-improvement-audit.md +0 -93
- package/docs/specs/profile-versioning.md +0 -291
- package/docs/three-package-architecture.md +0 -168
- /package/dist/{chunk-TBDR6PAI.js.map → chunk-IZCEK2HR.js.map} +0 -0
- /package/dist/{chunk-KW53MSA5.js.map → chunk-X74V6ESX.js.map} +0 -0
- /package/dist/{chunk-2KNZHH3P.js.map → chunk-Z6L6YSU6.js.map} +0 -0
- /package/dist/{run-campaign-7WNXMDSN.js.map → run-campaign-WXY7KI67.js.map} +0 -0
package/dist/campaign/index.d.ts
CHANGED
|
@@ -1,16 +1,17 @@
|
|
|
1
1
|
import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
|
|
2
|
-
import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig,
|
|
3
|
-
export {
|
|
4
|
-
import { b as RunCampaignOptions, c as RunImprovementLoopOptions, C as CampaignStorage } from '../
|
|
5
|
-
export { e as
|
|
6
|
-
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, k as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, l as EmitLoopProvenanceArgs, m as EmitLoopProvenanceResult, E as EvidenceVector, b as
|
|
7
|
-
import {
|
|
8
|
-
import { T as TraceAnalystKindSpec } from '../kind-factory-0BhLSI27.js';
|
|
9
|
-
import { c as AnalystFinding } from '../types-Ce17tDlG.js';
|
|
10
|
-
import { S as SignedManifest, A as AgentProfile, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../pre-registration-mAnCugl9.js';
|
|
2
|
+
import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, n as GenerationRecord, G as Gate, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, c as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, y as LabeledScenarioSource, C as CampaignResult, j as CodeSurface } from '../types-DQRY8ZT-.js';
|
|
3
|
+
export { e as CampaignAggregates, f as CampaignArtifactWriter, g as CampaignCellResult, h as CampaignCostMeter, z as CampaignTokenUsage, i as CampaignTraceWriter, D as DispatchFn, k as GateContext, d as GateDecision, l as GateResult, m as GenerationCandidate, A as JudgeAggregate, o as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-DQRY8ZT-.js';
|
|
4
|
+
import { b as RunCampaignOptions, c as RunImprovementLoopOptions, C as CampaignStorage } from '../gepa-C1NCIZ9o.js';
|
|
5
|
+
export { e as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, h as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, f as fsCampaignStorage, g as gepaProposer, i as inMemoryCampaignStorage, o as openAutoPr, r as runCampaign, d as runImprovementLoop, n as runOptimization, s as surfaceHash } from '../gepa-C1NCIZ9o.js';
|
|
6
|
+
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, k as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, l as EmitLoopProvenanceArgs, m as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, n as LoopProvenanceBackend, o as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, P as ParetoSignificanceGateOptions, c as PromotionObjective, d as PromotionPolicy, R as RunEvalOptions, e as buildEvidenceVector, q as buildLoopProvenanceRecord, f as composeGate, g as defaultProductionGate, s as emitLoopProvenance, h as evolutionaryProposer, i as heldOutGate, t as loopProvenanceSpans, p as paretoPolicy, j as paretoSignificanceGate, u as provenanceRecordPath, v as provenanceSpansPath, r as runEval, w as surfaceContentHash } from '../provenance-CncDq9qE.js';
|
|
7
|
+
import { S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../pre-registration-nfUdc9EQ.js';
|
|
11
8
|
import { E as EProcessState, a as PairedBootstrapResult } from '../statistics-CCJpTGOS.js';
|
|
9
|
+
import { L as LlmClientOptions } from '../llm-client-Bj7g0rqu.js';
|
|
10
|
+
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
12
11
|
import { A as AgentEvalError } from '../errors-CzMUYo7b.js';
|
|
13
|
-
import { a as RunSplitTag, R as RunRecord } from '../run-record-
|
|
12
|
+
import { a as RunSplitTag, R as RunRecord } from '../run-record-CP2ObebC.js';
|
|
13
|
+
import { T as TraceAnalystKindSpec } from '../kind-factory-X3eDYbKn.js';
|
|
14
|
+
import { c as AnalystFinding } from '../types-B5x54y6n.js';
|
|
14
15
|
import '@ax-llm/ax';
|
|
15
16
|
import '../store-C1YxJDEK.js';
|
|
16
17
|
import '../red-team-BWdoyleI.js';
|
|
@@ -19,26 +20,24 @@ import '../store-BcFXE6LG.js';
|
|
|
19
20
|
import '../schema-m0gsnbt3.js';
|
|
20
21
|
import '../pareto-E-pembql.js';
|
|
21
22
|
import '../hosted/index.js';
|
|
22
|
-
import '../insight-report-
|
|
23
|
-
import '../summary-report-
|
|
23
|
+
import '../insight-report-BnRjTibG.js';
|
|
24
|
+
import '../summary-report-CInXwsza.js';
|
|
24
25
|
import '../failure-cluster-DH9Flgcf.js';
|
|
25
26
|
import '../judge-calibration-0p2QcWNE.js';
|
|
26
|
-
import '../raw-provider-sink-C46HDghv.js';
|
|
27
|
-
import 'zod';
|
|
28
|
-
import '../types-C7DGg5ex.js';
|
|
29
27
|
import '@tangle-network/tcloud';
|
|
30
28
|
import '../verdict-C9MlYujm.js';
|
|
29
|
+
import '../types-C7DGg5ex.js';
|
|
30
|
+
import '../raw-provider-sink-C46HDghv.js';
|
|
31
|
+
import 'zod';
|
|
31
32
|
|
|
32
33
|
/**
|
|
33
|
-
* @experimental
|
|
34
|
-
*
|
|
35
34
|
* Make the trace-analyst's OWN prompt a GEPA-optimizable surface.
|
|
36
35
|
*
|
|
37
36
|
* The analyst that drives self-improvement is itself a prompt — and a
|
|
38
37
|
* hand-tuned one (a hardcoded, hand-versioned `const`). This module lets the
|
|
39
38
|
* loop optimize it: the analyst `actorDescription` becomes a `MutableSurface`
|
|
40
|
-
* that `
|
|
41
|
-
* `runImprovementLoop` or `
|
|
39
|
+
* that `gepaProposer` / `haloProposer` / any `SurfaceProposer` can mutate inside
|
|
40
|
+
* `runImprovementLoop` or `compareProposers`. That is the second-order loop —
|
|
42
41
|
* optimizing the optimizer's eyes, not just the agent's prompt.
|
|
43
42
|
*
|
|
44
43
|
* Two pieces, both deliberately small (the loop engine already exists — this
|
|
@@ -64,7 +63,7 @@ import '../verdict-C9MlYujm.js';
|
|
|
64
63
|
* holdoutScenarios: heldOutScenarios,
|
|
65
64
|
* dispatchWithSurface,
|
|
66
65
|
* judges: [failureModeRecallJudge()],
|
|
67
|
-
*
|
|
66
|
+
* proposer: gepaProposer({ baseUrl, apiKey }),
|
|
68
67
|
* gate: heldOutGate({ minDelta: 0.02 }),
|
|
69
68
|
* autoOnPromote: 'none',
|
|
70
69
|
* })
|
|
@@ -139,493 +138,162 @@ interface FailureModeRecallJudgeOptions {
|
|
|
139
138
|
declare function failureModeRecallJudge(opts?: FailureModeRecallJudgeOptions): JudgeConfig<AnalystArtifact, AnalystScenario>;
|
|
140
139
|
|
|
141
140
|
/**
|
|
142
|
-
*
|
|
143
|
-
*
|
|
144
|
-
*
|
|
145
|
-
*
|
|
146
|
-
*
|
|
147
|
-
*
|
|
148
|
-
*
|
|
149
|
-
*
|
|
141
|
+
* Anytime-valid sequential promotion gate — an e-process (betting
|
|
142
|
+
* test-martingale, see `eProcess` in `statistics.ts`) over paired
|
|
143
|
+
* per-scenario deltas, so a campaign stops the MOMENT evidence decides
|
|
144
|
+
* instead of burning a fixed-n budget. Decisions remain valid at any
|
|
145
|
+
* data-dependent stopping time (Ville's inequality), which is exactly what
|
|
146
|
+
* fixed-n machinery cannot offer: peeking at a bootstrap CI after every
|
|
147
|
+
* observation and stopping on the first significant peek inflates type-I
|
|
148
|
+
* error far beyond alpha.
|
|
150
149
|
*
|
|
151
|
-
*
|
|
152
|
-
*
|
|
153
|
-
*
|
|
154
|
-
*
|
|
155
|
-
* a genuinely NEW lesson always is, even if similar to an old one);
|
|
156
|
-
* 3. appends the new lessons as `- [gN] <lesson>` deltas and re-emits the block.
|
|
150
|
+
* REPLACES, never layers on, a fixed-n gate. Running `heldoutSignificance`
|
|
151
|
+
* or `paretoSignificanceGate` repeatedly on a growing sample and stopping
|
|
152
|
+
* early is optional stopping no matter how it is dressed up; this gate is
|
|
153
|
+
* the valid way to stop early. Use one or the other per evidence stream.
|
|
157
154
|
*
|
|
158
|
-
*
|
|
159
|
-
*
|
|
160
|
-
*
|
|
161
|
-
*
|
|
155
|
+
* Pre-registration binding: anytime validity holds only for the
|
|
156
|
+
* PRE-REGISTERED statistic. When a `SignedManifest` is bound, the gate takes
|
|
157
|
+
* alpha from `manifest.alpha`, the observation budget from
|
|
158
|
+
* `manifest.preRegisteredN`, orients deltas by `manifest.direction`, and
|
|
159
|
+
* shifts the null boundary by `manifest.minEffect` — re-deciding the same
|
|
160
|
+
* stream under different parameters after seeing data would reopen optional
|
|
161
|
+
* stopping under a fancier name. The manifest's content hash is verified at
|
|
162
|
+
* construction (sync, same `sha256-content` scheme as `signManifest`).
|
|
162
163
|
*
|
|
163
|
-
*
|
|
164
|
-
*
|
|
164
|
+
* Non-iid caveat (stated honestly): the supermartingale guarantee needs each
|
|
165
|
+
* delta's conditional mean under H0 to stay ≤ the null boundary given the
|
|
166
|
+
* past — exchangeable scenario deltas suffice. Scenario streams ordered by
|
|
167
|
+
* difficulty or by scenario family violate this; `decide(ctx)` therefore
|
|
168
|
+
* shuffles the paired deltas with a SEEDED permutation by default (the
|
|
169
|
+
* permutation is data-independent, so bet predictability is preserved).
|
|
170
|
+
* Stratified betting (per-stratum λ) is future work, not implemented here.
|
|
165
171
|
*/
|
|
166
172
|
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
+
type SequentialDecision = 'promote' | 'continue' | 'undecided-at-maxN';
|
|
174
|
+
interface SequentialObservation {
|
|
175
|
+
decision: SequentialDecision;
|
|
176
|
+
/** Current e-value (the betting wealth) against H0. */
|
|
177
|
+
eValue: number;
|
|
178
|
+
/** Paired deltas consumed so far. */
|
|
179
|
+
n: number;
|
|
180
|
+
/** Names the decision basis. For 'undecided-at-maxN' it states explicitly
|
|
181
|
+
* that exhausting the budget is NOT evidence of no effect. */
|
|
182
|
+
reason: string;
|
|
183
|
+
}
|
|
184
|
+
interface SequentialPairedGateOptions {
|
|
185
|
+
/** Type-I budget. With `preRegistration` bound this MUST match
|
|
186
|
+
* `manifest.alpha` (conflict throws). Default 0.05. */
|
|
187
|
+
alpha?: number;
|
|
188
|
+
/** Minimum paired deltas before a promote may fire. The stopping rule is
|
|
189
|
+
* "first n ≥ minN with e-value ≥ 1/alpha" — still a valid stopping time.
|
|
190
|
+
* Default 5. */
|
|
191
|
+
minN?: number;
|
|
192
|
+
/** Pre-registered observation budget. Required unless `preRegistration`
|
|
193
|
+
* supplies it via `preRegisteredN` (conflict throws). */
|
|
194
|
+
maxN?: number;
|
|
195
|
+
/** Bet truncation forwarded to `eProcess`. Default 0.5. */
|
|
196
|
+
maxBet?: number;
|
|
197
|
+
/** Bound on |delta| in the judge's native scale; deltas are mapped to
|
|
198
|
+
* x = (d/scale + 1)/2 ∈ [0,1]. A delta outside ±scale throws (use
|
|
199
|
+
* `detectScale` to pick 1 vs 100 BEFORE streaming). Default 1. */
|
|
200
|
+
scale?: number;
|
|
201
|
+
/** Seed for the data-independent shuffle of paired deltas in `decide(ctx)`
|
|
202
|
+
* (exchangeability guard). Default 1337. */
|
|
203
|
+
shuffleSeed?: number;
|
|
204
|
+
/** Bind the pre-registered hypothesis. Verified (content hash) at
|
|
205
|
+
* construction; alpha/maxN/direction/minEffect come FROM the manifest. */
|
|
206
|
+
preRegistration?: SignedManifest;
|
|
207
|
+
/** Override the gate name in reports. */
|
|
208
|
+
name?: string;
|
|
209
|
+
}
|
|
210
|
+
interface SequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario> extends Gate<TArtifact, TScenario> {
|
|
211
|
+
/** Streaming entry point: feed one paired per-scenario delta
|
|
212
|
+
* (candidate − baseline, native scale). Each gate instance carries ONE
|
|
213
|
+
* observe-stream; `decide(ctx)` runs on its own fresh stream and never
|
|
214
|
+
* consumes or advances this one. 'promote' is sticky; observing past the
|
|
215
|
+
* pre-registered maxN throws (extending a finished stream after seeing
|
|
216
|
+
* the result reopens optional stopping — start a NEW pre-registered
|
|
217
|
+
* test). */
|
|
218
|
+
observe(delta: number): SequentialObservation;
|
|
219
|
+
/** Read-only snapshot of the observe-stream. */
|
|
220
|
+
state(): EProcessState & {
|
|
221
|
+
decision: SequentialDecision;
|
|
222
|
+
};
|
|
173
223
|
}
|
|
174
|
-
declare function aceDriver(opts?: AceDriverOptions): ImprovementDriver;
|
|
175
|
-
|
|
176
224
|
/**
|
|
177
|
-
*
|
|
178
|
-
*
|
|
179
|
-
* `
|
|
180
|
-
*
|
|
181
|
-
*
|
|
182
|
-
* `gepaDriver` — and with our own `traceAnalystDriver` — inside `compareDrivers`
|
|
183
|
-
* on identical traces / scenarios / held-out scoring.
|
|
184
|
-
*
|
|
185
|
-
* It PRESERVES halo's actual working usage — `analyze` shells out to the
|
|
186
|
-
* published CLI (`halo <traces.jsonl> -p <prompt> -m <model>`) and uses its real
|
|
187
|
-
* RLM findings verbatim. We do NOT reimplement its analysis; that would make the
|
|
188
|
-
* benchmark meaningless. The materialize/apply pipeline is the shared
|
|
189
|
-
* `analysisEditDriver` — identical to `traceAnalystDriver`, which is what makes
|
|
190
|
-
* the comparison apples-to-apples.
|
|
225
|
+
* Anytime-valid sequential paired gate. Conforms to the existing `Gate`
|
|
226
|
+
* contract (`decide(ctx)` consumes candidate vs baseline judge scores via
|
|
227
|
+
* `pairHoldout` — same pairing granularity as the fixed-n gates: full cellId,
|
|
228
|
+
* never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
|
|
229
|
+
* that score cells incrementally and want to stop mid-stream.
|
|
191
230
|
*
|
|
192
|
-
*
|
|
231
|
+
* Decision mapping onto the substrate's five-valued `GateDecision`:
|
|
232
|
+
* - 'promote' → 'ship'
|
|
233
|
+
* - 'continue' → 'need_more_work' (stream ended before maxN with
|
|
234
|
+
* the e-value undecided — more reps could decide)
|
|
235
|
+
* - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
|
|
236
|
+
* evidence of no effect (never a silent default)
|
|
193
237
|
*/
|
|
194
|
-
|
|
195
|
-
interface
|
|
196
|
-
/**
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
/**
|
|
213
|
-
|
|
214
|
-
maxTurns?: number;
|
|
215
|
-
/** Test seam: inject a fetch for the apply-step callLlm (no network in unit tests). */
|
|
216
|
-
fetchImpl?: LlmClientOptions['fetch'];
|
|
238
|
+
declare function sequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: SequentialPairedGateOptions): SequentialPairedGate<TArtifact, TScenario>;
|
|
239
|
+
interface SequentialDecideOptions {
|
|
240
|
+
/** Type-I budget for the early-stop evidence. Default 0.05. */
|
|
241
|
+
alpha?: number;
|
|
242
|
+
/** Minimum paired deltas before a stop may fire. Default 5. */
|
|
243
|
+
minN?: number;
|
|
244
|
+
/** Bet truncation forwarded to `eProcess`. Default 0.5. */
|
|
245
|
+
maxBet?: number;
|
|
246
|
+
/** Bound on |per-scenario composite delta|. Default 1. */
|
|
247
|
+
scale?: number;
|
|
248
|
+
}
|
|
249
|
+
interface SequentialDecideFn {
|
|
250
|
+
(args: {
|
|
251
|
+
history: GenerationRecord[];
|
|
252
|
+
}): {
|
|
253
|
+
stop: boolean;
|
|
254
|
+
reason?: string;
|
|
255
|
+
};
|
|
256
|
+
/** Read-only snapshot of the accumulated e-process (observability + tests). */
|
|
257
|
+
state(): EProcessState;
|
|
217
258
|
}
|
|
218
|
-
/** Wrap the real halo-engine CLI as an ImprovementDriver (prompt-tier). */
|
|
219
|
-
declare function haloDriver(opts: HaloDriverOptions): ImprovementDriver;
|
|
220
|
-
|
|
221
259
|
/**
|
|
222
|
-
*
|
|
223
|
-
*
|
|
224
|
-
*
|
|
225
|
-
* OPTIMIZER drivers (`gepaDriver` rewrites the prompt; this one BUILDS a
|
|
226
|
-
* searchable memory of what prior trajectories taught and grafts the most
|
|
227
|
-
* relevant lessons onto the surface).
|
|
228
|
-
*
|
|
229
|
-
* Each generation it:
|
|
230
|
-
* 1. collects lessons — this generation's trace-analyst `findings` PLUS the
|
|
231
|
-
* memory already carried in the parent surface (so memory accumulates
|
|
232
|
-
* across generations instead of resetting);
|
|
233
|
-
* 2. curates them — normalizes, deduplicates near-identical lessons, and ranks
|
|
234
|
-
* by recurrence (a lesson seen across many findings outranks a one-off);
|
|
235
|
-
* 3. retrieves the top-K and writes them back as a single delimited memory
|
|
236
|
-
* block in the surface (idempotent — the block is replaced, never stacked,
|
|
237
|
-
* so the prompt does not grow without bound).
|
|
260
|
+
* `SurfaceProposer.decide` adapter — stops the optimization loop the moment
|
|
261
|
+
* the e-process decides the loop has produced a real improvement, instead of
|
|
262
|
+
* always running `maxGenerations`.
|
|
238
263
|
*
|
|
239
|
-
*
|
|
240
|
-
*
|
|
241
|
-
*
|
|
242
|
-
*
|
|
243
|
-
*
|
|
264
|
+
* Stream: for each generation g ≥ 1, the per-scenario composite deltas of
|
|
265
|
+
* generation g's top candidate vs the generation-0 top candidate (the
|
|
266
|
+
* incumbent the loop set out to beat), paired by scenarioId. H0: no proposed
|
|
267
|
+
* surface improves any scenario's expected composite over the incumbent —
|
|
268
|
+
* under it every delta has conditional mean ≤ 0 and the e-process is valid.
|
|
269
|
+
* Once wealth ≥ 1/alpha the loop stops and hands the winner to the promotion
|
|
270
|
+
* gate (which re-scores on HELD-OUT data — this adapter only spends the
|
|
271
|
+
* exploration budget, it never promotes).
|
|
244
272
|
*
|
|
245
|
-
*
|
|
246
|
-
*
|
|
247
|
-
*
|
|
273
|
+
* Honesty caveats: (1) the incumbent's scores are measured once and shared
|
|
274
|
+
* across all generations' deltas, so type-I control is exact only insofar as
|
|
275
|
+
* those scores approximate the incumbent's true per-scenario means (more reps
|
|
276
|
+
* → tighter); (2) an UNDECIDED process never stops the loop — absence of a
|
|
277
|
+
* crossing is NOT evidence of no effect, so the loop simply runs its normal
|
|
278
|
+
* course. Calling the adapter repeatedly with a growing history consumes each
|
|
279
|
+
* generation exactly once (re-feeding an already-seen record would double-count
|
|
280
|
+
* evidence).
|
|
248
281
|
*/
|
|
249
|
-
|
|
250
|
-
interface MemoryCurationDriverOptions {
|
|
251
|
-
/** Top-K lessons retained in the surface memory block. Default 12. */
|
|
252
|
-
maxEntries?: number;
|
|
253
|
-
/** Heading rendered above the lessons inside the block. Default below. */
|
|
254
|
-
sectionHeading?: string;
|
|
255
|
-
/**
|
|
256
|
-
* Optional LLM distillation: compress raw findings into crisp, generalizable
|
|
257
|
-
* one-line imperatives before curating. Omit for verbatim (deterministic).
|
|
258
|
-
*/
|
|
259
|
-
distill?: {
|
|
260
|
-
baseUrl: string;
|
|
261
|
-
apiKey?: string;
|
|
262
|
-
model: string;
|
|
263
|
-
fetchImpl?: LlmClientOptions['fetch'];
|
|
264
|
-
};
|
|
265
|
-
}
|
|
266
|
-
/** Build the CURATOR driver. */
|
|
267
|
-
declare function memoryCurationDriver(opts?: MemoryCurationDriverOptions): ImprovementDriver;
|
|
282
|
+
declare function sequentialDecide(options?: SequentialDecideOptions): SequentialDecideFn;
|
|
268
283
|
|
|
269
284
|
/**
|
|
270
|
-
*
|
|
285
|
+
* Statistical held-out promotion machinery — the trustworthy core the
|
|
286
|
+
* point-estimate `heldout-delta` gate lacked.
|
|
271
287
|
*
|
|
272
|
-
*
|
|
273
|
-
*
|
|
274
|
-
*
|
|
275
|
-
*
|
|
276
|
-
* a
|
|
277
|
-
*
|
|
278
|
-
* not
|
|
279
|
-
*
|
|
280
|
-
*
|
|
281
|
-
* applied and what could not (a missing anchor is a rejected op, never a
|
|
282
|
-
* silently dropped one). Pure, no I/O.
|
|
283
|
-
*/
|
|
284
|
-
/** A single bounded edit against a skill surface.
|
|
285
|
-
* - `add` — insert `text` after the first line containing `after`
|
|
286
|
-
* (append to the end when `after` is absent/empty).
|
|
287
|
-
* - `delete` — remove the first line containing `anchor`.
|
|
288
|
-
* - `replace` — replace the first line containing `anchor` with `text`.
|
|
289
|
-
* `text` may be multi-line; it is spliced in as multiple lines. Anchors match
|
|
290
|
-
* the FIRST line that contains the substring (deterministic; SkillOpt is
|
|
291
|
-
* expected to anchor on unique text). */
|
|
292
|
-
type SkillPatchOp = {
|
|
293
|
-
op: 'add';
|
|
294
|
-
after?: string;
|
|
295
|
-
text: string;
|
|
296
|
-
} | {
|
|
297
|
-
op: 'delete';
|
|
298
|
-
anchor: string;
|
|
299
|
-
} | {
|
|
300
|
-
op: 'replace';
|
|
301
|
-
anchor: string;
|
|
302
|
-
text: string;
|
|
303
|
-
};
|
|
304
|
-
/** A named, attributable bundle of ops the optimizer proposes as one edit. */
|
|
305
|
-
interface SkillPatch {
|
|
306
|
-
label: string;
|
|
307
|
-
rationale: string;
|
|
308
|
-
ops: SkillPatchOp[];
|
|
309
|
-
}
|
|
310
|
-
interface SkillPatchRejection {
|
|
311
|
-
op: SkillPatchOp;
|
|
312
|
-
reason: string;
|
|
313
|
-
}
|
|
314
|
-
interface ApplySkillPatchResult {
|
|
315
|
-
surface: string;
|
|
316
|
-
/** Count of ops that mutated the surface. */
|
|
317
|
-
applied: number;
|
|
318
|
-
/** Ops that could not apply (unanchored / empty), with the reason. The
|
|
319
|
-
* surface still reflects every APPLIED op — partial application is honest,
|
|
320
|
-
* and the caller decides whether a partial patch is worth scoring. */
|
|
321
|
-
rejected: SkillPatchRejection[];
|
|
322
|
-
}
|
|
323
|
-
/**
|
|
324
|
-
* Apply a SkillOpt patch to a text surface. Ops apply in array order against
|
|
325
|
-
* the evolving line buffer (an `add after X` followed by a `delete X` sees the
|
|
326
|
-
* inserted lines). A missing anchor rejects only that op; the rest still apply.
|
|
327
|
-
*/
|
|
328
|
-
declare function applySkillPatch(surface: string, patch: SkillPatch): ApplySkillPatchResult;
|
|
329
|
-
/** Total ops in a patch — the edit-budget axis (SkillOpt's "textual learning
|
|
330
|
-
* rate" caps this per epoch). */
|
|
331
|
-
declare function patchEditCount(patch: SkillPatch): number;
|
|
332
|
-
|
|
333
|
-
/**
|
|
334
|
-
* @experimental
|
|
335
|
-
*
|
|
336
|
-
* `skillOptDriver` — a patch-mode `ImprovementDriver` implementing SkillOpt
|
|
337
|
-
* (Microsoft, arXiv:2605.23904). Where `gepaDriver` regenerates the whole
|
|
338
|
-
* surface by reflection, SkillOpt proposes BOUNDED, anchored edits
|
|
339
|
-
* (add/delete/replace) to ONE skill document, so a good rule introduced
|
|
340
|
-
* earlier is not clobbered by a later sweeping rewrite. The edit budget is the
|
|
341
|
-
* paper's "textual learning rate"; a rejected-edit buffer + a slow-update
|
|
342
|
-
* meta-note steer the optimizer away from dead ends.
|
|
343
|
-
*
|
|
344
|
-
* This module is the PROPOSER — the LLM call that turns evidence into
|
|
345
|
-
* structured patches. The accept-only-if-held-out-improves loop, the budget
|
|
346
|
-
* annealing, and the rejected buffer live in the `runSkillOpt` preset, which
|
|
347
|
-
* owns the epoch hill-climb. The driver also conforms to `ImprovementDriver`
|
|
348
|
-
* (`propose` applies its patches to the current surface and returns the
|
|
349
|
-
* candidate surfaces) so it is a drop-in for `runOptimization` and a fair
|
|
350
|
-
* entrant in `compareDrivers`.
|
|
351
|
-
*/
|
|
352
|
-
|
|
353
|
-
/** Evidence the optimizer reflects on: where the current surface is weakest.
|
|
354
|
-
* Computed by the caller (the preset uses a TRAIN campaign so proposals never
|
|
355
|
-
* see the held-out split; the generic loop derives it from history). */
|
|
356
|
-
interface SkillOptEvidence {
|
|
357
|
-
/** Lowest-scoring scenarios (drives WHICH behavior to patch). */
|
|
358
|
-
weakScenarios: Array<{
|
|
359
|
-
scenarioId: string;
|
|
360
|
-
composite: number;
|
|
361
|
-
}>;
|
|
362
|
-
/** Lowest-scoring judge dimensions (drives WHAT to patch for). */
|
|
363
|
-
weakDimensions: Array<{
|
|
364
|
-
dimension: string;
|
|
365
|
-
score: number;
|
|
366
|
-
}>;
|
|
367
|
-
}
|
|
368
|
-
/** A patch that was tried and not accepted — fed back to the model so it does
|
|
369
|
-
* not re-propose a dead end (SkillOpt's rejected-edit buffer). */
|
|
370
|
-
interface RejectedEdit {
|
|
371
|
-
label: string;
|
|
372
|
-
rationale: string;
|
|
373
|
-
reason: string;
|
|
374
|
-
}
|
|
375
|
-
interface ProposePatchesArgs {
|
|
376
|
-
surface: string;
|
|
377
|
-
evidence: SkillOptEvidence;
|
|
378
|
-
/** Max ops per patch this round (the annealed textual learning rate). */
|
|
379
|
-
editBudget: number;
|
|
380
|
-
rejectedBuffer: RejectedEdit[];
|
|
381
|
-
/** Slow-update meta guidance accumulated across epochs. */
|
|
382
|
-
metaNote?: string;
|
|
383
|
-
/** Analyst findings + research report rendered as a prompt block (the
|
|
384
|
-
* EYES→HANDS wire) so a patch targets a NAMED diagnosed root cause. Built by
|
|
385
|
-
* the driver from `ctx.findings`/`ctx.report`; the patch-native `runSkillOpt`
|
|
386
|
-
* path may also supply it. */
|
|
387
|
-
findingsNote?: string;
|
|
388
|
-
/** How many candidate patches to propose. */
|
|
389
|
-
count: number;
|
|
390
|
-
signal: AbortSignal;
|
|
391
|
-
}
|
|
392
|
-
interface SkillOptDriverOptions {
|
|
393
|
-
llm: LlmClientOptions;
|
|
394
|
-
model: string;
|
|
395
|
-
/** What the skill document governs — orients the prompt. */
|
|
396
|
-
target: string;
|
|
397
|
-
/** Default ops-per-patch cap when used as a bare `ImprovementDriver`. The
|
|
398
|
-
* `runSkillOpt` preset overrides this per epoch as it anneals. Default 3. */
|
|
399
|
-
editBudget?: number;
|
|
400
|
-
temperature?: number;
|
|
401
|
-
maxTokens?: number;
|
|
402
|
-
/** Top-K weak scenarios/dimensions surfaced as evidence. Default 3. */
|
|
403
|
-
evidenceK?: number;
|
|
404
|
-
}
|
|
405
|
-
interface SkillOptDriver extends ImprovementDriver {
|
|
406
|
-
/** Patch-native path used by `runSkillOpt` (the SkillOpt epoch loop owns
|
|
407
|
-
* acceptance/budget/buffer). Returns structured patches, NOT surfaces. */
|
|
408
|
-
proposePatches(args: ProposePatchesArgs): Promise<SkillPatch[]>;
|
|
409
|
-
}
|
|
410
|
-
declare function skillOptDriver(opts: SkillOptDriverOptions): SkillOptDriver;
|
|
411
|
-
/** Parse + validate the patch response. Throws `SkillPatchParseError` when the
|
|
412
|
-
* response is not valid JSON at all (a router/model failure the caller must
|
|
413
|
-
* see — never a silent no-op epoch). Returns `[]` only for the legitimate
|
|
414
|
-
* "valid JSON, zero usable patches" case. Malformed ops within a patch are
|
|
415
|
-
* dropped (not silently mutated); each patch is truncated to the edit budget. */
|
|
416
|
-
declare class SkillPatchParseError extends Error {
|
|
417
|
-
constructor(message: string);
|
|
418
|
-
}
|
|
419
|
-
declare function parseSkillPatchResponse(raw: string, maxPatches: number, editBudget: number): SkillPatch[];
|
|
420
|
-
|
|
421
|
-
/**
|
|
422
|
-
* @experimental
|
|
423
|
-
*
|
|
424
|
-
* `traceAnalystDriver` — wraps agent-eval's OWN trace-analyst engine
|
|
425
|
-
* (`AnalystRegistry` over the agentic OTLP reader) as an `ImprovementDriver`.
|
|
426
|
-
* It is the symmetric opponent to `haloDriver`: both run the SAME shared
|
|
427
|
-
* `analysisEditDriver` pipeline (materialize identical traces → apply via one
|
|
428
|
-
* identical LLM edit), so a `compareDrivers` lift delta isolates a single
|
|
429
|
-
* variable — ANALYSIS QUALITY. The benchmark answers "is our HALO clone as good
|
|
430
|
-
* as the real HALO?" as a held-out lift CI, not a vibe.
|
|
431
|
-
*
|
|
432
|
-
* Findings come from the REGISTRY (structured `AnalystFinding[]` carrying
|
|
433
|
-
* area / severity / recommended_action), rendered into the report the shared
|
|
434
|
-
* apply step consumes.
|
|
435
|
-
*
|
|
436
|
-
* Fail-loud: no traces → throw; analyst run errors → throw; zero findings →
|
|
437
|
-
* throw. Never fabricate a candidate.
|
|
438
|
-
*/
|
|
439
|
-
|
|
440
|
-
interface TraceAnalystDriverOptions {
|
|
441
|
-
/** OpenAI-compatible base URL for BOTH the analyst's agentic reads and the
|
|
442
|
-
* apply step (e.g. `https://api.deepseek.com/v1` or the Tangle router). */
|
|
443
|
-
baseUrl: string;
|
|
444
|
-
/** Bearer key. Required — the Ax AI service has no env fallback here. */
|
|
445
|
-
apiKey: string;
|
|
446
|
-
/** Model the analyst kinds use for their agentic trace reads. */
|
|
447
|
-
model: string;
|
|
448
|
-
/** Model used to APPLY findings to the prompt surface. Default = `model`.
|
|
449
|
-
* Keep this EQUAL to haloDriver's `applyModel` for an apples-to-apples run. */
|
|
450
|
-
applyModel?: string;
|
|
451
|
-
/** Ax provider name. Default 'openai' — works for any OpenAI-compatible base
|
|
452
|
-
* via `apiURL`. Use 'deepseek' to hit DeepSeek's native provider. */
|
|
453
|
-
provider?: string;
|
|
454
|
-
/** Which analyst kinds to run. Default = the full shipped suite. */
|
|
455
|
-
kinds?: readonly TraceAnalystKindSpec[];
|
|
456
|
-
/** Resolve the OTLP traces (JSONL string) the analyst should read for THIS
|
|
457
|
-
* generation — identical contract to `haloDriver.resolveTraces`. */
|
|
458
|
-
resolveTraces: (ctx: ProposeContext) => string | Promise<string>;
|
|
459
|
-
/** Override the findings producer. Default: the shipped `AnalystRegistry`
|
|
460
|
-
* over `kinds`. The unit suite injects canned findings here. */
|
|
461
|
-
analyze?: (tracePath: string, ctx: ProposeContext) => Promise<ReadonlyArray<AnalystFinding>>;
|
|
462
|
-
/** Test seam: inject a fetch for the apply-step `callLlm`. */
|
|
463
|
-
fetchImpl?: LlmClientOptions['fetch'];
|
|
464
|
-
}
|
|
465
|
-
/** Wrap agent-eval's trace-analyst registry as an ImprovementDriver (prompt-tier). */
|
|
466
|
-
declare function traceAnalystDriver(opts: TraceAnalystDriverOptions): ImprovementDriver;
|
|
467
|
-
|
|
468
|
-
/**
|
|
469
|
-
* @experimental
|
|
470
|
-
*
|
|
471
|
-
* Anytime-valid sequential promotion gate — an e-process (betting
|
|
472
|
-
* test-martingale, see `eProcess` in `statistics.ts`) over paired
|
|
473
|
-
* per-scenario deltas, so a campaign stops the MOMENT evidence decides
|
|
474
|
-
* instead of burning a fixed-n budget. Decisions remain valid at any
|
|
475
|
-
* data-dependent stopping time (Ville's inequality), which is exactly what
|
|
476
|
-
* fixed-n machinery cannot offer: peeking at a bootstrap CI after every
|
|
477
|
-
* observation and stopping on the first significant peek inflates type-I
|
|
478
|
-
* error far beyond alpha.
|
|
479
|
-
*
|
|
480
|
-
* REPLACES, never layers on, a fixed-n gate. Running `heldoutSignificance`
|
|
481
|
-
* or `paretoSignificanceGate` repeatedly on a growing sample and stopping
|
|
482
|
-
* early is optional stopping no matter how it is dressed up; this gate is
|
|
483
|
-
* the valid way to stop early. Use one or the other per evidence stream.
|
|
484
|
-
*
|
|
485
|
-
* Pre-registration binding: anytime validity holds only for the
|
|
486
|
-
* PRE-REGISTERED statistic. When a `SignedManifest` is bound, the gate takes
|
|
487
|
-
* alpha from `manifest.alpha`, the observation budget from
|
|
488
|
-
* `manifest.preRegisteredN`, orients deltas by `manifest.direction`, and
|
|
489
|
-
* shifts the null boundary by `manifest.minEffect` — re-deciding the same
|
|
490
|
-
* stream under different parameters after seeing data would reopen optional
|
|
491
|
-
* stopping under a fancier name. The manifest's content hash is verified at
|
|
492
|
-
* construction (sync, same `sha256-content` scheme as `signManifest`).
|
|
493
|
-
*
|
|
494
|
-
* Non-iid caveat (stated honestly): the supermartingale guarantee needs each
|
|
495
|
-
* delta's conditional mean under H0 to stay ≤ the null boundary given the
|
|
496
|
-
* past — exchangeable scenario deltas suffice. Scenario streams ordered by
|
|
497
|
-
* difficulty or by scenario family violate this; `decide(ctx)` therefore
|
|
498
|
-
* shuffles the paired deltas with a SEEDED permutation by default (the
|
|
499
|
-
* permutation is data-independent, so bet predictability is preserved).
|
|
500
|
-
* Stratified betting (per-stratum λ) is future work, not implemented here.
|
|
501
|
-
*/
|
|
502
|
-
|
|
503
|
-
type SequentialDecision = 'promote' | 'continue' | 'undecided-at-maxN';
|
|
504
|
-
interface SequentialObservation {
|
|
505
|
-
decision: SequentialDecision;
|
|
506
|
-
/** Current e-value (the betting wealth) against H0. */
|
|
507
|
-
eValue: number;
|
|
508
|
-
/** Paired deltas consumed so far. */
|
|
509
|
-
n: number;
|
|
510
|
-
/** Names the decision basis. For 'undecided-at-maxN' it states explicitly
|
|
511
|
-
* that exhausting the budget is NOT evidence of no effect. */
|
|
512
|
-
reason: string;
|
|
513
|
-
}
|
|
514
|
-
interface SequentialPairedGateOptions {
|
|
515
|
-
/** Type-I budget. With `preRegistration` bound this MUST match
|
|
516
|
-
* `manifest.alpha` (conflict throws). Default 0.05. */
|
|
517
|
-
alpha?: number;
|
|
518
|
-
/** Minimum paired deltas before a promote may fire. The stopping rule is
|
|
519
|
-
* "first n ≥ minN with e-value ≥ 1/alpha" — still a valid stopping time.
|
|
520
|
-
* Default 5. */
|
|
521
|
-
minN?: number;
|
|
522
|
-
/** Pre-registered observation budget. Required unless `preRegistration`
|
|
523
|
-
* supplies it via `preRegisteredN` (conflict throws). */
|
|
524
|
-
maxN?: number;
|
|
525
|
-
/** Bet truncation forwarded to `eProcess`. Default 0.5. */
|
|
526
|
-
maxBet?: number;
|
|
527
|
-
/** Bound on |delta| in the judge's native scale; deltas are mapped to
|
|
528
|
-
* x = (d/scale + 1)/2 ∈ [0,1]. A delta outside ±scale throws (use
|
|
529
|
-
* `detectScale` to pick 1 vs 100 BEFORE streaming). Default 1. */
|
|
530
|
-
scale?: number;
|
|
531
|
-
/** Seed for the data-independent shuffle of paired deltas in `decide(ctx)`
|
|
532
|
-
* (exchangeability guard). Default 1337. */
|
|
533
|
-
shuffleSeed?: number;
|
|
534
|
-
/** Bind the pre-registered hypothesis. Verified (content hash) at
|
|
535
|
-
* construction; alpha/maxN/direction/minEffect come FROM the manifest. */
|
|
536
|
-
preRegistration?: SignedManifest;
|
|
537
|
-
/** Override the gate name in reports. */
|
|
538
|
-
name?: string;
|
|
539
|
-
}
|
|
540
|
-
interface SequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario> extends Gate<TArtifact, TScenario> {
|
|
541
|
-
/** Streaming entry point: feed one paired per-scenario delta
|
|
542
|
-
* (candidate − baseline, native scale). Each gate instance carries ONE
|
|
543
|
-
* observe-stream; `decide(ctx)` runs on its own fresh stream and never
|
|
544
|
-
* consumes or advances this one. 'promote' is sticky; observing past the
|
|
545
|
-
* pre-registered maxN throws (extending a finished stream after seeing
|
|
546
|
-
* the result reopens optional stopping — start a NEW pre-registered
|
|
547
|
-
* test). */
|
|
548
|
-
observe(delta: number): SequentialObservation;
|
|
549
|
-
/** Read-only snapshot of the observe-stream. */
|
|
550
|
-
state(): EProcessState & {
|
|
551
|
-
decision: SequentialDecision;
|
|
552
|
-
};
|
|
553
|
-
}
|
|
554
|
-
/**
|
|
555
|
-
* Anytime-valid sequential paired gate. Conforms to the existing `Gate`
|
|
556
|
-
* contract (`decide(ctx)` consumes candidate vs baseline judge scores via
|
|
557
|
-
* `pairHoldout` — same pairing granularity as the fixed-n gates: full cellId,
|
|
558
|
-
* never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
|
|
559
|
-
* that score cells incrementally and want to stop mid-stream.
|
|
560
|
-
*
|
|
561
|
-
* Decision mapping onto the substrate's five-valued `GateDecision`:
|
|
562
|
-
* - 'promote' → 'ship'
|
|
563
|
-
* - 'continue' → 'need_more_work' (stream ended before maxN with
|
|
564
|
-
* the e-value undecided — more reps could decide)
|
|
565
|
-
* - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
|
|
566
|
-
* evidence of no effect (never a silent default)
|
|
567
|
-
*/
|
|
568
|
-
declare function sequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: SequentialPairedGateOptions): SequentialPairedGate<TArtifact, TScenario>;
|
|
569
|
-
interface SequentialDecideOptions {
|
|
570
|
-
/** Type-I budget for the early-stop evidence. Default 0.05. */
|
|
571
|
-
alpha?: number;
|
|
572
|
-
/** Minimum paired deltas before a stop may fire. Default 5. */
|
|
573
|
-
minN?: number;
|
|
574
|
-
/** Bet truncation forwarded to `eProcess`. Default 0.5. */
|
|
575
|
-
maxBet?: number;
|
|
576
|
-
/** Bound on |per-scenario composite delta|. Default 1. */
|
|
577
|
-
scale?: number;
|
|
578
|
-
}
|
|
579
|
-
interface SequentialDecideFn {
|
|
580
|
-
(args: {
|
|
581
|
-
history: GenerationRecord[];
|
|
582
|
-
}): {
|
|
583
|
-
stop: boolean;
|
|
584
|
-
reason?: string;
|
|
585
|
-
};
|
|
586
|
-
/** Read-only snapshot of the accumulated e-process (observability + tests). */
|
|
587
|
-
state(): EProcessState;
|
|
588
|
-
}
|
|
589
|
-
/**
|
|
590
|
-
* `ImprovementDriver.decide` adapter — stops the optimization loop the moment
|
|
591
|
-
* the e-process decides the loop has produced a real improvement, instead of
|
|
592
|
-
* always running `maxGenerations`.
|
|
593
|
-
*
|
|
594
|
-
* Stream: for each generation g ≥ 1, the per-scenario composite deltas of
|
|
595
|
-
* generation g's top candidate vs the generation-0 top candidate (the
|
|
596
|
-
* incumbent the loop set out to beat), paired by scenarioId. H0: no proposed
|
|
597
|
-
* surface improves any scenario's expected composite over the incumbent —
|
|
598
|
-
* under it every delta has conditional mean ≤ 0 and the e-process is valid.
|
|
599
|
-
* Once wealth ≥ 1/alpha the loop stops and hands the winner to the promotion
|
|
600
|
-
* gate (which re-scores on HELD-OUT data — this adapter only spends the
|
|
601
|
-
* exploration budget, it never promotes).
|
|
602
|
-
*
|
|
603
|
-
* Honesty caveats: (1) the incumbent's scores are measured once and shared
|
|
604
|
-
* across all generations' deltas, so type-I control is exact only insofar as
|
|
605
|
-
* those scores approximate the incumbent's true per-scenario means (more reps
|
|
606
|
-
* → tighter); (2) an UNDECIDED process never stops the loop — absence of a
|
|
607
|
-
* crossing is NOT evidence of no effect, so the loop simply runs its normal
|
|
608
|
-
* course. Calling the adapter repeatedly with a growing history consumes each
|
|
609
|
-
* generation exactly once (re-feeding an already-seen record would double-count
|
|
610
|
-
* evidence).
|
|
611
|
-
*/
|
|
612
|
-
declare function sequentialDecide(options?: SequentialDecideOptions): SequentialDecideFn;
|
|
613
|
-
|
|
614
|
-
/**
|
|
615
|
-
* @experimental
|
|
616
|
-
*
|
|
617
|
-
* Statistical held-out promotion machinery — the trustworthy core the
|
|
618
|
-
* point-estimate `heldout-delta` gate lacked.
|
|
619
|
-
*
|
|
620
|
-
* The shipped false positive it prevents: a winner re-scored against the
|
|
621
|
-
* baseline on the holdout read run-to-run model NOISE (e.g. 91 vs 95) as a
|
|
622
|
-
* "+4 lift" and shipped, because the gate compared point estimates with no
|
|
623
|
-
* confidence interval. Here we pair candidate vs baseline holdout observations
|
|
624
|
-
* and bootstrap a CI on the paired delta — a candidate ships only when the CI
|
|
625
|
-
* lower bound clears the effect-size threshold (the gain is real at the
|
|
626
|
-
* confidence level, not noise), and is blocked when a critical dimension
|
|
627
|
-
* (e.g. `hallucination_free` for a legal agent) significantly regresses even if
|
|
628
|
-
* the net composite rose (anti-Goodhart).
|
|
288
|
+
* The shipped false positive it prevents: a winner re-scored against the
|
|
289
|
+
* baseline on the holdout read run-to-run model NOISE (e.g. 91 vs 95) as a
|
|
290
|
+
* "+4 lift" and shipped, because the gate compared point estimates with no
|
|
291
|
+
* confidence interval. Here we pair candidate vs baseline holdout observations
|
|
292
|
+
* and bootstrap a CI on the paired delta — a candidate ships only when the CI
|
|
293
|
+
* lower bound clears the effect-size threshold (the gain is real at the
|
|
294
|
+
* confidence level, not noise), and is blocked when a critical dimension
|
|
295
|
+
* (e.g. `hallucination_free` for a legal agent) significantly regresses even if
|
|
296
|
+
* the net composite rose (anti-Goodhart).
|
|
629
297
|
*
|
|
630
298
|
* Two traps this module is built around (both produce a NEW false positive if
|
|
631
299
|
* gotten wrong):
|
|
@@ -712,8 +380,6 @@ declare function dimensionRegressions(candidate: Map<string, Record<string, Judg
|
|
|
712
380
|
}): DimensionRegression[];
|
|
713
381
|
|
|
714
382
|
/**
|
|
715
|
-
* @experimental
|
|
716
|
-
*
|
|
717
383
|
* Filesystem `LabeledScenarioStore` adapter. The default capture sink for
|
|
718
384
|
* traces + eval artifacts. Production deployments typically swap for a
|
|
719
385
|
* Turso/SQLite adapter (same interface).
|
|
@@ -773,26 +439,145 @@ declare class FsLabeledScenarioStore implements LabeledScenarioStore {
|
|
|
773
439
|
}
|
|
774
440
|
|
|
775
441
|
/**
|
|
776
|
-
*
|
|
442
|
+
* FAPO (Fully Autonomous Prompt Optimization) is an orchestration policy, not
|
|
443
|
+
* a new prompt mutation primitive. The paper's loop evaluates an inspectable
|
|
444
|
+
* workflow, attributes failures to an edit level, proposes ONE scoped change,
|
|
445
|
+
* reviews it, measures it, and escalates from prompt -> parameters -> structure
|
|
446
|
+
* only when prompt-level search is exhausted and attribution supports the
|
|
447
|
+
* higher-cost edit.
|
|
777
448
|
*
|
|
778
|
-
*
|
|
779
|
-
*
|
|
780
|
-
*
|
|
781
|
-
*
|
|
782
|
-
*
|
|
449
|
+
* This substrate proposer encodes that policy while keeping the edit generators
|
|
450
|
+
* pluggable:
|
|
451
|
+
* - prompt: usually `gepaProposer`, `skillOptProposer`, or a runtime reflective proposer
|
|
452
|
+
* - parameter: `parameterSweepProposer` or a caller-supplied config proposer
|
|
453
|
+
* - structural: a caller-supplied code/worktree proposer from agent-runtime
|
|
454
|
+
*
|
|
455
|
+
* It deliberately does not import Claude Code, LangGraph, or Cisco's tenant
|
|
456
|
+
* runtime. agent-eval owns measurement and proposer contracts; runtime-shaped
|
|
457
|
+
* code generation stays downstream.
|
|
458
|
+
*
|
|
459
|
+
* Grounded sources:
|
|
460
|
+
* - Kassianik et al., "FAPO: Fully Autonomous Prompt Optimization of
|
|
461
|
+
* Multi-Step LLM Pipelines", arXiv:2606.19605v1.
|
|
462
|
+
* - cisco-foundation-ai/fully-automated-prompt-optimization at
|
|
463
|
+
* 376b50c57d5e2423e5a5de46e0d315a761d272d8.
|
|
464
|
+
*/
|
|
465
|
+
|
|
466
|
+
type FapoOptimizationLevel = 'prompt' | 'parameter' | 'structural';
|
|
467
|
+
interface FapoScopeContract {
|
|
468
|
+
/** Levels the tenant/playbook allows. Defaults to the levels with proposers. */
|
|
469
|
+
allowedLevels?: readonly FapoOptimizationLevel[];
|
|
470
|
+
/** Explicitly forbidden levels; wins over `allowedLevels`. */
|
|
471
|
+
forbiddenLevels?: readonly FapoOptimizationLevel[];
|
|
472
|
+
}
|
|
473
|
+
interface FapoFailureCluster {
|
|
474
|
+
label: string;
|
|
475
|
+
level: FapoOptimizationLevel;
|
|
476
|
+
count: number;
|
|
477
|
+
confidence?: 'high' | 'medium' | 'low';
|
|
478
|
+
suggestedFix?: string;
|
|
479
|
+
caseIds?: string[];
|
|
480
|
+
}
|
|
481
|
+
interface FapoAttributionSignals {
|
|
482
|
+
counts: Record<FapoOptimizationLevel, number>;
|
|
483
|
+
clusters: FapoFailureCluster[];
|
|
484
|
+
}
|
|
485
|
+
interface FapoReviewIssue {
|
|
486
|
+
checkName: string;
|
|
487
|
+
severity: 'block' | 'warn';
|
|
488
|
+
description: string;
|
|
489
|
+
location?: string;
|
|
490
|
+
}
|
|
491
|
+
interface FapoReviewResult {
|
|
492
|
+
verdict: 'pass' | 'warn' | 'fail';
|
|
493
|
+
issues?: FapoReviewIssue[];
|
|
494
|
+
suggestions?: string[];
|
|
495
|
+
}
|
|
496
|
+
interface FapoReviewInput<TFindings = unknown> {
|
|
497
|
+
level: FapoOptimizationLevel;
|
|
498
|
+
candidate: ProposedCandidate;
|
|
499
|
+
context: ProposeContext<TFindings>;
|
|
500
|
+
reason: string;
|
|
501
|
+
}
|
|
502
|
+
interface FapoProposerOptions<TFindings = unknown> {
|
|
503
|
+
/** Level-specific candidate proposers. At least one is required. */
|
|
504
|
+
proposers?: Partial<Record<FapoOptimizationLevel, SurfaceProposer<TFindings>>>;
|
|
505
|
+
/** Convenience aliases for `proposers.<level>`. */
|
|
506
|
+
promptProposer?: SurfaceProposer<TFindings>;
|
|
507
|
+
parameterProposer?: SurfaceProposer<TFindings>;
|
|
508
|
+
structuralProposer?: SurfaceProposer<TFindings>;
|
|
509
|
+
/** Tenant/playbook-derived allowed and forbidden edit levels. */
|
|
510
|
+
scope?: FapoScopeContract;
|
|
511
|
+
/**
|
|
512
|
+
* Independent reviewer hook. Return `fail` to block a candidate before eval,
|
|
513
|
+
* mirroring FAPO's variant-reviewer phase.
|
|
514
|
+
*/
|
|
515
|
+
reviewCandidate?: (input: FapoReviewInput<TFindings>) => Promise<FapoReviewResult>;
|
|
516
|
+
/**
|
|
517
|
+
* FAPO proposes one scoped change per cycle. Keep this at 1 unless you are
|
|
518
|
+
* intentionally relaxing the paper loop for a batch experiment.
|
|
519
|
+
*/
|
|
520
|
+
proposalsPerCycle?: number;
|
|
521
|
+
/** Consecutive non-improving variants required before a level is exhausted. Default 3. */
|
|
522
|
+
plateauWindow?: number;
|
|
523
|
+
/** Distinct strategies required before declaring a plateau. Default 3. */
|
|
524
|
+
minDistinctStrategies?: number;
|
|
525
|
+
/** Candidate score delta needed to count as an improvement. Default 0. */
|
|
526
|
+
minImprovement?: number;
|
|
527
|
+
/** Paper-faithful default: try prompt edits before escalating. */
|
|
528
|
+
promptFirst?: boolean;
|
|
529
|
+
/** Paper-faithful default: try parameter/config edits before structural code edits. */
|
|
530
|
+
parameterBeforeStructural?: boolean;
|
|
531
|
+
}
|
|
532
|
+
/** Build a FAPO policy proposer from level-specific candidate generators. */
|
|
533
|
+
declare function fapoProposer<TFindings = unknown>(opts: FapoProposerOptions<TFindings>): SurfaceProposer<TFindings>;
|
|
534
|
+
declare function extractFapoAttributionSignals(findings: readonly unknown[]): FapoAttributionSignals;
|
|
535
|
+
type JsonPrimitive = string | number | boolean | null;
|
|
536
|
+
type JsonValue = JsonPrimitive | JsonValue[] | {
|
|
537
|
+
[key: string]: JsonValue;
|
|
538
|
+
};
|
|
539
|
+
interface ParameterChange {
|
|
540
|
+
path: string | readonly string[];
|
|
541
|
+
value: JsonValue;
|
|
542
|
+
}
|
|
543
|
+
interface ParameterCandidate {
|
|
544
|
+
label: string;
|
|
545
|
+
rationale: string;
|
|
546
|
+
/** Deep-merged into the current JSON surface. */
|
|
547
|
+
patch?: Record<string, JsonValue>;
|
|
548
|
+
/** Dot-path writes, e.g. `{ path: 'retrieval.k', value: 10 }`. */
|
|
549
|
+
changes?: readonly ParameterChange[];
|
|
550
|
+
}
|
|
551
|
+
interface ParameterSweepProposerOptions {
|
|
552
|
+
candidates: readonly ParameterCandidate[];
|
|
553
|
+
/** Optional parser for non-JSON config surface strings. */
|
|
554
|
+
parse?: (surface: string) => Record<string, JsonValue>;
|
|
555
|
+
/** Optional serializer for non-JSON config surface strings. */
|
|
556
|
+
stringify?: (config: Record<string, JsonValue>) => string;
|
|
557
|
+
}
|
|
558
|
+
/** Config/parameter-level proposer for FAPO's middle escalation level. */
|
|
559
|
+
declare function parameterSweepProposer(opts: ParameterSweepProposerOptions): SurfaceProposer;
|
|
560
|
+
|
|
561
|
+
/**
|
|
562
|
+
* `compareProposers` — a head-to-head lift benchmark across surface proposers
|
|
563
|
+
* on ONE corpus. This is the forcing function: optimizer quality (GEPA
|
|
564
|
+
* reflection vs GEPA+Pareto vs SkillOpt) becomes a NUMBER with a confidence
|
|
565
|
+
* interval, so a proposer regression — or shipping a simplified proposer and
|
|
566
|
+
* calling it the real one — turns a build red instead of going
|
|
567
|
+
* measurement-invisible.
|
|
783
568
|
*
|
|
784
|
-
* Every entrant is scored the SAME way: each
|
|
785
|
-
* promoted, then
|
|
569
|
+
* Every entrant is scored the SAME way: each proposer returns the surface it
|
|
570
|
+
* promoted, then the benchmark scores the baseline + every winner on the
|
|
786
571
|
* SAME held-out scenarios with the SAME judges. Apples-to-apples by
|
|
787
|
-
* construction — the comparison never depends on how a
|
|
572
|
+
* construction — the comparison never depends on how a proposer measured itself.
|
|
788
573
|
* The per-scenario held-out composites feed a paired bootstrap (`statistics.ts`)
|
|
789
|
-
* for each
|
|
574
|
+
* for each proposer's lift CI and for the pairwise "which proposer wins" CI.
|
|
790
575
|
*/
|
|
791
576
|
|
|
792
577
|
/** What an optimizer produced: the surface it promoted + what it cost to get
|
|
793
|
-
* there.
|
|
578
|
+
* there. The comparison does the held-out scoring itself, so an entry only
|
|
794
579
|
* needs to run its loop and hand back the winner. */
|
|
795
|
-
interface
|
|
580
|
+
interface ProposerEntry {
|
|
796
581
|
name: string;
|
|
797
582
|
optimize: () => Promise<{
|
|
798
583
|
winnerSurface: MutableSurface;
|
|
@@ -800,11 +585,11 @@ interface DriverEntry {
|
|
|
800
585
|
durationMs?: number;
|
|
801
586
|
}>;
|
|
802
587
|
}
|
|
803
|
-
interface
|
|
588
|
+
interface ProposerScore {
|
|
804
589
|
name: string;
|
|
805
|
-
/** Mean held-out composite of the baseline (identical across
|
|
590
|
+
/** Mean held-out composite of the baseline (identical across proposers). */
|
|
806
591
|
baselineComposite: number;
|
|
807
|
-
/** Mean held-out composite of this
|
|
592
|
+
/** Mean held-out composite of this proposer's promoted surface. */
|
|
808
593
|
winnerComposite: number;
|
|
809
594
|
/** Mean per-scenario held-out lift (winner − baseline). */
|
|
810
595
|
lift: number;
|
|
@@ -819,8 +604,8 @@ interface DriverScore {
|
|
|
819
604
|
/** 1-based, by descending lift. */
|
|
820
605
|
rank: number;
|
|
821
606
|
}
|
|
822
|
-
interface
|
|
823
|
-
/** Higher-ranked
|
|
607
|
+
interface ProposerPairwise {
|
|
608
|
+
/** Higher-ranked proposer. */
|
|
824
609
|
a: string;
|
|
825
610
|
b: string;
|
|
826
611
|
/** Mean per-scenario held-out delta (a − b). */
|
|
@@ -830,31 +615,31 @@ interface DriverPairwise {
|
|
|
830
615
|
/** `a` if the CI clears 0, `b` if it is entirely negative, else `'tie'`. */
|
|
831
616
|
favored: string;
|
|
832
617
|
}
|
|
833
|
-
interface
|
|
618
|
+
interface ProposerComparison {
|
|
834
619
|
/** Sorted by descending lift; `rank` set accordingly. */
|
|
835
|
-
scores:
|
|
836
|
-
best:
|
|
837
|
-
/** Best vs each other
|
|
838
|
-
pairwise:
|
|
620
|
+
scores: ProposerScore[];
|
|
621
|
+
best: ProposerScore;
|
|
622
|
+
/** Best vs each other proposer, paired-bootstrap on the held-out winners. */
|
|
623
|
+
pairwise: ProposerPairwise[];
|
|
839
624
|
holdoutScenarioIds: string[];
|
|
840
625
|
}
|
|
841
|
-
interface
|
|
842
|
-
|
|
626
|
+
interface CompareProposersOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch' | 'scenarios'> {
|
|
627
|
+
proposers: ProposerEntry[];
|
|
843
628
|
baselineSurface: MutableSurface;
|
|
844
629
|
/** The held-out scenarios every winner is scored on. */
|
|
845
630
|
holdoutScenarios: TScenario[];
|
|
846
|
-
/** Scores a surface on a scenario — the same dispatcher the
|
|
631
|
+
/** Scores a surface on a scenario — the same dispatcher the proposers used. */
|
|
847
632
|
dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
848
633
|
/** Bootstrap resamples for the lift CIs. Default 2000. */
|
|
849
634
|
resamples?: number;
|
|
850
635
|
/** CI confidence. Default 0.95. */
|
|
851
636
|
confidence?: number;
|
|
852
637
|
}
|
|
853
|
-
declare function
|
|
638
|
+
declare function compareProposers<TScenario extends Scenario, TArtifact>(opts: CompareProposersOptions<TScenario, TArtifact>): Promise<ProposerComparison>;
|
|
854
639
|
/** Shared corpus + transport for the three built-in optimizer entries. */
|
|
855
640
|
interface OptimizerEntryConfig<TScenario extends Scenario, TArtifact> {
|
|
856
641
|
baselineSurface: string;
|
|
857
|
-
/** Training scenarios the
|
|
642
|
+
/** Training scenarios the proposers reflect on. */
|
|
858
643
|
trainScenarios: TScenario[];
|
|
859
644
|
/** Held-out scenarios (the gate axis + the benchmark scoring axis). */
|
|
860
645
|
holdoutScenarios: TScenario[];
|
|
@@ -872,33 +657,50 @@ interface OptimizerEntryConfig<TScenario extends Scenario, TArtifact> {
|
|
|
872
657
|
/** SkillOpt epochs. Default 6. */
|
|
873
658
|
maxEpochs?: number;
|
|
874
659
|
mutationPrimitives?: string[];
|
|
875
|
-
/** Static findings seed forwarded to each GEPA
|
|
876
|
-
* `ctx.findings
|
|
877
|
-
* `
|
|
660
|
+
/** Static findings seed forwarded to each GEPA proposer's `propose()` as
|
|
661
|
+
* `ctx.findings`. Forwarded by `gepaReflectionEntry` / `gepaParetoEntry`;
|
|
662
|
+
* `skillOptEntry` runs without findings (see its doc). */
|
|
878
663
|
findings?: unknown[];
|
|
879
|
-
/** Per-generation findings producer
|
|
880
|
-
*
|
|
664
|
+
/** Per-generation findings producer: after each generation scores, this
|
|
665
|
+
* re-diagnoses and REPLACES `ctx.findings` for the
|
|
881
666
|
* next generation's `propose()`. Reuses the `runOptimization` field type so
|
|
882
667
|
* it cannot drift. GEPA entries only. */
|
|
883
668
|
analyzeGeneration?: RunImprovementLoopOptions<TScenario, TArtifact>['analyzeGeneration'];
|
|
884
|
-
/**
|
|
669
|
+
/** Optional analysis report forwarded to `propose()` as `ctx.report`. */
|
|
885
670
|
report?: unknown;
|
|
886
671
|
}
|
|
887
672
|
/** GEPA, reflection-only (single-parent, no Pareto combine). */
|
|
888
|
-
declare function gepaReflectionEntry<TScenario extends Scenario, TArtifact>(config: OptimizerEntryConfig<TScenario, TArtifact>, name?: string):
|
|
673
|
+
declare function gepaReflectionEntry<TScenario extends Scenario, TArtifact>(config: OptimizerEntryConfig<TScenario, TArtifact>, name?: string): ProposerEntry;
|
|
889
674
|
/** GEPA with the Pareto frontier + combine-complementary-lessons. */
|
|
890
|
-
declare function gepaParetoEntry<TScenario extends Scenario, TArtifact>(config: OptimizerEntryConfig<TScenario, TArtifact>, name?: string):
|
|
675
|
+
declare function gepaParetoEntry<TScenario extends Scenario, TArtifact>(config: OptimizerEntryConfig<TScenario, TArtifact>, name?: string): ProposerEntry;
|
|
891
676
|
/** SkillOpt patch-mode hill-climb. Runs findings-BLIND: `runSkillOpt` owns its
|
|
892
677
|
* own epoch acceptance/budget loop and does not thread `analyzeGeneration`, so
|
|
893
678
|
* `config.findings` is intentionally NOT forwarded here. In a findings-fed
|
|
894
679
|
* comparison this entry is the blind control — do not read its result as
|
|
895
680
|
* findings-fed. (Threading findings into the SkillOpt epoch loop is a separate
|
|
896
681
|
* refactor, deferred not faked.) */
|
|
897
|
-
declare function skillOptEntry<TScenario extends Scenario, TArtifact>(config: OptimizerEntryConfig<TScenario, TArtifact>, name?: string):
|
|
682
|
+
declare function skillOptEntry<TScenario extends Scenario, TArtifact>(config: OptimizerEntryConfig<TScenario, TArtifact>, name?: string): ProposerEntry;
|
|
683
|
+
/** FAPO reviewed-escalation policy. This is an orchestration layer over
|
|
684
|
+
* level-specific proposers, not a new mutation operator:
|
|
685
|
+
* prompt -> parameter -> structural, with scope + reviewer + plateau rules in
|
|
686
|
+
* `fapoProposer`. The prompt proposer defaults to GEPA+Pareto because that is
|
|
687
|
+
* the package's strongest prompt-tier proposer; parameter/structural proposers
|
|
688
|
+
* are opt-in so we do not fake code-generation inside agent-eval. */
|
|
689
|
+
interface FapoEntryConfig<TScenario extends Scenario, TArtifact> extends OptimizerEntryConfig<TScenario, TArtifact> {
|
|
690
|
+
/** Override the prompt-level proposer. Default: `gepaProposer({ combineParents: true })`. */
|
|
691
|
+
promptProposer?: SurfaceProposer;
|
|
692
|
+
/** Parameter/config-level proposer. If omitted, `parameterCandidates` builds one. */
|
|
693
|
+
parameterProposer?: SurfaceProposer;
|
|
694
|
+
/** Structural/code-level proposer, typically supplied by agent-runtime. */
|
|
695
|
+
structuralProposer?: SurfaceProposer;
|
|
696
|
+
/** Convenience: build a `parameterSweepProposer` from these candidates. */
|
|
697
|
+
parameterCandidates?: readonly ParameterCandidate[];
|
|
698
|
+
/** FAPO policy knobs: scope, reviewer, plateau thresholds. */
|
|
699
|
+
fapo?: Omit<FapoProposerOptions, 'proposers' | 'promptProposer' | 'parameterProposer' | 'structuralProposer'>;
|
|
700
|
+
}
|
|
701
|
+
declare function fapoEscalationEntry<TScenario extends Scenario, TArtifact>(config: FapoEntryConfig<TScenario, TArtifact>, name?: string): ProposerEntry;
|
|
898
702
|
|
|
899
703
|
/**
|
|
900
|
-
* @experimental
|
|
901
|
-
*
|
|
902
704
|
* `runProfileMatrix` — the missing keystone between `runAgentMatrix` and the
|
|
903
705
|
* backend-integrity guard.
|
|
904
706
|
*
|
|
@@ -1082,87 +884,233 @@ interface UserStory extends Scenario {
|
|
|
1082
884
|
interface PlaybackContext extends DispatchContext {
|
|
1083
885
|
profile: AgentProfile;
|
|
1084
886
|
}
|
|
1085
|
-
/**
|
|
1086
|
-
* Drives the real product through a story and returns the runtime event stream
|
|
1087
|
-
* `extractProducedState` consumes. Implemented by CONSUMERS —
|
|
1088
|
-
* `SandboxPlaybackDriver` (real API / sandbox workspace) and
|
|
1089
|
-
* `PlaywrightPlaybackDriver` (real UI) — because they depend on runtime /
|
|
1090
|
-
* browser infra the substrate must not import. The driver MUST report LLM
|
|
1091
|
-
* usage via `ctx.cost.observeTokens` so the backend-integrity guard sees real
|
|
1092
|
-
* tokens (a run that never reports tokens reads as a stub).
|
|
1093
|
-
*/
|
|
1094
|
-
interface PlaybackDriver<TStory extends UserStory = UserStory> {
|
|
1095
|
-
run(story: TStory, ctx: PlaybackContext): Promise<readonly RuntimeEventLike[]>;
|
|
887
|
+
/**
|
|
888
|
+
* Drives the real product through a story and returns the runtime event stream
|
|
889
|
+
* `extractProducedState` consumes. Implemented by CONSUMERS —
|
|
890
|
+
* `SandboxPlaybackDriver` (real API / sandbox workspace) and
|
|
891
|
+
* `PlaywrightPlaybackDriver` (real UI) — because they depend on runtime /
|
|
892
|
+
* browser infra the substrate must not import. The driver MUST report LLM
|
|
893
|
+
* usage via `ctx.cost.observeTokens` so the backend-integrity guard sees real
|
|
894
|
+
* tokens (a run that never reports tokens reads as a stub).
|
|
895
|
+
*/
|
|
896
|
+
interface PlaybackDriver<TStory extends UserStory = UserStory> {
|
|
897
|
+
run(story: TStory, ctx: PlaybackContext): Promise<readonly RuntimeEventLike[]>;
|
|
898
|
+
}
|
|
899
|
+
/**
|
|
900
|
+
* Adapt a `PlaybackDriver` into a `runProfileMatrix` dispatch. The artifact the
|
|
901
|
+
* matrix scores is the `ProducedState` extracted from the driver's event
|
|
902
|
+
* stream — grade it with `scoreUserStory` (or a judge wrapping it).
|
|
903
|
+
*/
|
|
904
|
+
declare function makePlaybackDispatch<TStory extends UserStory>(driver: PlaybackDriver<TStory>): ProfileDispatchFn<TStory, ProducedState>;
|
|
905
|
+
/** A scored user story — the completion verdict plus its human title. */
|
|
906
|
+
interface UserStoryVerdict extends CompletionVerdict {
|
|
907
|
+
title: string;
|
|
908
|
+
}
|
|
909
|
+
/**
|
|
910
|
+
* Score one story's produced state against its requirements. Thin wrapper over
|
|
911
|
+
* `verifyCompletion` that builds the gold from the story and returns a
|
|
912
|
+
* per-requirement PASS/FAIL verdict. `checkCorrectness` is injected — a
|
|
913
|
+
* deterministic stub in tests, `createLlmCorrectnessChecker` in production.
|
|
914
|
+
*/
|
|
915
|
+
declare function scoreUserStory(story: UserStory, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<UserStoryVerdict>;
|
|
916
|
+
/** One row of the launch scoreboard — story × requirement → PASS/FAIL. */
|
|
917
|
+
interface ScoreboardRow {
|
|
918
|
+
storyId: string;
|
|
919
|
+
storyTitle: string;
|
|
920
|
+
reqId: string;
|
|
921
|
+
reqTitle: string;
|
|
922
|
+
status: 'PASS' | 'FAIL';
|
|
923
|
+
evidence: string[];
|
|
924
|
+
}
|
|
925
|
+
/**
|
|
926
|
+
* Flatten story verdicts into the per-requirement scoreboard — the literal
|
|
927
|
+
* Jira tick-off: one row per (story, requirement) with PASS/FAIL and the
|
|
928
|
+
* evidence behind the verdict.
|
|
929
|
+
*/
|
|
930
|
+
declare function userStoryScoreboard(verdicts: readonly UserStoryVerdict[]): ScoreboardRow[];
|
|
931
|
+
/** Launch-readiness headline counts rolled up from the per-requirement rows. */
|
|
932
|
+
interface ScoreboardSummary {
|
|
933
|
+
/** Distinct user stories on the board. */
|
|
934
|
+
stories: number;
|
|
935
|
+
/** Stories whose every requirement passed. */
|
|
936
|
+
storiesFullyComplete: number;
|
|
937
|
+
/** Total (story, requirement) rows. */
|
|
938
|
+
requirements: number;
|
|
939
|
+
/** Rows with status PASS. */
|
|
940
|
+
passed: number;
|
|
941
|
+
/** Rows with status FAIL. */
|
|
942
|
+
failed: number;
|
|
943
|
+
/** passed / requirements; 0 when there are no rows. */
|
|
944
|
+
passRate: number;
|
|
945
|
+
}
|
|
946
|
+
/** Roll the per-requirement rows up into the launch headline counts. */
|
|
947
|
+
declare function scoreboardSummary(rows: readonly ScoreboardRow[]): ScoreboardSummary;
|
|
948
|
+
interface ScoreboardRenderOptions {
|
|
949
|
+
/** Document H1. Defaults to a generic playback title. */
|
|
950
|
+
title?: string;
|
|
951
|
+
/** Key/value run metadata rendered under the headline (runId, backend, model, date). */
|
|
952
|
+
meta?: Record<string, string>;
|
|
953
|
+
/** Max chars of joined evidence shown per row. Default 160. */
|
|
954
|
+
maxEvidenceChars?: number;
|
|
955
|
+
}
|
|
956
|
+
/**
|
|
957
|
+
* Render the scoreboard as a launch-readiness Markdown document — the literal
|
|
958
|
+
* "tick off every user story" artifact: a headline roll-up, the open tickets
|
|
959
|
+
* (FAIL rows) up top as the launch blockers, then a per-story table of
|
|
960
|
+
* requirement → PASS/FAIL with the evidence behind each verdict. Pure: same
|
|
961
|
+
* rows in, same bytes out (no clock/random), so it is safe to snapshot.
|
|
962
|
+
*/
|
|
963
|
+
declare function renderScoreboardMarkdown(rows: readonly ScoreboardRow[], opts?: ScoreboardRenderOptions): string;
|
|
964
|
+
|
|
965
|
+
/**
|
|
966
|
+
* SkillOpt patch primitives (Microsoft, arXiv:2605.23904 — "Executive
|
|
967
|
+
* Strategy for Self-Evolving Agent Skills"). Where GEPA regenerates a surface
|
|
968
|
+
* by reflection, SkillOpt emits BOUNDED, anchored edits to ONE skill document
|
|
969
|
+
* — add / delete / replace — and accepts an edit only if it strictly improves
|
|
970
|
+
* a held-out score. Bounded edits are the "textual learning rate": small,
|
|
971
|
+
* reversible, and cheap to accept/reject, so a good rule introduced earlier is
|
|
972
|
+
* not overwritten by a later sweeping rewrite.
|
|
973
|
+
*
|
|
974
|
+
* This module applies a patch deterministically and reports, per op, what
|
|
975
|
+
* applied and what could not (a missing anchor is a rejected op, never a
|
|
976
|
+
* silently dropped one). Pure, no I/O.
|
|
977
|
+
*/
|
|
978
|
+
/** A single bounded edit against a skill surface.
|
|
979
|
+
* - `add` — insert `text` after the first line containing `after`
|
|
980
|
+
* (append to the end when `after` is absent/empty).
|
|
981
|
+
* - `delete` — remove the first line containing `anchor`.
|
|
982
|
+
* - `replace` — replace the first line containing `anchor` with `text`.
|
|
983
|
+
* `text` may be multi-line; it is spliced in as multiple lines. Anchors match
|
|
984
|
+
* the FIRST line that contains the substring (deterministic; SkillOpt is
|
|
985
|
+
* expected to anchor on unique text). */
|
|
986
|
+
type SkillPatchOp = {
|
|
987
|
+
op: 'add';
|
|
988
|
+
after?: string;
|
|
989
|
+
text: string;
|
|
990
|
+
} | {
|
|
991
|
+
op: 'delete';
|
|
992
|
+
anchor: string;
|
|
993
|
+
} | {
|
|
994
|
+
op: 'replace';
|
|
995
|
+
anchor: string;
|
|
996
|
+
text: string;
|
|
997
|
+
};
|
|
998
|
+
/** A named, attributable bundle of ops the optimizer proposes as one edit. */
|
|
999
|
+
interface SkillPatch {
|
|
1000
|
+
label: string;
|
|
1001
|
+
rationale: string;
|
|
1002
|
+
ops: SkillPatchOp[];
|
|
1003
|
+
}
|
|
1004
|
+
interface SkillPatchRejection {
|
|
1005
|
+
op: SkillPatchOp;
|
|
1006
|
+
reason: string;
|
|
1007
|
+
}
|
|
1008
|
+
interface ApplySkillPatchResult {
|
|
1009
|
+
surface: string;
|
|
1010
|
+
/** Count of ops that mutated the surface. */
|
|
1011
|
+
applied: number;
|
|
1012
|
+
/** Ops that could not apply (unanchored / empty), with the reason. The
|
|
1013
|
+
* surface still reflects every APPLIED op — partial application is honest,
|
|
1014
|
+
* and the caller decides whether a partial patch is worth scoring. */
|
|
1015
|
+
rejected: SkillPatchRejection[];
|
|
1016
|
+
}
|
|
1017
|
+
/**
|
|
1018
|
+
* Apply a SkillOpt patch to a text surface. Ops apply in array order against
|
|
1019
|
+
* the evolving line buffer (an `add after X` followed by a `delete X` sees the
|
|
1020
|
+
* inserted lines). A missing anchor rejects only that op; the rest still apply.
|
|
1021
|
+
*/
|
|
1022
|
+
declare function applySkillPatch(surface: string, patch: SkillPatch): ApplySkillPatchResult;
|
|
1023
|
+
/** Total ops in a patch — the edit-budget axis (SkillOpt's "textual learning
|
|
1024
|
+
* rate" caps this per epoch). */
|
|
1025
|
+
declare function patchEditCount(patch: SkillPatch): number;
|
|
1026
|
+
|
|
1027
|
+
/**
|
|
1028
|
+
* `skillOptProposer` — a patch-mode `SurfaceProposer` implementing SkillOpt
|
|
1029
|
+
* (Microsoft, arXiv:2605.23904). Where `gepaProposer` regenerates the whole
|
|
1030
|
+
* surface by reflection, SkillOpt proposes BOUNDED, anchored edits
|
|
1031
|
+
* (add/delete/replace) to ONE skill document, so a good rule introduced
|
|
1032
|
+
* earlier is not clobbered by a later sweeping rewrite. The edit budget is the
|
|
1033
|
+
* paper's "textual learning rate"; a rejected-edit buffer + a slow-update
|
|
1034
|
+
* meta-note steer the optimizer away from dead ends.
|
|
1035
|
+
*
|
|
1036
|
+
* This module is the PROPOSER — the LLM call that turns evidence into
|
|
1037
|
+
* structured patches. The accept-only-if-held-out-improves loop, the budget
|
|
1038
|
+
* annealing, and the rejected buffer live in the `runSkillOpt` preset, which
|
|
1039
|
+
* owns the epoch hill-climb. The proposer also conforms to `SurfaceProposer`
|
|
1040
|
+
* (`propose` applies its patches to the current surface and returns the
|
|
1041
|
+
* candidate surfaces) so it is a drop-in for `runOptimization` and a fair
|
|
1042
|
+
* entrant in `compareProposers`.
|
|
1043
|
+
*/
|
|
1044
|
+
|
|
1045
|
+
/** Evidence the optimizer reflects on: where the current surface is weakest.
|
|
1046
|
+
* Computed by the caller (the preset uses a TRAIN campaign so proposals never
|
|
1047
|
+
* see the held-out split; the generic loop derives it from history). */
|
|
1048
|
+
interface SkillOptEvidence {
|
|
1049
|
+
/** Lowest-scoring scenarios (drives WHICH behavior to patch). */
|
|
1050
|
+
weakScenarios: Array<{
|
|
1051
|
+
scenarioId: string;
|
|
1052
|
+
composite: number;
|
|
1053
|
+
}>;
|
|
1054
|
+
/** Lowest-scoring judge dimensions (drives WHAT to patch for). */
|
|
1055
|
+
weakDimensions: Array<{
|
|
1056
|
+
dimension: string;
|
|
1057
|
+
score: number;
|
|
1058
|
+
}>;
|
|
1059
|
+
}
|
|
1060
|
+
/** A patch that was tried and not accepted — fed back to the model so it does
|
|
1061
|
+
* not re-propose a dead end (SkillOpt's rejected-edit buffer). */
|
|
1062
|
+
interface RejectedEdit {
|
|
1063
|
+
label: string;
|
|
1064
|
+
rationale: string;
|
|
1065
|
+
reason: string;
|
|
1096
1066
|
}
|
|
1097
|
-
|
|
1098
|
-
|
|
1099
|
-
|
|
1100
|
-
|
|
1101
|
-
|
|
1102
|
-
|
|
1103
|
-
/**
|
|
1104
|
-
|
|
1105
|
-
|
|
1067
|
+
interface ProposePatchesArgs {
|
|
1068
|
+
surface: string;
|
|
1069
|
+
evidence: SkillOptEvidence;
|
|
1070
|
+
/** Max ops per patch this round (the annealed textual learning rate). */
|
|
1071
|
+
editBudget: number;
|
|
1072
|
+
rejectedBuffer: RejectedEdit[];
|
|
1073
|
+
/** Slow-update meta guidance accumulated across epochs. */
|
|
1074
|
+
metaNote?: string;
|
|
1075
|
+
/** Analyst findings + research report rendered as a prompt block so a patch
|
|
1076
|
+
* targets a NAMED diagnosed root cause. Built by
|
|
1077
|
+
* the proposer from `ctx.findings`/`ctx.report`; the patch-native `runSkillOpt`
|
|
1078
|
+
* path may also supply it. */
|
|
1079
|
+
findingsNote?: string;
|
|
1080
|
+
/** How many candidate patches to propose. */
|
|
1081
|
+
count: number;
|
|
1082
|
+
signal: AbortSignal;
|
|
1106
1083
|
}
|
|
1107
|
-
|
|
1108
|
-
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
|
|
1112
|
-
|
|
1113
|
-
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
|
|
1119
|
-
reqTitle: string;
|
|
1120
|
-
status: 'PASS' | 'FAIL';
|
|
1121
|
-
evidence: string[];
|
|
1084
|
+
interface SkillOptProposerOptions {
|
|
1085
|
+
llm: LlmClientOptions;
|
|
1086
|
+
model: string;
|
|
1087
|
+
/** What the skill document governs — orients the prompt. */
|
|
1088
|
+
target: string;
|
|
1089
|
+
/** Default ops-per-patch cap when used as a bare `SurfaceProposer`. The
|
|
1090
|
+
* `runSkillOpt` preset overrides this per epoch as it anneals. Default 3. */
|
|
1091
|
+
editBudget?: number;
|
|
1092
|
+
temperature?: number;
|
|
1093
|
+
maxTokens?: number;
|
|
1094
|
+
/** Top-K weak scenarios/dimensions surfaced as evidence. Default 3. */
|
|
1095
|
+
evidenceK?: number;
|
|
1122
1096
|
}
|
|
1123
|
-
|
|
1124
|
-
|
|
1125
|
-
|
|
1126
|
-
|
|
1127
|
-
*/
|
|
1128
|
-
declare function userStoryScoreboard(verdicts: readonly UserStoryVerdict[]): ScoreboardRow[];
|
|
1129
|
-
/** Launch-readiness headline counts rolled up from the per-requirement rows. */
|
|
1130
|
-
interface ScoreboardSummary {
|
|
1131
|
-
/** Distinct user stories on the board. */
|
|
1132
|
-
stories: number;
|
|
1133
|
-
/** Stories whose every requirement passed. */
|
|
1134
|
-
storiesFullyComplete: number;
|
|
1135
|
-
/** Total (story, requirement) rows. */
|
|
1136
|
-
requirements: number;
|
|
1137
|
-
/** Rows with status PASS. */
|
|
1138
|
-
passed: number;
|
|
1139
|
-
/** Rows with status FAIL. */
|
|
1140
|
-
failed: number;
|
|
1141
|
-
/** passed / requirements; 0 when there are no rows. */
|
|
1142
|
-
passRate: number;
|
|
1097
|
+
interface SkillOptProposer extends SurfaceProposer {
|
|
1098
|
+
/** Patch-native path used by `runSkillOpt` (the SkillOpt epoch loop owns
|
|
1099
|
+
* acceptance/budget/buffer). Returns structured patches, NOT surfaces. */
|
|
1100
|
+
proposePatches(args: ProposePatchesArgs): Promise<SkillPatch[]>;
|
|
1143
1101
|
}
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
|
|
1152
|
-
maxEvidenceChars?: number;
|
|
1102
|
+
declare function skillOptProposer(opts: SkillOptProposerOptions): SkillOptProposer;
|
|
1103
|
+
/** Parse + validate the patch response. Throws `SkillPatchParseError` when the
|
|
1104
|
+
* response is not valid JSON at all (a router/model failure the caller must
|
|
1105
|
+
* see — never a silent no-op epoch). Returns `[]` only for the legitimate
|
|
1106
|
+
* "valid JSON, zero usable patches" case. Malformed ops within a patch are
|
|
1107
|
+
* dropped (not silently mutated); each patch is truncated to the edit budget. */
|
|
1108
|
+
declare class SkillPatchParseError extends Error {
|
|
1109
|
+
constructor(message: string);
|
|
1153
1110
|
}
|
|
1154
|
-
|
|
1155
|
-
* Render the scoreboard as a launch-readiness Markdown document — the literal
|
|
1156
|
-
* "tick off every user story" artifact: a headline roll-up, the open tickets
|
|
1157
|
-
* (FAIL rows) up top as the launch blockers, then a per-story table of
|
|
1158
|
-
* requirement → PASS/FAIL with the evidence behind each verdict. Pure: same
|
|
1159
|
-
* rows in, same bytes out (no clock/random), so it is safe to snapshot.
|
|
1160
|
-
*/
|
|
1161
|
-
declare function renderScoreboardMarkdown(rows: readonly ScoreboardRow[], opts?: ScoreboardRenderOptions): string;
|
|
1111
|
+
declare function parseSkillPatchResponse(raw: string, maxPatches: number, editBudget: number): SkillPatch[];
|
|
1162
1112
|
|
|
1163
1113
|
/**
|
|
1164
|
-
* @experimental
|
|
1165
|
-
*
|
|
1166
1114
|
* `runSkillOpt` — the SkillOpt epoch hill-climb (Microsoft, arXiv:2605.23904).
|
|
1167
1115
|
* Unlike `runOptimization`'s population/promote-top-K search, SkillOpt is a
|
|
1168
1116
|
* sequential, held-out-gated hill-climb on ONE skill document:
|
|
@@ -1183,7 +1131,7 @@ declare function renderScoreboardMarkdown(rows: readonly ScoreboardRow[], opts?:
|
|
|
1183
1131
|
* `HeldOutGate`/`defaultProductionGate`, applied per edit instead of once at
|
|
1184
1132
|
* the end — which is why the held-out composite is monotonically
|
|
1185
1133
|
* non-decreasing and a regression can never ship. `runCampaign` is the
|
|
1186
|
-
* measurement; `applySkillPatch` applies the edits; `
|
|
1134
|
+
* measurement; `applySkillPatch` applies the edits; `skillOptProposer` proposes
|
|
1187
1135
|
* them.
|
|
1188
1136
|
*/
|
|
1189
1137
|
|
|
@@ -1192,7 +1140,7 @@ interface RunSkillOptOptions<TScenario extends Scenario, TArtifact> extends Omit
|
|
|
1192
1140
|
baselineSurface: string;
|
|
1193
1141
|
/** Dispatcher taking the CURRENT skill surface + scenario → artifact. */
|
|
1194
1142
|
dispatchWithSurface: (surface: string, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1195
|
-
|
|
1143
|
+
proposer: SkillOptProposer;
|
|
1196
1144
|
/** Scenarios the optimizer reflects on for evidence. MUST be disjoint from
|
|
1197
1145
|
* `holdoutScenarios` — proposals never see the acceptance axis. */
|
|
1198
1146
|
trainScenarios: TScenario[];
|
|
@@ -1257,10 +1205,175 @@ interface RunSkillOptResult {
|
|
|
1257
1205
|
declare function runSkillOpt<TScenario extends Scenario, TArtifact>(opts: RunSkillOptOptions<TScenario, TArtifact>): Promise<RunSkillOptResult>;
|
|
1258
1206
|
|
|
1259
1207
|
/**
|
|
1260
|
-
*
|
|
1208
|
+
* `aceProposer` — Agentic Context Engineering: an APPEND-MOSTLY curator, the
|
|
1209
|
+
* deliberate contrast to `memoryCurationProposer`'s dedup-and-replace. ACE's
|
|
1210
|
+
* thesis (arXiv:2510.04618) is that aggressively deduping/rewriting a context
|
|
1211
|
+
* causes "context collapse" — hard-won specific lessons get summarized away. So
|
|
1212
|
+
* the playbook GROWS by appending each generation's new lessons as provenance-
|
|
1213
|
+
* tagged delta bullets; existing bullets are preserved verbatim, never merged.
|
|
1214
|
+
*
|
|
1215
|
+
* Each generation it:
|
|
1216
|
+
* 1. reads the playbook block already in the parent surface (verbatim);
|
|
1217
|
+
* 2. turns this generation's `findings` into lessons, keeping only the ones not
|
|
1218
|
+
* already present (idempotency — a recurring finding is not re-appended, but
|
|
1219
|
+
* a genuinely NEW lesson always is, even if similar to an old one);
|
|
1220
|
+
* 3. appends the new lessons as `- [gN] <lesson>` deltas and re-emits the block.
|
|
1221
|
+
*
|
|
1222
|
+
* Bounded WITHOUT collapse: when the playbook exceeds `maxEntries`, the OLDEST
|
|
1223
|
+
* deltas are evicted (FIFO) — recency is kept, but no two distinct lessons are
|
|
1224
|
+
* ever merged into one. Deterministic (no LLM) so a lift is attributable to the
|
|
1225
|
+
* accumulated lessons, not a rewrite's model noise.
|
|
1226
|
+
*
|
|
1227
|
+
* Fail-loud: with no new lesson this generation it returns NO candidate (the
|
|
1228
|
+
* playbook is unchanged — nothing to propose), never a fabricated bullet.
|
|
1229
|
+
*/
|
|
1230
|
+
|
|
1231
|
+
interface AceProposerOptions {
|
|
1232
|
+
/** Max delta bullets retained in the playbook. On overflow the OLDEST are
|
|
1233
|
+
* evicted (FIFO) — never merged. Default 50 (ACE keeps a long context). */
|
|
1234
|
+
maxEntries?: number;
|
|
1235
|
+
/** Heading rendered above the bullets inside the block. */
|
|
1236
|
+
sectionHeading?: string;
|
|
1237
|
+
}
|
|
1238
|
+
declare function aceProposer(opts?: AceProposerOptions): SurfaceProposer;
|
|
1239
|
+
|
|
1240
|
+
/**
|
|
1241
|
+
* `haloProposer` — wraps the REAL halo-engine (Inference.net's hierarchical
|
|
1242
|
+
* agentic trace analyzer, `pip install halo-engine`, repo context-labs/halo)
|
|
1243
|
+
* as an agent-eval `SurfaceProposer`, so HALO competes head-to-head with
|
|
1244
|
+
* `gepaProposer` — and with our own `traceAnalystProposer` — inside `compareProposers`
|
|
1245
|
+
* on identical traces / scenarios / held-out scoring.
|
|
1246
|
+
*
|
|
1247
|
+
* It PRESERVES halo's actual working usage — `analyze` shells out to the
|
|
1248
|
+
* published CLI (`halo <traces.jsonl> -p <prompt> -m <model>`) and uses its real
|
|
1249
|
+
* RLM findings verbatim. We do NOT reimplement its analysis; that would make the
|
|
1250
|
+
* benchmark meaningless. The materialize/apply pipeline is the shared
|
|
1251
|
+
* `analysisEditProposer` — identical to `traceAnalystProposer`, which is what makes
|
|
1252
|
+
* the comparison apples-to-apples.
|
|
1253
|
+
*
|
|
1254
|
+
* Fail-loud: no traces → throw; halo errors → throw; empty findings → throw.
|
|
1255
|
+
*/
|
|
1256
|
+
|
|
1257
|
+
interface HaloProposerOptions {
|
|
1258
|
+
/** OpenAI-compatible base URL for BOTH halo's RLM analysis and the apply
|
|
1259
|
+
* step (e.g. the Tangle router `https://router.tangle.tools/v1`). */
|
|
1260
|
+
baseUrl: string;
|
|
1261
|
+
/** Bearer key (else relies on OPENAI_API_KEY in the env halo inherits). */
|
|
1262
|
+
apiKey?: string;
|
|
1263
|
+
/** Model for halo's `--model` (its RLM). Default 'gpt-5.4-mini' (halo's own default). */
|
|
1264
|
+
model?: string;
|
|
1265
|
+
/** Model used to APPLY halo's findings to the prompt surface. Default = `model`. */
|
|
1266
|
+
applyModel?: string;
|
|
1267
|
+
/** The real halo binary. Default 'halo' (from `pip install halo-engine`). */
|
|
1268
|
+
haloBin?: string;
|
|
1269
|
+
/** Resolve the OTLP traces (JSONL string) halo should analyze for THIS
|
|
1270
|
+
* generation. Returning empty throws (halo has nothing to analyze). */
|
|
1271
|
+
resolveTraces: (ctx: ProposeContext) => string | Promise<string>;
|
|
1272
|
+
/** halo's analysis prompt (`-p`). Default targets the failure taxonomy. */
|
|
1273
|
+
analysisPrompt?: string;
|
|
1274
|
+
/** halo `--max-depth` / `--max-turns` passthrough. */
|
|
1275
|
+
maxDepth?: number;
|
|
1276
|
+
maxTurns?: number;
|
|
1277
|
+
/** Test seam: inject a fetch for the apply-step callLlm (no network in unit tests). */
|
|
1278
|
+
fetchImpl?: LlmClientOptions['fetch'];
|
|
1279
|
+
}
|
|
1280
|
+
/** Wrap the real halo-engine CLI as a SurfaceProposer (prompt-tier). */
|
|
1281
|
+
declare function haloProposer(opts: HaloProposerOptions): SurfaceProposer;
|
|
1282
|
+
|
|
1283
|
+
/**
|
|
1284
|
+
* `memoryCurationProposer` — a CURATOR `SurfaceProposer`, the complement to the
|
|
1285
|
+
* OPTIMIZER proposers (`gepaProposer` rewrites the prompt; this one BUILDS a
|
|
1286
|
+
* searchable memory of what prior trajectories taught and grafts the most
|
|
1287
|
+
* relevant lessons onto the surface).
|
|
1288
|
+
*
|
|
1289
|
+
* Each generation it:
|
|
1290
|
+
* 1. collects lessons — this generation's trace-analyst `findings` PLUS the
|
|
1291
|
+
* memory already carried in the parent surface (so memory accumulates
|
|
1292
|
+
* across generations instead of resetting);
|
|
1293
|
+
* 2. curates them — normalizes, deduplicates near-identical lessons, and ranks
|
|
1294
|
+
* by recurrence (a lesson seen across many findings outranks a one-off);
|
|
1295
|
+
* 3. retrieves the top-K and writes them back as a single delimited memory
|
|
1296
|
+
* block in the surface (idempotent — the block is replaced, never stacked,
|
|
1297
|
+
* so the prompt does not grow without bound).
|
|
1298
|
+
*
|
|
1299
|
+
* This is the substrate behind the "knowledge base of working trajectories" the
|
|
1300
|
+
* agent searches: the curated block IS the retrieved memory the next run reads.
|
|
1301
|
+
* Curation is DETERMINISTIC (no LLM) so a lift it produces is attributable to
|
|
1302
|
+
* the lessons, not to model noise in a rewrite. An optional `distill` LLM step
|
|
1303
|
+
* can compress raw findings into crisp imperatives; default is verbatim.
|
|
1304
|
+
*
|
|
1305
|
+
* Fail-loud: never fabricates a lesson. With no findings and no prior memory it
|
|
1306
|
+
* returns no candidate (nothing learned yet — gen 0). It does not throw on an
|
|
1307
|
+
* empty generation because early generations legitimately have no findings.
|
|
1308
|
+
*/
|
|
1309
|
+
|
|
1310
|
+
interface MemoryCurationProposerOptions {
|
|
1311
|
+
/** Top-K lessons retained in the surface memory block. Default 12. */
|
|
1312
|
+
maxEntries?: number;
|
|
1313
|
+
/** Heading rendered above the lessons inside the block. Default below. */
|
|
1314
|
+
sectionHeading?: string;
|
|
1315
|
+
/**
|
|
1316
|
+
* Optional LLM distillation: compress raw findings into crisp, generalizable
|
|
1317
|
+
* one-line imperatives before curating. Omit for verbatim (deterministic).
|
|
1318
|
+
*/
|
|
1319
|
+
distill?: {
|
|
1320
|
+
baseUrl: string;
|
|
1321
|
+
apiKey?: string;
|
|
1322
|
+
model: string;
|
|
1323
|
+
fetchImpl?: LlmClientOptions['fetch'];
|
|
1324
|
+
};
|
|
1325
|
+
}
|
|
1326
|
+
/** Build the CURATOR proposer. */
|
|
1327
|
+
declare function memoryCurationProposer(opts?: MemoryCurationProposerOptions): SurfaceProposer;
|
|
1328
|
+
|
|
1329
|
+
/**
|
|
1330
|
+
* `traceAnalystProposer` — wraps agent-eval's OWN trace-analyst engine
|
|
1331
|
+
* (`AnalystRegistry` over the agentic OTLP reader) as a `SurfaceProposer`.
|
|
1332
|
+
* It is the symmetric opponent to `haloProposer`: both run the SAME shared
|
|
1333
|
+
* `analysisEditProposer` pipeline (materialize identical traces → apply via one
|
|
1334
|
+
* identical LLM edit), so a `compareProposers` lift delta isolates a single
|
|
1335
|
+
* variable — ANALYSIS QUALITY. The benchmark answers "is our HALO clone as good
|
|
1336
|
+
* as the real HALO?" as a held-out lift CI, not a vibe.
|
|
1337
|
+
*
|
|
1338
|
+
* Findings come from the REGISTRY (structured `AnalystFinding[]` carrying
|
|
1339
|
+
* area / severity / recommended_action), rendered into the report the shared
|
|
1340
|
+
* apply step consumes.
|
|
1261
1341
|
*
|
|
1342
|
+
* Fail-loud: no traces → throw; analyst run errors → throw; zero findings →
|
|
1343
|
+
* throw. Never fabricate a candidate.
|
|
1344
|
+
*/
|
|
1345
|
+
|
|
1346
|
+
interface TraceAnalystProposerOptions {
|
|
1347
|
+
/** OpenAI-compatible base URL for BOTH the analyst's agentic reads and the
|
|
1348
|
+
* apply step (e.g. `https://api.deepseek.com/v1` or the Tangle router). */
|
|
1349
|
+
baseUrl: string;
|
|
1350
|
+
/** Bearer key. Required — the Ax AI service has no env fallback here. */
|
|
1351
|
+
apiKey: string;
|
|
1352
|
+
/** Model the analyst kinds use for their agentic trace reads. */
|
|
1353
|
+
model: string;
|
|
1354
|
+
/** Model used to APPLY findings to the prompt surface. Default = `model`.
|
|
1355
|
+
* Keep this EQUAL to haloProposer's `applyModel` for an apples-to-apples run. */
|
|
1356
|
+
applyModel?: string;
|
|
1357
|
+
/** Ax provider name. Default 'openai' — works for any OpenAI-compatible base
|
|
1358
|
+
* via `apiURL`. Use 'deepseek' to hit DeepSeek's native provider. */
|
|
1359
|
+
provider?: string;
|
|
1360
|
+
/** Which analyst kinds to run. Default = the full shipped suite. */
|
|
1361
|
+
kinds?: readonly TraceAnalystKindSpec[];
|
|
1362
|
+
/** Resolve the OTLP traces (JSONL string) the analyst should read for THIS
|
|
1363
|
+
* generation — identical contract to `haloProposer.resolveTraces`. */
|
|
1364
|
+
resolveTraces: (ctx: ProposeContext) => string | Promise<string>;
|
|
1365
|
+
/** Override the findings producer. Default: the shipped `AnalystRegistry`
|
|
1366
|
+
* over `kinds`. The unit suite injects canned findings here. */
|
|
1367
|
+
analyze?: (tracePath: string, ctx: ProposeContext) => Promise<ReadonlyArray<AnalystFinding>>;
|
|
1368
|
+
/** Test seam: inject a fetch for the apply-step `callLlm`. */
|
|
1369
|
+
fetchImpl?: LlmClientOptions['fetch'];
|
|
1370
|
+
}
|
|
1371
|
+
/** Wrap agent-eval's trace-analyst registry as a SurfaceProposer (prompt-tier). */
|
|
1372
|
+
declare function traceAnalystProposer(opts: TraceAnalystProposerOptions): SurfaceProposer;
|
|
1373
|
+
|
|
1374
|
+
/**
|
|
1262
1375
|
* Shared campaign-score reductions used by every optimizer preset
|
|
1263
|
-
* (`runOptimization`, `runSkillOpt`, `
|
|
1376
|
+
* (`runOptimization`, `runSkillOpt`, `compareProposers`). ONE definition of
|
|
1264
1377
|
* "composite of a campaign" and "per-scenario / per-dimension breakdown" so
|
|
1265
1378
|
* the optimizers cannot drift on how a surface's score is computed.
|
|
1266
1379
|
*/
|
|
@@ -1273,29 +1386,27 @@ interface CampaignBreakdown {
|
|
|
1273
1386
|
/** Mean score per judge dimension across all cells. */
|
|
1274
1387
|
dimensions: Record<string, number>;
|
|
1275
1388
|
/** Per-scenario composite (mean over reps + judges) + the judge's free-form
|
|
1276
|
-
* `notes` for that scenario (the "why" a reflective
|
|
1389
|
+
* `notes` for that scenario (the "why" a reflective proposer grounds on). */
|
|
1277
1390
|
scenarios: Array<{
|
|
1278
1391
|
scenarioId: string;
|
|
1279
1392
|
composite: number;
|
|
1280
1393
|
notes?: string;
|
|
1281
1394
|
}>;
|
|
1282
1395
|
}
|
|
1283
|
-
/** Per-candidate evidence a reflective/patch
|
|
1396
|
+
/** Per-candidate evidence a reflective/patch proposer grounds its next proposal
|
|
1284
1397
|
* on: mean score per judge dimension + per-scenario composite. */
|
|
1285
1398
|
declare function campaignBreakdown<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): CampaignBreakdown;
|
|
1286
1399
|
|
|
1287
1400
|
/**
|
|
1288
|
-
* @experimental
|
|
1289
|
-
*
|
|
1290
1401
|
* VCS-pluggable worktree adapter. One improvement = one worktree, PR-like
|
|
1291
|
-
* (multiple commits allowed). A code-tier
|
|
1402
|
+
* (multiple commits allowed). A code-tier proposer's `propose()` creates a
|
|
1292
1403
|
* worktree, an agent commits the change into it, and `finalize()` returns a
|
|
1293
1404
|
* `CodeSurface{ worktreeRef }` the measurement checks out to run the worker
|
|
1294
1405
|
* against the changed code. On promotion the worktree becomes the PR branch.
|
|
1295
1406
|
*
|
|
1296
1407
|
* The interface is VCS-agnostic so a future `jj` ([jj-vcs](https://github.com/jj-vcs/jj))
|
|
1297
|
-
* adapter can slot in without touching
|
|
1298
|
-
* ships today. See `docs/design/
|
|
1408
|
+
* adapter can slot in without touching proposer code. Only the git adapter
|
|
1409
|
+
* ships today. See `docs/design/loop-taxonomy.md`.
|
|
1299
1410
|
*/
|
|
1300
1411
|
|
|
1301
1412
|
interface Worktree {
|
|
@@ -1339,4 +1450,4 @@ declare function gitWorktreeAdapter(opts: GitWorktreeAdapterOptions): WorktreeAd
|
|
|
1339
1450
|
* as a ref under the adapter's worktree dir. */
|
|
1340
1451
|
declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
|
|
1341
1452
|
|
|
1342
|
-
export { type AcceptedEdit, type
|
|
1453
|
+
export { type AcceptedEdit, type AceProposerOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignStorage, CodeSurface, type CompareProposersOptions, type DimensionRegression, DispatchContext, type FailureModeRecallJudgeOptions, type FapoAttributionSignals, type FapoEntryConfig, type FapoFailureCluster, type FapoOptimizationLevel, type FapoProposerOptions, type FapoReviewInput, type FapoReviewIssue, type FapoReviewResult, type FapoScopeContract, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, Gate, GenerationRecord, type GitWorktreeAdapterOptions, type HaloProposerOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type JsonPrimitive, type JsonValue, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, type MemoryCurationProposerOptions, MutableSurface, type OptimizerEntryConfig, type PairedHoldout, type ParameterCandidate, type ParameterChange, type ParameterSweepProposerOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, ProposedCandidate, type ProposerComparison, type ProposerEntry, type ProposerPairwise, type ProposerScore, type RejectedEdit, RunCampaignOptions, RunImprovementLoopOptions, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, Scenario, type ScenarioRollup, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillOptProposer, type SkillOptProposerOptions, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, SurfaceProposer, type TraceAnalystProposerOptions, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceProposer, applySkillPatch, buildAnalystSurfaceDispatch, campaignBreakdown, campaignMeanComposite, compareProposers, detectScale, dimensionRegressions, extractFapoAttributionSignals, failureModeRecallJudge, fapoEscalationEntry, fapoProposer, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloProposer, heldoutSignificance, makePlaybackDispatch, memoryCurationProposer, pairHoldout, parameterSweepProposer, parseSkillPatchResponse, patchEditCount, renderScoreboardMarkdown, resolveWorktreePath, runProfileMatrix, runSkillOpt, scoreUserStory, scoreboardSummary, sequentialDecide, sequentialPairedGate, skillOptEntry, skillOptProposer, traceAnalystProposer, userStoryScoreboard };
|