@tangle-network/agent-runtime 0.95.0 → 0.97.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +64 -15
- package/dist/activation-B0ZD7nfX.d.ts +63 -0
- package/dist/agent.d.ts +5 -169
- package/dist/agent.js +8 -229
- package/dist/agent.js.map +1 -1
- package/dist/analyst-loop.d.ts +6 -9
- package/dist/analyst-loop.js +1 -2
- package/dist/candidate-execution/index.js +6 -7
- package/dist/chunk-3XKSBI2U.js +474 -0
- package/dist/chunk-3XKSBI2U.js.map +1 -0
- package/dist/{chunk-6YBA64Z2.js → chunk-6XKPVJAZ.js} +5 -18
- package/dist/chunk-6XKPVJAZ.js.map +1 -0
- package/dist/{chunk-MKGRLDWB.js → chunk-BLQIYRVR.js} +17 -2
- package/dist/chunk-BLQIYRVR.js.map +1 -0
- package/dist/{chunk-QDSOD7RC.js → chunk-FD2MBMOH.js} +13 -101
- package/dist/chunk-FD2MBMOH.js.map +1 -0
- package/dist/{chunk-YLUOTX6U.js → chunk-FXF2OL34.js} +7 -7
- package/dist/{chunk-EP6RVHMX.js → chunk-HZDEXTSL.js} +848 -2
- package/dist/chunk-HZDEXTSL.js.map +1 -0
- package/dist/chunk-PSOCBNM3.js +2069 -0
- package/dist/chunk-PSOCBNM3.js.map +1 -0
- package/dist/{chunk-BPGXIKK7.js → chunk-SGQ4YIQW.js} +4 -4
- package/dist/{chunk-IADLKE7I.js → chunk-UQ6PNNXM.js} +5 -7
- package/dist/{chunk-IADLKE7I.js.map → chunk-UQ6PNNXM.js.map} +1 -1
- package/dist/{chunk-Z5I642SY.js → chunk-WYC2XJF2.js} +2 -2
- package/dist/{chunk-ZEYAT33L.js → chunk-Y3SRWZMP.js} +2 -2
- package/dist/{chunk-WTZ37EQY.js → chunk-YOLKCWRV.js} +197 -90
- package/dist/chunk-YOLKCWRV.js.map +1 -0
- package/dist/conversation.js +0 -1
- package/dist/environment-provider.js +0 -1
- package/dist/{agentic-generator-hCaQRAes.d.ts → improve-g75IE2Cx.d.ts} +152 -3
- package/dist/{improvement-adapter-BieWeK5J.d.ts → improvement-adapter-HAZz-7vK.d.ts} +8 -31
- package/dist/index.d.ts +41 -11
- package/dist/index.js +180 -51
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +16 -9
- package/dist/intelligence.js +13 -8
- package/dist/intelligence.js.map +1 -1
- package/dist/knowledge.d.ts +30 -13
- package/dist/knowledge.js +11 -10
- package/dist/{loop-runner-bin-BIQldFS8.d.ts → loop-runner-bin-Cn1N2rRo.d.ts} +1 -1
- package/dist/loop-runner-bin.d.ts +2 -2
- package/dist/loop-runner-bin.js +6 -8
- package/dist/loops.d.ts +1 -1
- package/dist/loops.js +4 -6
- package/dist/mcp/bin.js +3 -5
- package/dist/mcp/bin.js.map +1 -1
- package/dist/mcp/index.js +10 -12
- package/dist/mcp/index.js.map +1 -1
- package/dist/platform.js +0 -2
- package/dist/platform.js.map +1 -1
- package/dist/primeintellect/index.js +0 -1
- package/dist/primeintellect/index.js.map +1 -1
- package/dist/profiles.js +0 -1
- package/dist/profiles.js.map +1 -1
- package/dist/{types-BC3bZpH0.d.ts → types-CmYCMbFT.d.ts} +12 -54
- package/package.json +8 -13
- package/skills/build-with-agent-runtime/SKILL.md +122 -213
- package/dist/chunk-6O73TRHW.js +0 -142
- package/dist/chunk-6O73TRHW.js.map +0 -1
- package/dist/chunk-6YBA64Z2.js.map +0 -1
- package/dist/chunk-AP7CPGMZ.js +0 -334
- package/dist/chunk-AP7CPGMZ.js.map +0 -1
- package/dist/chunk-DGUM43GV.js +0 -11
- package/dist/chunk-DGUM43GV.js.map +0 -1
- package/dist/chunk-DHCHL6OG.js +0 -625
- package/dist/chunk-DHCHL6OG.js.map +0 -1
- package/dist/chunk-EP6RVHMX.js.map +0 -1
- package/dist/chunk-G55QE4IQ.js +0 -1137
- package/dist/chunk-G55QE4IQ.js.map +0 -1
- package/dist/chunk-ISTDY47H.js +0 -849
- package/dist/chunk-ISTDY47H.js.map +0 -1
- package/dist/chunk-MKGRLDWB.js.map +0 -1
- package/dist/chunk-QDSOD7RC.js.map +0 -1
- package/dist/chunk-WTZ37EQY.js.map +0 -1
- package/dist/generator-YkAQrOoD.d.ts +0 -382
- package/dist/improve-B-UYaEH5.d.ts +0 -172
- package/dist/lifecycle.d.ts +0 -870
- package/dist/lifecycle.js +0 -981
- package/dist/lifecycle.js.map +0 -1
- package/dist/mcp-serve-verifier-Bs_n0xPc.d.ts +0 -34
- package/skills/agent-runtime-adoption/SKILL.md +0 -246
- /package/dist/{chunk-YLUOTX6U.js.map → chunk-FXF2OL34.js.map} +0 -0
- /package/dist/{chunk-BPGXIKK7.js.map → chunk-SGQ4YIQW.js.map} +0 -0
- /package/dist/{chunk-Z5I642SY.js.map → chunk-WYC2XJF2.js.map} +0 -0
- /package/dist/{chunk-ZEYAT33L.js.map → chunk-Y3SRWZMP.js.map} +0 -0
|
@@ -1,382 +0,0 @@
|
|
|
1
|
-
import { RunRecord, AnalystFinding } from '@tangle-network/agent-eval';
|
|
2
|
-
import { AgentProfileResourceRef, AgentProfileMcpServer, AgentProfileHookCommand, AgentSubagentProfile, AgentProfile } from '@tangle-network/agent-interface';
|
|
3
|
-
|
|
4
|
-
/**
|
|
5
|
-
* `@tangle-network/agent-runtime/lifecycle` — artifact-lifecycle FOUNDATION.
|
|
6
|
-
*
|
|
7
|
-
* The §1.5 law says an agent IS its `AgentProfile`, and the profile is the WHOLE
|
|
8
|
-
* agent: prompt + skills + tools + mcp + hooks + subagents. This module names the
|
|
9
|
-
* discrete, individually-promotable PIECES of that profile — "artifacts" — and
|
|
10
|
-
* gives them stable ids so the rest of a self-improvement lifecycle (propose →
|
|
11
|
-
* measure → promote → ship) has something concrete to hang off.
|
|
12
|
-
*
|
|
13
|
-
* This is PHASE 1: just the two primitives the rest hangs off.
|
|
14
|
-
* 1. `ArtifactRegistry` — a typed catalog of profile artifacts with stable ids.
|
|
15
|
-
* 2. `measureMarginalLift` — the with-vs-without ablation: how much score/cost a
|
|
16
|
-
* single artifact adds on top of a baseline profile.
|
|
17
|
-
*
|
|
18
|
-
* The per-surface lifecycles, the `BuildableSurface` author contract, and the
|
|
19
|
-
* promotion-gate wiring are deferred to later phases.
|
|
20
|
-
*/
|
|
21
|
-
|
|
22
|
-
/**
|
|
23
|
-
* The profile levers an artifact can target. One-to-one with the §1.5 profile
|
|
24
|
-
* surface (`prompt + skills + tools + mcp + hooks + subagents`). Each kind maps to
|
|
25
|
-
* exactly one field of `AgentProfile`, so an artifact can be applied onto a
|
|
26
|
-
* baseline profile deterministically (see `applyArtifact`).
|
|
27
|
-
*/
|
|
28
|
-
type ArtifactKind = 'skill' | 'tool' | 'mcp' | 'hook' | 'subagent' | 'prompt';
|
|
29
|
-
/**
|
|
30
|
-
* The payload for each `ArtifactKind`. The shapes are the SAME types the
|
|
31
|
-
* `AgentProfile` field carries, so applying an artifact is a structural merge
|
|
32
|
-
* onto the profile — never a bespoke per-kind transform.
|
|
33
|
-
*
|
|
34
|
-
* - `prompt` — an instruction line appended to `profile.prompt.instructions`.
|
|
35
|
-
* - `skill` — a `SKILL.md`-style resource ref added to `profile.resources.skills`.
|
|
36
|
-
* - `tool` — a tool grant: `{ enabled }` set under `profile.tools[name]`.
|
|
37
|
-
* - `mcp` — one MCP server added under `profile.mcp[name]`.
|
|
38
|
-
* - `hook` — one or more hook commands added under `profile.hooks[event]`.
|
|
39
|
-
* - `subagent` — one subagent profile added under `profile.subagents[name]`.
|
|
40
|
-
*/
|
|
41
|
-
interface ArtifactPayloads {
|
|
42
|
-
prompt: {
|
|
43
|
-
instruction: string;
|
|
44
|
-
};
|
|
45
|
-
skill: {
|
|
46
|
-
resource: AgentProfileResourceRef;
|
|
47
|
-
};
|
|
48
|
-
tool: {
|
|
49
|
-
enabled: boolean;
|
|
50
|
-
};
|
|
51
|
-
mcp: {
|
|
52
|
-
server: AgentProfileMcpServer;
|
|
53
|
-
};
|
|
54
|
-
hook: {
|
|
55
|
-
event: string;
|
|
56
|
-
commands: AgentProfileHookCommand[];
|
|
57
|
-
};
|
|
58
|
-
subagent: {
|
|
59
|
-
profile: AgentSubagentProfile;
|
|
60
|
-
};
|
|
61
|
-
}
|
|
62
|
-
/**
|
|
63
|
-
* A discrete, individually-promotable piece of an agent profile.
|
|
64
|
-
*
|
|
65
|
-
* `kind` selects the profile lever; `payload` is the kind-specific value; `key`
|
|
66
|
-
* is the profile-field key the payload lands under (the tool name, the MCP server
|
|
67
|
-
* name, the subagent name — unused for `prompt`, which appends). `id` is stable:
|
|
68
|
-
* once registered, it never changes, so a marginal-lift measurement, a promotion
|
|
69
|
-
* decision, and a ship record all reference the same artifact.
|
|
70
|
-
*/
|
|
71
|
-
interface ProfileArtifact<K extends ArtifactKind = ArtifactKind> {
|
|
72
|
-
/** Stable id. Assigned by the registry at register time; never reassigned. */
|
|
73
|
-
id: string;
|
|
74
|
-
kind: K;
|
|
75
|
-
/**
|
|
76
|
-
* The profile-field key this artifact lands under (e.g. the tool name, the MCP
|
|
77
|
-
* server name, the subagent name, the hook event). Optional for `prompt`
|
|
78
|
-
* (instructions append, they have no key). Defaults to `id` when applying a
|
|
79
|
-
* keyed artifact without an explicit key.
|
|
80
|
-
*/
|
|
81
|
-
key?: string;
|
|
82
|
-
/** Human-facing label for review surfaces. */
|
|
83
|
-
name: string;
|
|
84
|
-
/** Optional one-line description of what this artifact does. */
|
|
85
|
-
description?: string;
|
|
86
|
-
payload: ArtifactPayloads[K];
|
|
87
|
-
/**
|
|
88
|
-
* Lifecycle status — the full artifact state machine:
|
|
89
|
-
*
|
|
90
|
-
* `candidate` → `active` → `decayed` (re-promotable)
|
|
91
|
-
* ↘ `retired` (terminal)
|
|
92
|
-
*
|
|
93
|
-
* - `candidate` — registered, not yet promoted. The default at register time.
|
|
94
|
-
* - `active` — passed the promotion gate and carries a measured held-back
|
|
95
|
-
* lift; the only status `composeProfile` folds into a profile.
|
|
96
|
-
* - `decayed` — was active, but a later re-measure (`driftWatch`) found its
|
|
97
|
-
* lift fell below the keep-bar. Demoted out of the composed
|
|
98
|
-
* profile; kept as an auditable record and a re-promotion
|
|
99
|
-
* candidate if a future re-measure recovers the lift.
|
|
100
|
-
* - `retired` — permanently removed from the active set (`dedupeArtifacts`
|
|
101
|
-
* retires the weaker half of a non-stacking pair). Terminal.
|
|
102
|
-
*
|
|
103
|
-
* The registry never auto-promotes; every transition is an explicit call
|
|
104
|
-
* (`promote`/`promoteWithLift`/`demote`/`retire`).
|
|
105
|
-
*/
|
|
106
|
-
status: ArtifactStatus;
|
|
107
|
-
/** Free-form metadata (provenance, generation id, the measured lift, …). */
|
|
108
|
-
metadata?: Record<string, unknown>;
|
|
109
|
-
}
|
|
110
|
-
/**
|
|
111
|
-
* The artifact lifecycle states. `active` is the load-bearing one — it is the
|
|
112
|
-
* sole status `composeProfile` folds into a deployable profile, and it is gated
|
|
113
|
-
* by a measured held-back lift (the registry invariant). `decayed` and `retired`
|
|
114
|
-
* are the two ways an artifact LEAVES the active set: a decayed artifact lost its
|
|
115
|
-
* lift on re-measure (reversible — `driftWatch`), a retired one was deduped away
|
|
116
|
-
* (terminal — `dedupeArtifacts`).
|
|
117
|
-
*/
|
|
118
|
-
type ArtifactStatus = 'candidate' | 'active' | 'decayed' | 'retired';
|
|
119
|
-
/** The input to `register` — everything on `ProfileArtifact` except the
|
|
120
|
-
* registry-owned `id` and `status`. An explicit `id` may be supplied for
|
|
121
|
-
* deterministic/idempotent registration; otherwise the registry assigns one. */
|
|
122
|
-
type ArtifactInput<K extends ArtifactKind = ArtifactKind> = Omit<ProfileArtifact<K>, 'id' | 'status'> & {
|
|
123
|
-
id?: string;
|
|
124
|
-
status?: ArtifactStatus;
|
|
125
|
-
};
|
|
126
|
-
|
|
127
|
-
/**
|
|
128
|
-
* `measureMarginalLift` — the with-vs-without ablation for one artifact.
|
|
129
|
-
*
|
|
130
|
-
* "What does THIS one piece add?" is the question a lifecycle has to answer
|
|
131
|
-
* before it can promote anything. The answer is an ablation: score the baseline
|
|
132
|
-
* profile, score the baseline-plus-candidate profile, and report the delta. This
|
|
133
|
-
* is the marginal contribution of the artifact — the same shape the project's
|
|
134
|
-
* `OutcomeMeasurement` reports for an apply pass, but isolated to ONE artifact so
|
|
135
|
-
* a registry can rank candidates by what they individually earn.
|
|
136
|
-
*
|
|
137
|
-
* The measurement is selector-agnostic: it takes an `EvalRunner` the caller
|
|
138
|
-
* supplies (any function that scores a profile and reports cost), runs it twice,
|
|
139
|
-
* and subtracts. It never judges, never picks a winner, never re-scores on a
|
|
140
|
-
* holdout — those are the gate's job. It only quantifies the ablation so the gate
|
|
141
|
-
* has a real number to decide on.
|
|
142
|
-
*/
|
|
143
|
-
|
|
144
|
-
/**
|
|
145
|
-
* The result of running an eval over ONE profile: a composite score and the cost
|
|
146
|
-
* to obtain it. This mirrors the project's score/cost convention (`composite`
|
|
147
|
-
* from `OutcomeMeasurement`, `costUsd` from `LoopResult`), so a caller can pass a
|
|
148
|
-
* thin wrapper over `runLoop` / `runBenchmark` / `runAgentEval` directly.
|
|
149
|
-
*/
|
|
150
|
-
interface EvalResult {
|
|
151
|
-
/** Composite score in `[0, 1]` (higher is better) for the profile under test. */
|
|
152
|
-
composite: number;
|
|
153
|
-
/** USD cost to produce this result. */
|
|
154
|
-
costUsd: number;
|
|
155
|
-
/**
|
|
156
|
-
* Per-task records the run produced, when the runner emits them. The marginal
|
|
157
|
-
* lift only needs `composite`, but the held-out promotion gate (`HeldOutGate`)
|
|
158
|
-
* pairs candidate vs baseline per-task holdout records by (experimentId, seed)
|
|
159
|
-
* — so a runner feeding `heldOutPromotionGate` MUST populate this with rows
|
|
160
|
-
* carrying both `search` and `holdout` split scores. Omit it for evals scored
|
|
161
|
-
* to a single composite (then use `thresholdPromotionGate`).
|
|
162
|
-
*/
|
|
163
|
-
runs?: RunRecord[];
|
|
164
|
-
/** Optional opaque passthrough (per-task cells, the raw report, …). */
|
|
165
|
-
details?: unknown;
|
|
166
|
-
}
|
|
167
|
-
/**
|
|
168
|
-
* Scores a profile. The caller wires this to whatever eval they run — a
|
|
169
|
-
* `runLoop` rollout, a `runBenchmark` campaign, a `runAgentEval` cohort — and
|
|
170
|
-
* returns the composite + cost. `signal` is forwarded for cancellation.
|
|
171
|
-
*/
|
|
172
|
-
type EvalRunner = (profile: AgentProfile, signal?: AbortSignal) => Promise<EvalResult>;
|
|
173
|
-
interface MeasureMarginalLiftOptions {
|
|
174
|
-
/** The profile the artifact is measured ON TOP OF (the "without" arm). */
|
|
175
|
-
baseline: AgentProfile;
|
|
176
|
-
/** The single artifact whose marginal contribution we want. */
|
|
177
|
-
candidate: ProfileArtifact;
|
|
178
|
-
/** The eval that scores a profile. Run once per arm (twice total). */
|
|
179
|
-
evalRunner: EvalRunner;
|
|
180
|
-
/**
|
|
181
|
-
* A pre-computed baseline result, to skip the "without" run when the caller
|
|
182
|
-
* already scored the baseline (e.g. measuring several candidates against the
|
|
183
|
-
* same baseline). When set, the baseline arm is NOT re-run.
|
|
184
|
-
*/
|
|
185
|
-
baselineResult?: EvalResult;
|
|
186
|
-
/** Forwarded to both `evalRunner` invocations for cancellation. */
|
|
187
|
-
signal?: AbortSignal;
|
|
188
|
-
}
|
|
189
|
-
/**
|
|
190
|
-
* The marginal lift of one artifact: the with/without ablation.
|
|
191
|
-
*
|
|
192
|
-
* `scoreDelta = with.composite − without.composite`. A positive `scoreDelta` is
|
|
193
|
-
* the evidence a gate needs to promote; a negative one is the signal to drop the
|
|
194
|
-
* artifact. `costDelta` is the extra USD the artifact costs (often positive — a
|
|
195
|
-
* new tool/MCP adds calls) and lets the gate weigh lift against spend.
|
|
196
|
-
*/
|
|
197
|
-
interface MarginalLift {
|
|
198
|
-
/** The artifact id this measurement is for (stable, from the registry). */
|
|
199
|
-
artifactId: string;
|
|
200
|
-
/** Eval of `applyArtifact(baseline, candidate)`. */
|
|
201
|
-
withArtifact: EvalResult;
|
|
202
|
-
/** Eval of `baseline` alone. */
|
|
203
|
-
withoutArtifact: EvalResult;
|
|
204
|
-
/** `withArtifact.composite − withoutArtifact.composite`. */
|
|
205
|
-
scoreDelta: number;
|
|
206
|
-
/** `withArtifact.costUsd − withoutArtifact.costUsd`. */
|
|
207
|
-
costDelta: number;
|
|
208
|
-
}
|
|
209
|
-
/**
|
|
210
|
-
* Run the with/without ablation for `candidate` over `baseline` and return its
|
|
211
|
-
* marginal score/cost contribution.
|
|
212
|
-
*
|
|
213
|
-
* The "without" arm scores the baseline profile unchanged; the "with" arm scores
|
|
214
|
-
* `applyArtifact(baseline, candidate)`. Both use the same `evalRunner`, so the
|
|
215
|
-
* delta isolates the artifact's effect (eval method held constant). The baseline
|
|
216
|
-
* arm is skipped when `baselineResult` is supplied.
|
|
217
|
-
*
|
|
218
|
-
* @example
|
|
219
|
-
* const lift = await measureMarginalLift({
|
|
220
|
-
* baseline,
|
|
221
|
-
* candidate: registry.get(id)!,
|
|
222
|
-
* evalRunner: (profile) => scoreProfileOnCohort(profile),
|
|
223
|
-
* })
|
|
224
|
-
* if (lift.scoreDelta > 0) registry.promote(lift.artifactId)
|
|
225
|
-
*/
|
|
226
|
-
declare function measureMarginalLift(opts: MeasureMarginalLiftOptions): Promise<MarginalLift>;
|
|
227
|
-
|
|
228
|
-
/**
|
|
229
|
-
* `PromotionGate` — the held-back exam that decides whether a measured candidate
|
|
230
|
-
* artifact is promoted into the registry.
|
|
231
|
-
*
|
|
232
|
-
* The lifecycle measures each candidate's marginal lift (`measureMarginalLift`),
|
|
233
|
-
* but a positive ablation delta on the practice problems is NOT enough to ship —
|
|
234
|
-
* that is how you overfit. The gate is the SECOND, stricter test: it decides
|
|
235
|
-
* promotion from a held-back split (fresh problems the candidate never tuned on)
|
|
236
|
-
* with a significance bar, so a promoted artifact earned its place rather than
|
|
237
|
-
* memorized the practice set.
|
|
238
|
-
*
|
|
239
|
-
* The interface is small and pluggable on purpose: production wires
|
|
240
|
-
* `heldOutPromotionGate`, which delegates to agent-eval's paper-grade
|
|
241
|
-
* `HeldOutGate` (paired-bootstrap CI on the held-out delta + an overfit-gap
|
|
242
|
-
* check). A deterministic test can inject a stub gate. The orchestrator never
|
|
243
|
-
* imports `HeldOutGate` directly — it only knows this interface — so the gate
|
|
244
|
-
* policy is a configuration choice, not a hardcode.
|
|
245
|
-
*/
|
|
246
|
-
|
|
247
|
-
/** The verdict a gate returns for one candidate. */
|
|
248
|
-
interface PromotionVerdict {
|
|
249
|
-
/** Whether to promote the candidate into the registry as `active`. */
|
|
250
|
-
promote: boolean;
|
|
251
|
-
/** Human-readable reason (surfaced in provenance + reports). */
|
|
252
|
-
reason: string;
|
|
253
|
-
/** Machine-readable rejection code, or `null` on promote. Mirrors the
|
|
254
|
-
* `HeldOutGate` rejection taxonomy when that gate is the backend. */
|
|
255
|
-
rejectionCode: string | null;
|
|
256
|
-
}
|
|
257
|
-
/**
|
|
258
|
-
* Decides whether ONE measured candidate is promoted. The lifecycle calls this
|
|
259
|
-
* once per candidate, after `measureMarginalLift` has produced the ablation.
|
|
260
|
-
*
|
|
261
|
-
* `lift` carries the with/without ablation (the marginal contribution); the gate
|
|
262
|
-
* MAY use it directly (the simple "positive lift on the held-back split" policy)
|
|
263
|
-
* OR ignore the scalar and decide from the paired per-task records the eval
|
|
264
|
-
* produced (`lift.withArtifact.runs` / `lift.withoutArtifact.runs`) when a
|
|
265
|
-
* significance gate like `HeldOutGate` is the backend.
|
|
266
|
-
*/
|
|
267
|
-
interface PromotionGate {
|
|
268
|
-
/** Stable label for the gate policy (provenance). */
|
|
269
|
-
kind: string;
|
|
270
|
-
decide(lift: MarginalLift): PromotionVerdict;
|
|
271
|
-
}
|
|
272
|
-
/**
|
|
273
|
-
* The simplest honest gate: promote iff the candidate's marginal lift on the
|
|
274
|
-
* held-back split clears `minDelta` (default `> 0`). It reads only the scalar
|
|
275
|
-
* `scoreDelta`, so it works with any `EvalRunner` (no per-task records needed).
|
|
276
|
-
*
|
|
277
|
-
* Use this when the eval is already scored ON the held-back split and you want a
|
|
278
|
-
* threshold, not a significance test — e.g. a deterministic fixture domain, or a
|
|
279
|
-
* cheap first pass before a paired-bootstrap gate. For paper-grade promotion on
|
|
280
|
-
* noisy live evals, prefer `heldOutPromotionGate`.
|
|
281
|
-
*/
|
|
282
|
-
declare function thresholdPromotionGate(minDelta?: number): PromotionGate;
|
|
283
|
-
interface HeldOutPromotionGateOptions {
|
|
284
|
-
/** Stable label of the baseline candidate the held-out records pair against. */
|
|
285
|
-
baselineKey: string;
|
|
286
|
-
/** Minimum paired (candidate, baseline) holdout observations. Default 3. */
|
|
287
|
-
minProductiveRuns?: number;
|
|
288
|
-
/** Lower-bound on the bootstrap-CI of the median paired holdout delta. Default 0. */
|
|
289
|
-
pairedDeltaThreshold?: number;
|
|
290
|
-
/** Max allowed worsening of the (search − holdout) overfit gap. Default 0.15. */
|
|
291
|
-
overfitGapThreshold?: number;
|
|
292
|
-
/** Deterministic bootstrap seed (reproducible CIs). Default unseeded. */
|
|
293
|
-
seed?: number;
|
|
294
|
-
/** Hard ceiling on the candidate's median per-task USD cost. Default none. */
|
|
295
|
-
costPerTaskCeiling?: number;
|
|
296
|
-
}
|
|
297
|
-
/**
|
|
298
|
-
* The paper-grade promotion gate: delegate to agent-eval's `HeldOutGate`, which
|
|
299
|
-
* pairs the candidate and baseline per-task holdout records by (experimentId,
|
|
300
|
-
* seed), runs a paired-bootstrap CI on the median delta, and checks the
|
|
301
|
-
* overfit gap. Promotes only when the held-out generalization is real, not luck.
|
|
302
|
-
*
|
|
303
|
-
* REQUIRES the eval to surface per-task records: `lift.withArtifact.runs` (the
|
|
304
|
-
* candidate arm) and `lift.withoutArtifact.runs` (the baseline arm), each
|
|
305
|
-
* carrying matched seeds with both `search` and `holdout` split scores. When the
|
|
306
|
-
* records are absent, this gate FAILS LOUD — a held-out significance claim with
|
|
307
|
-
* no per-task data behind it would be a fabricated number, which the
|
|
308
|
-
* no-silent-fallback doctrine forbids. Use `thresholdPromotionGate` for evals
|
|
309
|
-
* that only produce a scalar composite.
|
|
310
|
-
*/
|
|
311
|
-
declare function heldOutPromotionGate(opts: HeldOutPromotionGateOptions): PromotionGate;
|
|
312
|
-
|
|
313
|
-
/**
|
|
314
|
-
* `CandidateGenerator` — the ONE per-surface seam of the artifact lifecycle.
|
|
315
|
-
*
|
|
316
|
-
* Everything else in the lifecycle (measure → gate → store → compose) is
|
|
317
|
-
* surface-agnostic: it operates on `ProfileArtifact`s regardless of whether they
|
|
318
|
-
* are skills, tools, prompts, or MCP servers. The ONLY thing that varies by
|
|
319
|
-
* surface is HOW a fresh candidate piece is produced from the agent's history —
|
|
320
|
-
* a skill is DISTILLED from traces then refined; a tool grant is proposed from a
|
|
321
|
-
* "the agent lacked an action" finding; a prompt line is drafted from a recurring
|
|
322
|
-
* mistake. That per-surface logic is isolated here, behind one interface, so the
|
|
323
|
-
* orchestrator never grows a `switch (surface)`.
|
|
324
|
-
*
|
|
325
|
-
* A generator is pure-ish: given the lifecycle context (the baseline profile, the
|
|
326
|
-
* captured traces/findings, the registry of what already exists), it returns zero
|
|
327
|
-
* or more `ArtifactInput`s — candidate pieces that have NOT been measured yet.
|
|
328
|
-
* The orchestrator owns registration, measurement, gating, and storage; the
|
|
329
|
-
* generator owns only "what new pieces could this surface contribute".
|
|
330
|
-
*
|
|
331
|
-
* This is the interface the per-surface stages (skills, tools, prompt, mcp)
|
|
332
|
-
* implement. The skill generator (`skillGenerator`) is the reference
|
|
333
|
-
* implementation and the literal answer to "an empty profile has no skills":
|
|
334
|
-
* its `distill` step CREATES a skill from traces (the step `skillOpt` cannot do),
|
|
335
|
-
* then `refine` optimizes it.
|
|
336
|
-
*/
|
|
337
|
-
|
|
338
|
-
/**
|
|
339
|
-
* The read-only context a generator sees when proposing candidates. It is the
|
|
340
|
-
* agent's HISTORY (what to learn from) plus the agent's CURRENT shape (what
|
|
341
|
-
* already exists, so the generator does not re-propose a duplicate).
|
|
342
|
-
*/
|
|
343
|
-
interface GenerateContext {
|
|
344
|
-
/** The baseline profile candidates are proposed on top of. A generator reads
|
|
345
|
-
* it to avoid re-proposing something the profile already has. */
|
|
346
|
-
baseline: AgentProfile;
|
|
347
|
-
/** The domain/agent id the lifecycle is running for — namespaces provenance
|
|
348
|
-
* and lets a generator scope its proposals (e.g. distill from this domain's
|
|
349
|
-
* traces only). */
|
|
350
|
-
domain: string;
|
|
351
|
-
/** Trace-analyst findings to ground proposals in observed behavior. The
|
|
352
|
-
* firewall holds: these are OBSERVED signals, never judge verdicts. */
|
|
353
|
-
findings: ReadonlyArray<AnalystFinding>;
|
|
354
|
-
/** Raw captured trace text/records to distill from, opaque to the
|
|
355
|
-
* orchestrator. A skill generator's `distill` reads this; a tool generator
|
|
356
|
-
* may ignore it and read `findings` instead. */
|
|
357
|
-
traces?: unknown;
|
|
358
|
-
/** Cooperative cancellation, forwarded from `runLifecycle`. */
|
|
359
|
-
signal?: AbortSignal;
|
|
360
|
-
}
|
|
361
|
-
/**
|
|
362
|
-
* Produces fresh, UNMEASURED candidate artifacts for ONE profile surface.
|
|
363
|
-
*
|
|
364
|
-
* `kind` is the `ArtifactKind` this generator targets (so the orchestrator can
|
|
365
|
-
* report and group by surface). `generate` returns candidate inputs the
|
|
366
|
-
* orchestrator will register, measure (`measureMarginalLift`), gate
|
|
367
|
-
* (`HeldOutGate`), and — on a pass — promote into the registry. Returning `[]`
|
|
368
|
-
* is valid: the surface had nothing to contribute this round.
|
|
369
|
-
*/
|
|
370
|
-
interface CandidateGenerator<K extends ArtifactKind = ArtifactKind> {
|
|
371
|
-
/** The profile surface this generator targets. */
|
|
372
|
-
kind: K;
|
|
373
|
-
/**
|
|
374
|
-
* Propose candidate artifacts from the lifecycle context. MUST NOT measure,
|
|
375
|
-
* gate, or register — that is the orchestrator's job. MUST NOT mutate
|
|
376
|
-
* `ctx.baseline`. Returns unmeasured `ArtifactInput`s (no lift score yet); the
|
|
377
|
-
* orchestrator stamps provenance + the measured lift before promotion.
|
|
378
|
-
*/
|
|
379
|
-
generate(ctx: GenerateContext): Promise<ArtifactInput<K>[]>;
|
|
380
|
-
}
|
|
381
|
-
|
|
382
|
-
export { type ArtifactKind as A, type CandidateGenerator as C, type EvalRunner as E, type GenerateContext as G, type HeldOutPromotionGateOptions as H, type MarginalLift as M, type PromotionGate as P, type ProfileArtifact as a, type ArtifactStatus as b, type ArtifactInput as c, type EvalResult as d, type PromotionVerdict as e, type ArtifactPayloads as f, type MeasureMarginalLiftOptions as g, heldOutPromotionGate as h, measureMarginalLift as m, thresholdPromotionGate as t };
|
|
@@ -1,172 +0,0 @@
|
|
|
1
|
-
import { WorktreeAdapter } from '@tangle-network/agent-eval/campaign';
|
|
2
|
-
import { Scenario, SelfImproveOptions, SurfaceProposer, SelfImproveResult, MutableSurface } from '@tangle-network/agent-eval/contract';
|
|
3
|
-
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
4
|
-
import { L as LocalHarness } from './local-harness-ZqCx51u7.js';
|
|
5
|
-
import { V as Verifier, C as CandidateGenerator } from './agentic-generator-hCaQRAes.js';
|
|
6
|
-
|
|
7
|
-
/**
|
|
8
|
-
*
|
|
9
|
-
* `improve` — the ONE public, surface-pluggable RSI verb.
|
|
10
|
-
*
|
|
11
|
-
* A thin facade over agent-eval's `selfImprove` (the held-out-gated closed
|
|
12
|
-
* loop). It removes the two things a caller otherwise has to know to drive the
|
|
13
|
-
* loop by hand: WHICH `MutableSurface` of the profile is being optimized, and
|
|
14
|
-
* WHICH `SurfaceProposer` mutates that surface. You name a `surface`; the
|
|
15
|
-
* facade picks the matching default proposer, extracts the baseline surface from
|
|
16
|
-
* the profile, runs `selfImprove`, and (on a ship verdict) writes the promoted
|
|
17
|
-
* winner back into the corresponding profile field.
|
|
18
|
-
*
|
|
19
|
-
* - `surface: 'prompt'` → `gepaProposer` mutates `profile.prompt.systemPrompt`.
|
|
20
|
-
* - `surface: 'skills'` → `skillOptProposer` mutates a skills document string.
|
|
21
|
-
* - `surface: 'memory'` → `memoryCurationProposer` curates a bounded durable
|
|
22
|
-
* lesson document supplied through `opts.memory`.
|
|
23
|
-
* - `surface: 'agent-profile'` → caller-supplied proposer mutates the complete
|
|
24
|
-
* canonical AgentProfile JSON in one candidate.
|
|
25
|
-
* - `surface` ∈ {`tools`, `mcp`, `hooks`, `subagents`, `agent-profile`} → no zero-config default
|
|
26
|
-
* proposer exists (a code/config proposer needs caller-supplied wiring — a
|
|
27
|
-
* worktree repo root, a candidate generator, a serializer). The facade
|
|
28
|
-
* requires an explicit `opts.generator` for these and throws a `ConfigError`
|
|
29
|
-
* otherwise. This is a designed boundary, not a missing default: there is
|
|
30
|
-
* no safe value the facade could invent for those surfaces. Code instead
|
|
31
|
-
* requires `opts.code.repoRoot` and accepts only the runtime-owned
|
|
32
|
-
* `opts.code.generator` path so every isolated checkout can be released.
|
|
33
|
-
*
|
|
34
|
-
* Everything else (`scenarios`, `judge`, `agent`, `budget`, `llm`) passes
|
|
35
|
-
* straight through to `selfImprove`.
|
|
36
|
-
*
|
|
37
|
-
* @experimental
|
|
38
|
-
*/
|
|
39
|
-
|
|
40
|
-
/** The executable agent lever `improve` optimizes. Profile fields remain
|
|
41
|
-
* portable AgentProfile coordinates; implementation and orchestration files
|
|
42
|
-
* use the code surface so a winner can be sealed into an exact candidate. */
|
|
43
|
-
type ImproveSurface = 'prompt' | 'skills' | 'tools' | 'mcp' | 'hooks' | 'subagents' | 'agent-profile' | 'memory' | 'code';
|
|
44
|
-
type ImproveOptions<TScenario extends Scenario, TArtifact> = Omit<SelfImproveOptions<TScenario, TArtifact>, 'analyzeGeneration' | 'baselineSurface' | 'findings' | 'gate' | 'proposer'> & {
|
|
45
|
-
/** Which profile lever to optimize. Default `'prompt'`. Selects the default
|
|
46
|
-
* generator + the baseline-surface extraction shape. */
|
|
47
|
-
surface?: ImproveSurface;
|
|
48
|
-
/** The `SurfaceProposer` that mutates a profile surface. When unset, the facade
|
|
49
|
-
* picks the default for prompt, skills, and memory; surfaces
|
|
50
|
-
* with no default REQUIRE this (fail-loud otherwise). Forbidden for code;
|
|
51
|
-
* use `code.generator` so the runtime owns candidate cleanup. */
|
|
52
|
-
generator?: SurfaceProposer;
|
|
53
|
-
/** Gate mode. `'holdout'` (default) runs the held-out promotion gate;
|
|
54
|
-
* `'none'` is a baseline-only run (`budget.generations = 0`). */
|
|
55
|
-
gate?: 'holdout' | 'none';
|
|
56
|
-
/** Restrict the run to this subset of models. When set, the reflection model
|
|
57
|
-
* (`llm.model`, or the default when unset) must be a member, or `improve()` throws
|
|
58
|
-
* a `ConfigError` before the generator is built. Unset = unrestricted. */
|
|
59
|
-
allowedModels?: readonly string[];
|
|
60
|
-
/** Per-generation findings producer passthrough (see selfImprove.analyzeGeneration).
|
|
61
|
-
* DEFAULT: the built-in failure distiller — after each generation it turns the
|
|
62
|
-
* worst-scoring/errored cells into structured findings ({ scenario, composite,
|
|
63
|
-
* notes, error }) for the NEXT proposal round, so the proposer reasons over what
|
|
64
|
-
* actually failed instead of a static seed. Pass your own producer (e.g. a
|
|
65
|
-
* trace-analyst over the runDir's traces) to replace it; pass `null` to disable
|
|
66
|
-
* and keep the static `findings` all the way through. */
|
|
67
|
-
analyzeGeneration?: SelfImproveOptions<TScenario, TArtifact>['analyzeGeneration'] | null;
|
|
68
|
-
/** META-HARNESS mode: instead of the ~1500-char distilled findings, feed the
|
|
69
|
-
* proposer RAW-TRACE FILESYSTEM CONTEXT — the PATHS into the prior generation's
|
|
70
|
-
* real run traces under `runDir` (per-cell `spans.jsonl` event logs +
|
|
71
|
-
* `cached-result.json` scores + artifacts) plus a `grep`/`cat`-to-diagnose
|
|
72
|
-
* instruction — so the coding agent reads the actual failures itself rather than
|
|
73
|
-
* a pre-summary. Requires a REAL `runDir` (that is where the traces live).
|
|
74
|
-
* Ignored when `analyzeGeneration` is set explicitly (that wins) or is `null`
|
|
75
|
-
* (disabled). Equivalent to `analyzeGeneration: rawTraceDistiller()`; this flag
|
|
76
|
-
* is the one-line enable. Default `false` (the distiller stays the default). */
|
|
77
|
-
rawTraceContext?: boolean;
|
|
78
|
-
/** CODE-surface wiring: name `surface: 'code'`, point at a repo, and the
|
|
79
|
-
* facade assembles the whole candidate pipeline — an isolated incumbent plus git worktrees
|
|
80
|
-
* (`gitWorktreeAdapter`) driven by `improvementDriver` with the full agentic
|
|
81
|
-
* generator (a real coding harness edits each candidate worktree; a `verify`
|
|
82
|
-
* hook gates candidates before they are ever measured). Ignored when
|
|
83
|
-
* `opts.generator` is supplied. Required for every code run because a real
|
|
84
|
-
* repository and base ref are necessary to measure the incumbent. */
|
|
85
|
-
code?: ImproveCodeOptions;
|
|
86
|
-
/** SKILLS-surface wiring for real skill-DOCUMENT optimization. Without this,
|
|
87
|
-
* `surface: 'skills'` optimizes the profile's skills REFS array (file pointers)
|
|
88
|
-
* — which `skillOptProposer` (a document patcher) cannot meaningfully edit.
|
|
89
|
-
* Provide the document CONTENT to optimize + a `writeBack` to persist the
|
|
90
|
-
* shipped winner (the profile ref points at a file the caller owns). This is
|
|
91
|
-
* what makes skillOpt reachable through improve(). */
|
|
92
|
-
skills?: ImproveSkillsOptions;
|
|
93
|
-
/** MEMORY-surface wiring for a curated durable memory document. The default
|
|
94
|
-
* deterministic proposer deduplicates and ranks lessons from findings, then
|
|
95
|
-
* replaces its managed block instead of growing memory without bound. */
|
|
96
|
-
memory?: ImproveMemoryOptions;
|
|
97
|
-
/** Custom held-back-exam decision. The string `gate` above controls whether
|
|
98
|
-
* the exam runs; this callback controls how its evidence decides promotion. */
|
|
99
|
-
promotionGate?: SelfImproveOptions<TScenario, TArtifact>['gate'];
|
|
100
|
-
};
|
|
101
|
-
interface ImproveSkillsOptions {
|
|
102
|
-
/** The skill document's current text — the baseline `skillOptProposer` patches. */
|
|
103
|
-
document: string;
|
|
104
|
-
/** Persist the shipped winner document (write the file the profile ref points at).
|
|
105
|
-
* Called only on a ship verdict. When omitted, the winner is still returned in
|
|
106
|
-
* `result.raw.winner.surface` for the caller to materialize. */
|
|
107
|
-
writeBack?: (winnerDocument: string) => void | Promise<void>;
|
|
108
|
-
}
|
|
109
|
-
interface ImproveMemoryOptions {
|
|
110
|
-
/** Current durable memory text used as the measured baseline. */
|
|
111
|
-
document: string;
|
|
112
|
-
/** Persist the promoted memory document. Never called on hold or error. */
|
|
113
|
-
writeBack?: (winnerDocument: string) => void | Promise<void>;
|
|
114
|
-
}
|
|
115
|
-
interface ImproveCodeOptions {
|
|
116
|
-
/** Repo root candidate worktrees fork from. */
|
|
117
|
-
repoRoot: string;
|
|
118
|
-
/** Base ref candidates fork from. Default `main`. */
|
|
119
|
-
baseRef?: string;
|
|
120
|
-
/** Directory worktrees are created under. Default `<repoRoot>/.worktrees`. */
|
|
121
|
-
worktreeDir?: string;
|
|
122
|
-
/** Git-compatible adapter override, primarily for tests. Candidate advancement
|
|
123
|
-
* still requires normal Git worktree and commit semantics. */
|
|
124
|
-
worktree?: WorktreeAdapter;
|
|
125
|
-
/** Coding harness the agentic generator runs in each worktree. Default `claude`. */
|
|
126
|
-
harness?: LocalHarness;
|
|
127
|
-
/** Verify a candidate worktree before it becomes a measurable surface; failures
|
|
128
|
-
* feed the next shot (see `agenticGenerator.verify` / `commandVerifier`). */
|
|
129
|
-
verify?: Verifier;
|
|
130
|
-
/** Per-shot wall-clock timeout for the harness (ms). */
|
|
131
|
-
timeoutMs?: number;
|
|
132
|
-
/** Byte-producer override — the test seam and the escape hatch for custom
|
|
133
|
-
* candidate production. When set, `harness`/`verify`/`timeoutMs` are unused. */
|
|
134
|
-
generator?: CandidateGenerator;
|
|
135
|
-
}
|
|
136
|
-
interface ImproveResult<TScenario extends Scenario, TArtifact> {
|
|
137
|
-
/** The profile after improvement: the winner surface applied back into the
|
|
138
|
-
* matching field when the gate shipped, else the input profile unchanged. */
|
|
139
|
-
profile: AgentProfile;
|
|
140
|
-
/** True when `gateDecision === 'ship'`. */
|
|
141
|
-
shipped: boolean;
|
|
142
|
-
/** Held-out lift (`winner − baseline` composite). */
|
|
143
|
-
lift: number;
|
|
144
|
-
/** The five-valued gate verdict from `selfImprove`. */
|
|
145
|
-
gateDecision: SelfImproveResult<TScenario, TArtifact>['gateDecision'];
|
|
146
|
-
/** Full `selfImprove` result for advanced inspection. For code runs,
|
|
147
|
-
* `raw.winner.surface.worktreeRef` remains live after return whether the
|
|
148
|
-
* candidate shipped or held; call `dispose()` after consuming it. */
|
|
149
|
-
raw: SelfImproveResult<TScenario, TArtifact>;
|
|
150
|
-
/** Release resources owned by this result. Idempotent; currently disposes
|
|
151
|
-
* the returned code worktree and is a no-op for profile-only surfaces. */
|
|
152
|
-
dispose(): Promise<void>;
|
|
153
|
-
}
|
|
154
|
-
/** Apply a promoted winner surface back into the profile field for `surface`.
|
|
155
|
-
* Returns a shallow copy; never mutates the input profile. */
|
|
156
|
-
declare function applyImprovementWinnerToProfile(profile: AgentProfile, surface: ImproveSurface, winner: MutableSurface): AgentProfile;
|
|
157
|
-
/**
|
|
158
|
-
* Run the held-out-gated self-improvement loop on ONE profile surface.
|
|
159
|
-
*
|
|
160
|
-
* @example Optimize the system prompt, default holdout gate:
|
|
161
|
-
*
|
|
162
|
-
* const out = await improve(profile, findings, {
|
|
163
|
-
* surface: 'prompt',
|
|
164
|
-
* scenarios,
|
|
165
|
-
* judge,
|
|
166
|
-
* agent: (surface, scenario, ctx) => runAgent(surface, scenario, ctx.signal),
|
|
167
|
-
* })
|
|
168
|
-
* if (out.shipped) deploy(out.profile)
|
|
169
|
-
*/
|
|
170
|
-
declare function improve<TScenario extends Scenario, TArtifact>(profile: AgentProfile, findings: unknown[], opts: ImproveOptions<TScenario, TArtifact>): Promise<ImproveResult<TScenario, TArtifact>>;
|
|
171
|
-
|
|
172
|
-
export { type ImproveOptions as I, type ImproveResult as a, type ImproveCodeOptions as b, type ImproveMemoryOptions as c, type ImproveSkillsOptions as d, type ImproveSurface as e, applyImprovementWinnerToProfile as f, improve as i };
|