@tangle-network/agent-runtime 0.95.0 → 0.96.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/README.md +64 -15
  2. package/dist/activation-B0ZD7nfX.d.ts +63 -0
  3. package/dist/agent.d.ts +5 -169
  4. package/dist/agent.js +8 -229
  5. package/dist/agent.js.map +1 -1
  6. package/dist/analyst-loop.d.ts +6 -9
  7. package/dist/analyst-loop.js +1 -2
  8. package/dist/candidate-execution/index.js +6 -7
  9. package/dist/chunk-3XKSBI2U.js +474 -0
  10. package/dist/chunk-3XKSBI2U.js.map +1 -0
  11. package/dist/{chunk-6YBA64Z2.js → chunk-6XKPVJAZ.js} +5 -18
  12. package/dist/chunk-6XKPVJAZ.js.map +1 -0
  13. package/dist/{chunk-MKGRLDWB.js → chunk-BLQIYRVR.js} +17 -2
  14. package/dist/chunk-BLQIYRVR.js.map +1 -0
  15. package/dist/{chunk-QDSOD7RC.js → chunk-FD2MBMOH.js} +13 -101
  16. package/dist/chunk-FD2MBMOH.js.map +1 -0
  17. package/dist/{chunk-YLUOTX6U.js → chunk-FXF2OL34.js} +7 -7
  18. package/dist/{chunk-EP6RVHMX.js → chunk-HZDEXTSL.js} +848 -2
  19. package/dist/chunk-HZDEXTSL.js.map +1 -0
  20. package/dist/chunk-PSOCBNM3.js +2069 -0
  21. package/dist/chunk-PSOCBNM3.js.map +1 -0
  22. package/dist/{chunk-BPGXIKK7.js → chunk-SGQ4YIQW.js} +4 -4
  23. package/dist/{chunk-IADLKE7I.js → chunk-UQ6PNNXM.js} +5 -7
  24. package/dist/{chunk-IADLKE7I.js.map → chunk-UQ6PNNXM.js.map} +1 -1
  25. package/dist/{chunk-Z5I642SY.js → chunk-WYC2XJF2.js} +2 -2
  26. package/dist/{chunk-ZEYAT33L.js → chunk-Y3SRWZMP.js} +2 -2
  27. package/dist/{chunk-WTZ37EQY.js → chunk-YOLKCWRV.js} +197 -90
  28. package/dist/chunk-YOLKCWRV.js.map +1 -0
  29. package/dist/conversation.js +0 -1
  30. package/dist/environment-provider.js +0 -1
  31. package/dist/{agentic-generator-hCaQRAes.d.ts → improve-g75IE2Cx.d.ts} +152 -3
  32. package/dist/{improvement-adapter-BieWeK5J.d.ts → improvement-adapter-HAZz-7vK.d.ts} +8 -31
  33. package/dist/index.d.ts +41 -11
  34. package/dist/index.js +180 -51
  35. package/dist/index.js.map +1 -1
  36. package/dist/intelligence.d.ts +16 -9
  37. package/dist/intelligence.js +13 -8
  38. package/dist/intelligence.js.map +1 -1
  39. package/dist/knowledge.d.ts +30 -13
  40. package/dist/knowledge.js +11 -10
  41. package/dist/{loop-runner-bin-BIQldFS8.d.ts → loop-runner-bin-Cn1N2rRo.d.ts} +1 -1
  42. package/dist/loop-runner-bin.d.ts +2 -2
  43. package/dist/loop-runner-bin.js +6 -8
  44. package/dist/loops.d.ts +1 -1
  45. package/dist/loops.js +4 -6
  46. package/dist/mcp/bin.js +3 -5
  47. package/dist/mcp/bin.js.map +1 -1
  48. package/dist/mcp/index.js +10 -12
  49. package/dist/mcp/index.js.map +1 -1
  50. package/dist/platform.js +0 -2
  51. package/dist/platform.js.map +1 -1
  52. package/dist/primeintellect/index.js +0 -1
  53. package/dist/primeintellect/index.js.map +1 -1
  54. package/dist/profiles.js +0 -1
  55. package/dist/profiles.js.map +1 -1
  56. package/dist/{types-BC3bZpH0.d.ts → types-CmYCMbFT.d.ts} +12 -54
  57. package/package.json +7 -12
  58. package/skills/build-with-agent-runtime/SKILL.md +122 -213
  59. package/dist/chunk-6O73TRHW.js +0 -142
  60. package/dist/chunk-6O73TRHW.js.map +0 -1
  61. package/dist/chunk-6YBA64Z2.js.map +0 -1
  62. package/dist/chunk-AP7CPGMZ.js +0 -334
  63. package/dist/chunk-AP7CPGMZ.js.map +0 -1
  64. package/dist/chunk-DGUM43GV.js +0 -11
  65. package/dist/chunk-DGUM43GV.js.map +0 -1
  66. package/dist/chunk-DHCHL6OG.js +0 -625
  67. package/dist/chunk-DHCHL6OG.js.map +0 -1
  68. package/dist/chunk-EP6RVHMX.js.map +0 -1
  69. package/dist/chunk-G55QE4IQ.js +0 -1137
  70. package/dist/chunk-G55QE4IQ.js.map +0 -1
  71. package/dist/chunk-ISTDY47H.js +0 -849
  72. package/dist/chunk-ISTDY47H.js.map +0 -1
  73. package/dist/chunk-MKGRLDWB.js.map +0 -1
  74. package/dist/chunk-QDSOD7RC.js.map +0 -1
  75. package/dist/chunk-WTZ37EQY.js.map +0 -1
  76. package/dist/generator-YkAQrOoD.d.ts +0 -382
  77. package/dist/improve-B-UYaEH5.d.ts +0 -172
  78. package/dist/lifecycle.d.ts +0 -870
  79. package/dist/lifecycle.js +0 -981
  80. package/dist/lifecycle.js.map +0 -1
  81. package/dist/mcp-serve-verifier-Bs_n0xPc.d.ts +0 -34
  82. package/skills/agent-runtime-adoption/SKILL.md +0 -246
  83. /package/dist/{chunk-YLUOTX6U.js.map → chunk-FXF2OL34.js.map} +0 -0
  84. /package/dist/{chunk-BPGXIKK7.js.map → chunk-SGQ4YIQW.js.map} +0 -0
  85. /package/dist/{chunk-Z5I642SY.js.map → chunk-WYC2XJF2.js.map} +0 -0
  86. /package/dist/{chunk-ZEYAT33L.js.map → chunk-Y3SRWZMP.js.map} +0 -0
@@ -1,382 +0,0 @@
1
- import { RunRecord, AnalystFinding } from '@tangle-network/agent-eval';
2
- import { AgentProfileResourceRef, AgentProfileMcpServer, AgentProfileHookCommand, AgentSubagentProfile, AgentProfile } from '@tangle-network/agent-interface';
3
-
4
- /**
5
- * `@tangle-network/agent-runtime/lifecycle` — artifact-lifecycle FOUNDATION.
6
- *
7
- * The §1.5 law says an agent IS its `AgentProfile`, and the profile is the WHOLE
8
- * agent: prompt + skills + tools + mcp + hooks + subagents. This module names the
9
- * discrete, individually-promotable PIECES of that profile — "artifacts" — and
10
- * gives them stable ids so the rest of a self-improvement lifecycle (propose →
11
- * measure → promote → ship) has something concrete to hang off.
12
- *
13
- * This is PHASE 1: just the two primitives the rest hangs off.
14
- * 1. `ArtifactRegistry` — a typed catalog of profile artifacts with stable ids.
15
- * 2. `measureMarginalLift` — the with-vs-without ablation: how much score/cost a
16
- * single artifact adds on top of a baseline profile.
17
- *
18
- * The per-surface lifecycles, the `BuildableSurface` author contract, and the
19
- * promotion-gate wiring are deferred to later phases.
20
- */
21
-
22
- /**
23
- * The profile levers an artifact can target. One-to-one with the §1.5 profile
24
- * surface (`prompt + skills + tools + mcp + hooks + subagents`). Each kind maps to
25
- * exactly one field of `AgentProfile`, so an artifact can be applied onto a
26
- * baseline profile deterministically (see `applyArtifact`).
27
- */
28
- type ArtifactKind = 'skill' | 'tool' | 'mcp' | 'hook' | 'subagent' | 'prompt';
29
- /**
30
- * The payload for each `ArtifactKind`. The shapes are the SAME types the
31
- * `AgentProfile` field carries, so applying an artifact is a structural merge
32
- * onto the profile — never a bespoke per-kind transform.
33
- *
34
- * - `prompt` — an instruction line appended to `profile.prompt.instructions`.
35
- * - `skill` — a `SKILL.md`-style resource ref added to `profile.resources.skills`.
36
- * - `tool` — a tool grant: `{ enabled }` set under `profile.tools[name]`.
37
- * - `mcp` — one MCP server added under `profile.mcp[name]`.
38
- * - `hook` — one or more hook commands added under `profile.hooks[event]`.
39
- * - `subagent` — one subagent profile added under `profile.subagents[name]`.
40
- */
41
- interface ArtifactPayloads {
42
- prompt: {
43
- instruction: string;
44
- };
45
- skill: {
46
- resource: AgentProfileResourceRef;
47
- };
48
- tool: {
49
- enabled: boolean;
50
- };
51
- mcp: {
52
- server: AgentProfileMcpServer;
53
- };
54
- hook: {
55
- event: string;
56
- commands: AgentProfileHookCommand[];
57
- };
58
- subagent: {
59
- profile: AgentSubagentProfile;
60
- };
61
- }
62
- /**
63
- * A discrete, individually-promotable piece of an agent profile.
64
- *
65
- * `kind` selects the profile lever; `payload` is the kind-specific value; `key`
66
- * is the profile-field key the payload lands under (the tool name, the MCP server
67
- * name, the subagent name — unused for `prompt`, which appends). `id` is stable:
68
- * once registered, it never changes, so a marginal-lift measurement, a promotion
69
- * decision, and a ship record all reference the same artifact.
70
- */
71
- interface ProfileArtifact<K extends ArtifactKind = ArtifactKind> {
72
- /** Stable id. Assigned by the registry at register time; never reassigned. */
73
- id: string;
74
- kind: K;
75
- /**
76
- * The profile-field key this artifact lands under (e.g. the tool name, the MCP
77
- * server name, the subagent name, the hook event). Optional for `prompt`
78
- * (instructions append, they have no key). Defaults to `id` when applying a
79
- * keyed artifact without an explicit key.
80
- */
81
- key?: string;
82
- /** Human-facing label for review surfaces. */
83
- name: string;
84
- /** Optional one-line description of what this artifact does. */
85
- description?: string;
86
- payload: ArtifactPayloads[K];
87
- /**
88
- * Lifecycle status — the full artifact state machine:
89
- *
90
- * `candidate` → `active` → `decayed` (re-promotable)
91
- * ↘ `retired` (terminal)
92
- *
93
- * - `candidate` — registered, not yet promoted. The default at register time.
94
- * - `active` — passed the promotion gate and carries a measured held-back
95
- * lift; the only status `composeProfile` folds into a profile.
96
- * - `decayed` — was active, but a later re-measure (`driftWatch`) found its
97
- * lift fell below the keep-bar. Demoted out of the composed
98
- * profile; kept as an auditable record and a re-promotion
99
- * candidate if a future re-measure recovers the lift.
100
- * - `retired` — permanently removed from the active set (`dedupeArtifacts`
101
- * retires the weaker half of a non-stacking pair). Terminal.
102
- *
103
- * The registry never auto-promotes; every transition is an explicit call
104
- * (`promote`/`promoteWithLift`/`demote`/`retire`).
105
- */
106
- status: ArtifactStatus;
107
- /** Free-form metadata (provenance, generation id, the measured lift, …). */
108
- metadata?: Record<string, unknown>;
109
- }
110
- /**
111
- * The artifact lifecycle states. `active` is the load-bearing one — it is the
112
- * sole status `composeProfile` folds into a deployable profile, and it is gated
113
- * by a measured held-back lift (the registry invariant). `decayed` and `retired`
114
- * are the two ways an artifact LEAVES the active set: a decayed artifact lost its
115
- * lift on re-measure (reversible — `driftWatch`), a retired one was deduped away
116
- * (terminal — `dedupeArtifacts`).
117
- */
118
- type ArtifactStatus = 'candidate' | 'active' | 'decayed' | 'retired';
119
- /** The input to `register` — everything on `ProfileArtifact` except the
120
- * registry-owned `id` and `status`. An explicit `id` may be supplied for
121
- * deterministic/idempotent registration; otherwise the registry assigns one. */
122
- type ArtifactInput<K extends ArtifactKind = ArtifactKind> = Omit<ProfileArtifact<K>, 'id' | 'status'> & {
123
- id?: string;
124
- status?: ArtifactStatus;
125
- };
126
-
127
- /**
128
- * `measureMarginalLift` — the with-vs-without ablation for one artifact.
129
- *
130
- * "What does THIS one piece add?" is the question a lifecycle has to answer
131
- * before it can promote anything. The answer is an ablation: score the baseline
132
- * profile, score the baseline-plus-candidate profile, and report the delta. This
133
- * is the marginal contribution of the artifact — the same shape the project's
134
- * `OutcomeMeasurement` reports for an apply pass, but isolated to ONE artifact so
135
- * a registry can rank candidates by what they individually earn.
136
- *
137
- * The measurement is selector-agnostic: it takes an `EvalRunner` the caller
138
- * supplies (any function that scores a profile and reports cost), runs it twice,
139
- * and subtracts. It never judges, never picks a winner, never re-scores on a
140
- * holdout — those are the gate's job. It only quantifies the ablation so the gate
141
- * has a real number to decide on.
142
- */
143
-
144
- /**
145
- * The result of running an eval over ONE profile: a composite score and the cost
146
- * to obtain it. This mirrors the project's score/cost convention (`composite`
147
- * from `OutcomeMeasurement`, `costUsd` from `LoopResult`), so a caller can pass a
148
- * thin wrapper over `runLoop` / `runBenchmark` / `runAgentEval` directly.
149
- */
150
- interface EvalResult {
151
- /** Composite score in `[0, 1]` (higher is better) for the profile under test. */
152
- composite: number;
153
- /** USD cost to produce this result. */
154
- costUsd: number;
155
- /**
156
- * Per-task records the run produced, when the runner emits them. The marginal
157
- * lift only needs `composite`, but the held-out promotion gate (`HeldOutGate`)
158
- * pairs candidate vs baseline per-task holdout records by (experimentId, seed)
159
- * — so a runner feeding `heldOutPromotionGate` MUST populate this with rows
160
- * carrying both `search` and `holdout` split scores. Omit it for evals scored
161
- * to a single composite (then use `thresholdPromotionGate`).
162
- */
163
- runs?: RunRecord[];
164
- /** Optional opaque passthrough (per-task cells, the raw report, …). */
165
- details?: unknown;
166
- }
167
- /**
168
- * Scores a profile. The caller wires this to whatever eval they run — a
169
- * `runLoop` rollout, a `runBenchmark` campaign, a `runAgentEval` cohort — and
170
- * returns the composite + cost. `signal` is forwarded for cancellation.
171
- */
172
- type EvalRunner = (profile: AgentProfile, signal?: AbortSignal) => Promise<EvalResult>;
173
- interface MeasureMarginalLiftOptions {
174
- /** The profile the artifact is measured ON TOP OF (the "without" arm). */
175
- baseline: AgentProfile;
176
- /** The single artifact whose marginal contribution we want. */
177
- candidate: ProfileArtifact;
178
- /** The eval that scores a profile. Run once per arm (twice total). */
179
- evalRunner: EvalRunner;
180
- /**
181
- * A pre-computed baseline result, to skip the "without" run when the caller
182
- * already scored the baseline (e.g. measuring several candidates against the
183
- * same baseline). When set, the baseline arm is NOT re-run.
184
- */
185
- baselineResult?: EvalResult;
186
- /** Forwarded to both `evalRunner` invocations for cancellation. */
187
- signal?: AbortSignal;
188
- }
189
- /**
190
- * The marginal lift of one artifact: the with/without ablation.
191
- *
192
- * `scoreDelta = with.composite − without.composite`. A positive `scoreDelta` is
193
- * the evidence a gate needs to promote; a negative one is the signal to drop the
194
- * artifact. `costDelta` is the extra USD the artifact costs (often positive — a
195
- * new tool/MCP adds calls) and lets the gate weigh lift against spend.
196
- */
197
- interface MarginalLift {
198
- /** The artifact id this measurement is for (stable, from the registry). */
199
- artifactId: string;
200
- /** Eval of `applyArtifact(baseline, candidate)`. */
201
- withArtifact: EvalResult;
202
- /** Eval of `baseline` alone. */
203
- withoutArtifact: EvalResult;
204
- /** `withArtifact.composite − withoutArtifact.composite`. */
205
- scoreDelta: number;
206
- /** `withArtifact.costUsd − withoutArtifact.costUsd`. */
207
- costDelta: number;
208
- }
209
- /**
210
- * Run the with/without ablation for `candidate` over `baseline` and return its
211
- * marginal score/cost contribution.
212
- *
213
- * The "without" arm scores the baseline profile unchanged; the "with" arm scores
214
- * `applyArtifact(baseline, candidate)`. Both use the same `evalRunner`, so the
215
- * delta isolates the artifact's effect (eval method held constant). The baseline
216
- * arm is skipped when `baselineResult` is supplied.
217
- *
218
- * @example
219
- * const lift = await measureMarginalLift({
220
- * baseline,
221
- * candidate: registry.get(id)!,
222
- * evalRunner: (profile) => scoreProfileOnCohort(profile),
223
- * })
224
- * if (lift.scoreDelta > 0) registry.promote(lift.artifactId)
225
- */
226
- declare function measureMarginalLift(opts: MeasureMarginalLiftOptions): Promise<MarginalLift>;
227
-
228
- /**
229
- * `PromotionGate` — the held-back exam that decides whether a measured candidate
230
- * artifact is promoted into the registry.
231
- *
232
- * The lifecycle measures each candidate's marginal lift (`measureMarginalLift`),
233
- * but a positive ablation delta on the practice problems is NOT enough to ship —
234
- * that is how you overfit. The gate is the SECOND, stricter test: it decides
235
- * promotion from a held-back split (fresh problems the candidate never tuned on)
236
- * with a significance bar, so a promoted artifact earned its place rather than
237
- * memorized the practice set.
238
- *
239
- * The interface is small and pluggable on purpose: production wires
240
- * `heldOutPromotionGate`, which delegates to agent-eval's paper-grade
241
- * `HeldOutGate` (paired-bootstrap CI on the held-out delta + an overfit-gap
242
- * check). A deterministic test can inject a stub gate. The orchestrator never
243
- * imports `HeldOutGate` directly — it only knows this interface — so the gate
244
- * policy is a configuration choice, not a hardcode.
245
- */
246
-
247
- /** The verdict a gate returns for one candidate. */
248
- interface PromotionVerdict {
249
- /** Whether to promote the candidate into the registry as `active`. */
250
- promote: boolean;
251
- /** Human-readable reason (surfaced in provenance + reports). */
252
- reason: string;
253
- /** Machine-readable rejection code, or `null` on promote. Mirrors the
254
- * `HeldOutGate` rejection taxonomy when that gate is the backend. */
255
- rejectionCode: string | null;
256
- }
257
- /**
258
- * Decides whether ONE measured candidate is promoted. The lifecycle calls this
259
- * once per candidate, after `measureMarginalLift` has produced the ablation.
260
- *
261
- * `lift` carries the with/without ablation (the marginal contribution); the gate
262
- * MAY use it directly (the simple "positive lift on the held-back split" policy)
263
- * OR ignore the scalar and decide from the paired per-task records the eval
264
- * produced (`lift.withArtifact.runs` / `lift.withoutArtifact.runs`) when a
265
- * significance gate like `HeldOutGate` is the backend.
266
- */
267
- interface PromotionGate {
268
- /** Stable label for the gate policy (provenance). */
269
- kind: string;
270
- decide(lift: MarginalLift): PromotionVerdict;
271
- }
272
- /**
273
- * The simplest honest gate: promote iff the candidate's marginal lift on the
274
- * held-back split clears `minDelta` (default `> 0`). It reads only the scalar
275
- * `scoreDelta`, so it works with any `EvalRunner` (no per-task records needed).
276
- *
277
- * Use this when the eval is already scored ON the held-back split and you want a
278
- * threshold, not a significance test — e.g. a deterministic fixture domain, or a
279
- * cheap first pass before a paired-bootstrap gate. For paper-grade promotion on
280
- * noisy live evals, prefer `heldOutPromotionGate`.
281
- */
282
- declare function thresholdPromotionGate(minDelta?: number): PromotionGate;
283
- interface HeldOutPromotionGateOptions {
284
- /** Stable label of the baseline candidate the held-out records pair against. */
285
- baselineKey: string;
286
- /** Minimum paired (candidate, baseline) holdout observations. Default 3. */
287
- minProductiveRuns?: number;
288
- /** Lower-bound on the bootstrap-CI of the median paired holdout delta. Default 0. */
289
- pairedDeltaThreshold?: number;
290
- /** Max allowed worsening of the (search − holdout) overfit gap. Default 0.15. */
291
- overfitGapThreshold?: number;
292
- /** Deterministic bootstrap seed (reproducible CIs). Default unseeded. */
293
- seed?: number;
294
- /** Hard ceiling on the candidate's median per-task USD cost. Default none. */
295
- costPerTaskCeiling?: number;
296
- }
297
- /**
298
- * The paper-grade promotion gate: delegate to agent-eval's `HeldOutGate`, which
299
- * pairs the candidate and baseline per-task holdout records by (experimentId,
300
- * seed), runs a paired-bootstrap CI on the median delta, and checks the
301
- * overfit gap. Promotes only when the held-out generalization is real, not luck.
302
- *
303
- * REQUIRES the eval to surface per-task records: `lift.withArtifact.runs` (the
304
- * candidate arm) and `lift.withoutArtifact.runs` (the baseline arm), each
305
- * carrying matched seeds with both `search` and `holdout` split scores. When the
306
- * records are absent, this gate FAILS LOUD — a held-out significance claim with
307
- * no per-task data behind it would be a fabricated number, which the
308
- * no-silent-fallback doctrine forbids. Use `thresholdPromotionGate` for evals
309
- * that only produce a scalar composite.
310
- */
311
- declare function heldOutPromotionGate(opts: HeldOutPromotionGateOptions): PromotionGate;
312
-
313
- /**
314
- * `CandidateGenerator` — the ONE per-surface seam of the artifact lifecycle.
315
- *
316
- * Everything else in the lifecycle (measure → gate → store → compose) is
317
- * surface-agnostic: it operates on `ProfileArtifact`s regardless of whether they
318
- * are skills, tools, prompts, or MCP servers. The ONLY thing that varies by
319
- * surface is HOW a fresh candidate piece is produced from the agent's history —
320
- * a skill is DISTILLED from traces then refined; a tool grant is proposed from a
321
- * "the agent lacked an action" finding; a prompt line is drafted from a recurring
322
- * mistake. That per-surface logic is isolated here, behind one interface, so the
323
- * orchestrator never grows a `switch (surface)`.
324
- *
325
- * A generator is pure-ish: given the lifecycle context (the baseline profile, the
326
- * captured traces/findings, the registry of what already exists), it returns zero
327
- * or more `ArtifactInput`s — candidate pieces that have NOT been measured yet.
328
- * The orchestrator owns registration, measurement, gating, and storage; the
329
- * generator owns only "what new pieces could this surface contribute".
330
- *
331
- * This is the interface the per-surface stages (skills, tools, prompt, mcp)
332
- * implement. The skill generator (`skillGenerator`) is the reference
333
- * implementation and the literal answer to "an empty profile has no skills":
334
- * its `distill` step CREATES a skill from traces (the step `skillOpt` cannot do),
335
- * then `refine` optimizes it.
336
- */
337
-
338
- /**
339
- * The read-only context a generator sees when proposing candidates. It is the
340
- * agent's HISTORY (what to learn from) plus the agent's CURRENT shape (what
341
- * already exists, so the generator does not re-propose a duplicate).
342
- */
343
- interface GenerateContext {
344
- /** The baseline profile candidates are proposed on top of. A generator reads
345
- * it to avoid re-proposing something the profile already has. */
346
- baseline: AgentProfile;
347
- /** The domain/agent id the lifecycle is running for — namespaces provenance
348
- * and lets a generator scope its proposals (e.g. distill from this domain's
349
- * traces only). */
350
- domain: string;
351
- /** Trace-analyst findings to ground proposals in observed behavior. The
352
- * firewall holds: these are OBSERVED signals, never judge verdicts. */
353
- findings: ReadonlyArray<AnalystFinding>;
354
- /** Raw captured trace text/records to distill from, opaque to the
355
- * orchestrator. A skill generator's `distill` reads this; a tool generator
356
- * may ignore it and read `findings` instead. */
357
- traces?: unknown;
358
- /** Cooperative cancellation, forwarded from `runLifecycle`. */
359
- signal?: AbortSignal;
360
- }
361
- /**
362
- * Produces fresh, UNMEASURED candidate artifacts for ONE profile surface.
363
- *
364
- * `kind` is the `ArtifactKind` this generator targets (so the orchestrator can
365
- * report and group by surface). `generate` returns candidate inputs the
366
- * orchestrator will register, measure (`measureMarginalLift`), gate
367
- * (`HeldOutGate`), and — on a pass — promote into the registry. Returning `[]`
368
- * is valid: the surface had nothing to contribute this round.
369
- */
370
- interface CandidateGenerator<K extends ArtifactKind = ArtifactKind> {
371
- /** The profile surface this generator targets. */
372
- kind: K;
373
- /**
374
- * Propose candidate artifacts from the lifecycle context. MUST NOT measure,
375
- * gate, or register — that is the orchestrator's job. MUST NOT mutate
376
- * `ctx.baseline`. Returns unmeasured `ArtifactInput`s (no lift score yet); the
377
- * orchestrator stamps provenance + the measured lift before promotion.
378
- */
379
- generate(ctx: GenerateContext): Promise<ArtifactInput<K>[]>;
380
- }
381
-
382
- export { type ArtifactKind as A, type CandidateGenerator as C, type EvalRunner as E, type GenerateContext as G, type HeldOutPromotionGateOptions as H, type MarginalLift as M, type PromotionGate as P, type ProfileArtifact as a, type ArtifactStatus as b, type ArtifactInput as c, type EvalResult as d, type PromotionVerdict as e, type ArtifactPayloads as f, type MeasureMarginalLiftOptions as g, heldOutPromotionGate as h, measureMarginalLift as m, thresholdPromotionGate as t };
@@ -1,172 +0,0 @@
1
- import { WorktreeAdapter } from '@tangle-network/agent-eval/campaign';
2
- import { Scenario, SelfImproveOptions, SurfaceProposer, SelfImproveResult, MutableSurface } from '@tangle-network/agent-eval/contract';
3
- import { AgentProfile } from '@tangle-network/agent-interface';
4
- import { L as LocalHarness } from './local-harness-ZqCx51u7.js';
5
- import { V as Verifier, C as CandidateGenerator } from './agentic-generator-hCaQRAes.js';
6
-
7
- /**
8
- *
9
- * `improve` — the ONE public, surface-pluggable RSI verb.
10
- *
11
- * A thin facade over agent-eval's `selfImprove` (the held-out-gated closed
12
- * loop). It removes the two things a caller otherwise has to know to drive the
13
- * loop by hand: WHICH `MutableSurface` of the profile is being optimized, and
14
- * WHICH `SurfaceProposer` mutates that surface. You name a `surface`; the
15
- * facade picks the matching default proposer, extracts the baseline surface from
16
- * the profile, runs `selfImprove`, and (on a ship verdict) writes the promoted
17
- * winner back into the corresponding profile field.
18
- *
19
- * - `surface: 'prompt'` → `gepaProposer` mutates `profile.prompt.systemPrompt`.
20
- * - `surface: 'skills'` → `skillOptProposer` mutates a skills document string.
21
- * - `surface: 'memory'` → `memoryCurationProposer` curates a bounded durable
22
- * lesson document supplied through `opts.memory`.
23
- * - `surface: 'agent-profile'` → caller-supplied proposer mutates the complete
24
- * canonical AgentProfile JSON in one candidate.
25
- * - `surface` ∈ {`tools`, `mcp`, `hooks`, `subagents`, `agent-profile`} → no zero-config default
26
- * proposer exists (a code/config proposer needs caller-supplied wiring — a
27
- * worktree repo root, a candidate generator, a serializer). The facade
28
- * requires an explicit `opts.generator` for these and throws a `ConfigError`
29
- * otherwise. This is a designed boundary, not a missing default: there is
30
- * no safe value the facade could invent for those surfaces. Code instead
31
- * requires `opts.code.repoRoot` and accepts only the runtime-owned
32
- * `opts.code.generator` path so every isolated checkout can be released.
33
- *
34
- * Everything else (`scenarios`, `judge`, `agent`, `budget`, `llm`) passes
35
- * straight through to `selfImprove`.
36
- *
37
- * @experimental
38
- */
39
-
40
- /** The executable agent lever `improve` optimizes. Profile fields remain
41
- * portable AgentProfile coordinates; implementation and orchestration files
42
- * use the code surface so a winner can be sealed into an exact candidate. */
43
- type ImproveSurface = 'prompt' | 'skills' | 'tools' | 'mcp' | 'hooks' | 'subagents' | 'agent-profile' | 'memory' | 'code';
44
- type ImproveOptions<TScenario extends Scenario, TArtifact> = Omit<SelfImproveOptions<TScenario, TArtifact>, 'analyzeGeneration' | 'baselineSurface' | 'findings' | 'gate' | 'proposer'> & {
45
- /** Which profile lever to optimize. Default `'prompt'`. Selects the default
46
- * generator + the baseline-surface extraction shape. */
47
- surface?: ImproveSurface;
48
- /** The `SurfaceProposer` that mutates a profile surface. When unset, the facade
49
- * picks the default for prompt, skills, and memory; surfaces
50
- * with no default REQUIRE this (fail-loud otherwise). Forbidden for code;
51
- * use `code.generator` so the runtime owns candidate cleanup. */
52
- generator?: SurfaceProposer;
53
- /** Gate mode. `'holdout'` (default) runs the held-out promotion gate;
54
- * `'none'` is a baseline-only run (`budget.generations = 0`). */
55
- gate?: 'holdout' | 'none';
56
- /** Restrict the run to this subset of models. When set, the reflection model
57
- * (`llm.model`, or the default when unset) must be a member, or `improve()` throws
58
- * a `ConfigError` before the generator is built. Unset = unrestricted. */
59
- allowedModels?: readonly string[];
60
- /** Per-generation findings producer passthrough (see selfImprove.analyzeGeneration).
61
- * DEFAULT: the built-in failure distiller — after each generation it turns the
62
- * worst-scoring/errored cells into structured findings ({ scenario, composite,
63
- * notes, error }) for the NEXT proposal round, so the proposer reasons over what
64
- * actually failed instead of a static seed. Pass your own producer (e.g. a
65
- * trace-analyst over the runDir's traces) to replace it; pass `null` to disable
66
- * and keep the static `findings` all the way through. */
67
- analyzeGeneration?: SelfImproveOptions<TScenario, TArtifact>['analyzeGeneration'] | null;
68
- /** META-HARNESS mode: instead of the ~1500-char distilled findings, feed the
69
- * proposer RAW-TRACE FILESYSTEM CONTEXT — the PATHS into the prior generation's
70
- * real run traces under `runDir` (per-cell `spans.jsonl` event logs +
71
- * `cached-result.json` scores + artifacts) plus a `grep`/`cat`-to-diagnose
72
- * instruction — so the coding agent reads the actual failures itself rather than
73
- * a pre-summary. Requires a REAL `runDir` (that is where the traces live).
74
- * Ignored when `analyzeGeneration` is set explicitly (that wins) or is `null`
75
- * (disabled). Equivalent to `analyzeGeneration: rawTraceDistiller()`; this flag
76
- * is the one-line enable. Default `false` (the distiller stays the default). */
77
- rawTraceContext?: boolean;
78
- /** CODE-surface wiring: name `surface: 'code'`, point at a repo, and the
79
- * facade assembles the whole candidate pipeline — an isolated incumbent plus git worktrees
80
- * (`gitWorktreeAdapter`) driven by `improvementDriver` with the full agentic
81
- * generator (a real coding harness edits each candidate worktree; a `verify`
82
- * hook gates candidates before they are ever measured). Ignored when
83
- * `opts.generator` is supplied. Required for every code run because a real
84
- * repository and base ref are necessary to measure the incumbent. */
85
- code?: ImproveCodeOptions;
86
- /** SKILLS-surface wiring for real skill-DOCUMENT optimization. Without this,
87
- * `surface: 'skills'` optimizes the profile's skills REFS array (file pointers)
88
- * — which `skillOptProposer` (a document patcher) cannot meaningfully edit.
89
- * Provide the document CONTENT to optimize + a `writeBack` to persist the
90
- * shipped winner (the profile ref points at a file the caller owns). This is
91
- * what makes skillOpt reachable through improve(). */
92
- skills?: ImproveSkillsOptions;
93
- /** MEMORY-surface wiring for a curated durable memory document. The default
94
- * deterministic proposer deduplicates and ranks lessons from findings, then
95
- * replaces its managed block instead of growing memory without bound. */
96
- memory?: ImproveMemoryOptions;
97
- /** Custom held-back-exam decision. The string `gate` above controls whether
98
- * the exam runs; this callback controls how its evidence decides promotion. */
99
- promotionGate?: SelfImproveOptions<TScenario, TArtifact>['gate'];
100
- };
101
- interface ImproveSkillsOptions {
102
- /** The skill document's current text — the baseline `skillOptProposer` patches. */
103
- document: string;
104
- /** Persist the shipped winner document (write the file the profile ref points at).
105
- * Called only on a ship verdict. When omitted, the winner is still returned in
106
- * `result.raw.winner.surface` for the caller to materialize. */
107
- writeBack?: (winnerDocument: string) => void | Promise<void>;
108
- }
109
- interface ImproveMemoryOptions {
110
- /** Current durable memory text used as the measured baseline. */
111
- document: string;
112
- /** Persist the promoted memory document. Never called on hold or error. */
113
- writeBack?: (winnerDocument: string) => void | Promise<void>;
114
- }
115
- interface ImproveCodeOptions {
116
- /** Repo root candidate worktrees fork from. */
117
- repoRoot: string;
118
- /** Base ref candidates fork from. Default `main`. */
119
- baseRef?: string;
120
- /** Directory worktrees are created under. Default `<repoRoot>/.worktrees`. */
121
- worktreeDir?: string;
122
- /** Git-compatible adapter override, primarily for tests. Candidate advancement
123
- * still requires normal Git worktree and commit semantics. */
124
- worktree?: WorktreeAdapter;
125
- /** Coding harness the agentic generator runs in each worktree. Default `claude`. */
126
- harness?: LocalHarness;
127
- /** Verify a candidate worktree before it becomes a measurable surface; failures
128
- * feed the next shot (see `agenticGenerator.verify` / `commandVerifier`). */
129
- verify?: Verifier;
130
- /** Per-shot wall-clock timeout for the harness (ms). */
131
- timeoutMs?: number;
132
- /** Byte-producer override — the test seam and the escape hatch for custom
133
- * candidate production. When set, `harness`/`verify`/`timeoutMs` are unused. */
134
- generator?: CandidateGenerator;
135
- }
136
- interface ImproveResult<TScenario extends Scenario, TArtifact> {
137
- /** The profile after improvement: the winner surface applied back into the
138
- * matching field when the gate shipped, else the input profile unchanged. */
139
- profile: AgentProfile;
140
- /** True when `gateDecision === 'ship'`. */
141
- shipped: boolean;
142
- /** Held-out lift (`winner − baseline` composite). */
143
- lift: number;
144
- /** The five-valued gate verdict from `selfImprove`. */
145
- gateDecision: SelfImproveResult<TScenario, TArtifact>['gateDecision'];
146
- /** Full `selfImprove` result for advanced inspection. For code runs,
147
- * `raw.winner.surface.worktreeRef` remains live after return whether the
148
- * candidate shipped or held; call `dispose()` after consuming it. */
149
- raw: SelfImproveResult<TScenario, TArtifact>;
150
- /** Release resources owned by this result. Idempotent; currently disposes
151
- * the returned code worktree and is a no-op for profile-only surfaces. */
152
- dispose(): Promise<void>;
153
- }
154
- /** Apply a promoted winner surface back into the profile field for `surface`.
155
- * Returns a shallow copy; never mutates the input profile. */
156
- declare function applyImprovementWinnerToProfile(profile: AgentProfile, surface: ImproveSurface, winner: MutableSurface): AgentProfile;
157
- /**
158
- * Run the held-out-gated self-improvement loop on ONE profile surface.
159
- *
160
- * @example Optimize the system prompt, default holdout gate:
161
- *
162
- * const out = await improve(profile, findings, {
163
- * surface: 'prompt',
164
- * scenarios,
165
- * judge,
166
- * agent: (surface, scenario, ctx) => runAgent(surface, scenario, ctx.signal),
167
- * })
168
- * if (out.shipped) deploy(out.profile)
169
- */
170
- declare function improve<TScenario extends Scenario, TArtifact>(profile: AgentProfile, findings: unknown[], opts: ImproveOptions<TScenario, TArtifact>): Promise<ImproveResult<TScenario, TArtifact>>;
171
-
172
- export { type ImproveOptions as I, type ImproveResult as a, type ImproveCodeOptions as b, type ImproveMemoryOptions as c, type ImproveSkillsOptions as d, type ImproveSurface as e, applyImprovementWinnerToProfile as f, improve as i };