@tangle-network/agent-runtime 0.94.13 → 0.96.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/README.md +64 -15
  2. package/dist/activation-B0ZD7nfX.d.ts +63 -0
  3. package/dist/agent.d.ts +6 -193
  4. package/dist/agent.js +10 -234
  5. package/dist/agent.js.map +1 -1
  6. package/dist/analyst-loop.d.ts +7 -10
  7. package/dist/analyst-loop.js +1 -2
  8. package/dist/candidate-execution/index.d.ts +43 -16
  9. package/dist/candidate-execution/index.js +17 -8
  10. package/dist/{chunk-VYA2YEKA.js → chunk-2KGAN2HM.js} +83 -14
  11. package/dist/chunk-2KGAN2HM.js.map +1 -0
  12. package/dist/chunk-3XKSBI2U.js +474 -0
  13. package/dist/chunk-3XKSBI2U.js.map +1 -0
  14. package/dist/{chunk-TVJQAYQM.js → chunk-6XKPVJAZ.js} +107 -716
  15. package/dist/chunk-6XKPVJAZ.js.map +1 -0
  16. package/dist/chunk-BLQIYRVR.js +699 -0
  17. package/dist/chunk-BLQIYRVR.js.map +1 -0
  18. package/dist/{chunk-QDSOD7RC.js → chunk-FD2MBMOH.js} +13 -101
  19. package/dist/chunk-FD2MBMOH.js.map +1 -0
  20. package/dist/{chunk-U33YZ7B2.js → chunk-FXF2OL34.js} +8 -8
  21. package/dist/{chunk-EP6RVHMX.js → chunk-HZDEXTSL.js} +848 -2
  22. package/dist/chunk-HZDEXTSL.js.map +1 -0
  23. package/dist/{chunk-WIPGQ4GT.js → chunk-IMSNJSXH.js} +1 -1
  24. package/dist/{chunk-WIPGQ4GT.js.map → chunk-IMSNJSXH.js.map} +1 -1
  25. package/dist/{chunk-VSWBYWFK.js → chunk-M6MD6JBS.js} +20 -27
  26. package/dist/chunk-M6MD6JBS.js.map +1 -0
  27. package/dist/chunk-PSOCBNM3.js +2069 -0
  28. package/dist/chunk-PSOCBNM3.js.map +1 -0
  29. package/dist/{chunk-AEG3NGJ2.js → chunk-Q2JSAVQ3.js} +34 -2
  30. package/dist/chunk-Q2JSAVQ3.js.map +1 -0
  31. package/dist/{chunk-XP5KDM3R.js → chunk-SGQ4YIQW.js} +4 -4
  32. package/dist/{chunk-C3UKLQ54.js → chunk-UQ6PNNXM.js} +18 -10
  33. package/dist/chunk-UQ6PNNXM.js.map +1 -0
  34. package/dist/{chunk-33OG2NN3.js → chunk-WYC2XJF2.js} +2 -2
  35. package/dist/{chunk-ZEYAT33L.js → chunk-Y3SRWZMP.js} +2 -2
  36. package/dist/{chunk-CNH7DF7Z.js → chunk-YOLKCWRV.js} +1116 -591
  37. package/dist/chunk-YOLKCWRV.js.map +1 -0
  38. package/dist/{completion-gate-D1gX1-hg.d.ts → completion-gate-C80jiRfN.d.ts} +1 -1
  39. package/dist/conversation.d.ts +12 -1
  40. package/dist/conversation.js +2 -3
  41. package/dist/{coordination-Dr_axlAf.d.ts → coordination-BFE3Den7.d.ts} +10 -11
  42. package/dist/environment-provider.d.ts +2 -2
  43. package/dist/environment-provider.js +1 -2
  44. package/dist/{agentic-generator-DDMM45kZ.d.ts → improve-g75IE2Cx.d.ts} +152 -4
  45. package/dist/{improvement-adapter-BieWeK5J.d.ts → improvement-adapter-HAZz-7vK.d.ts} +8 -31
  46. package/dist/index.d.ts +55 -28
  47. package/dist/index.js +206 -82
  48. package/dist/index.js.map +1 -1
  49. package/dist/intelligence.d.ts +185 -120
  50. package/dist/intelligence.js +509 -345
  51. package/dist/intelligence.js.map +1 -1
  52. package/dist/knowledge.d.ts +40 -12
  53. package/dist/knowledge.js +13 -7
  54. package/dist/{loop-runner-bin-BRQSQdHa.d.ts → loop-runner-bin-Cn1N2rRo.d.ts} +3 -3
  55. package/dist/loop-runner-bin.d.ts +6 -6
  56. package/dist/loop-runner-bin.js +8 -10
  57. package/dist/loops.d.ts +13 -13
  58. package/dist/loops.js +6 -8
  59. package/dist/mcp/bin.js +5 -7
  60. package/dist/mcp/bin.js.map +1 -1
  61. package/dist/mcp/index.d.ts +6 -6
  62. package/dist/mcp/index.js +12 -14
  63. package/dist/mcp/index.js.map +1 -1
  64. package/dist/platform.js +0 -2
  65. package/dist/platform.js.map +1 -1
  66. package/dist/primeintellect/index.js +1 -2
  67. package/dist/primeintellect/index.js.map +1 -1
  68. package/dist/profile-DbfaMTdk.d.ts +233 -0
  69. package/dist/profiles.d.ts +1 -1
  70. package/dist/profiles.js +0 -1
  71. package/dist/profiles.js.map +1 -1
  72. package/dist/{supervise-DmYOug5f.d.ts → supervise-BLPI50-w.d.ts} +3 -3
  73. package/dist/{types-ByAYqlVb.d.ts → types-B3vAW0Oq.d.ts} +1 -1
  74. package/dist/{prepare-CtdtsFNG.d.ts → types-CWqfCO8s.d.ts} +67 -298
  75. package/dist/{types-BC3bZpH0.d.ts → types-CmYCMbFT.d.ts} +12 -54
  76. package/dist/{types-1d5QGK3t.d.ts → types-CmnA2iL3.d.ts} +3 -3
  77. package/dist/{worktree-fanout-CPprU-qI.d.ts → worktree-fanout-DCA3G4bO.d.ts} +3 -3
  78. package/package.json +14 -16
  79. package/skills/build-with-agent-runtime/SKILL.md +122 -213
  80. package/dist/chunk-6O73TRHW.js +0 -142
  81. package/dist/chunk-6O73TRHW.js.map +0 -1
  82. package/dist/chunk-AEG3NGJ2.js.map +0 -1
  83. package/dist/chunk-C3UKLQ54.js.map +0 -1
  84. package/dist/chunk-CNH7DF7Z.js.map +0 -1
  85. package/dist/chunk-D3H7F6L2.js +0 -626
  86. package/dist/chunk-D3H7F6L2.js.map +0 -1
  87. package/dist/chunk-DGUM43GV.js +0 -11
  88. package/dist/chunk-DGUM43GV.js.map +0 -1
  89. package/dist/chunk-EP6RVHMX.js.map +0 -1
  90. package/dist/chunk-HGRW27YY.js +0 -214
  91. package/dist/chunk-HGRW27YY.js.map +0 -1
  92. package/dist/chunk-ISTDY47H.js +0 -849
  93. package/dist/chunk-ISTDY47H.js.map +0 -1
  94. package/dist/chunk-PCURO3DL.js +0 -661
  95. package/dist/chunk-PCURO3DL.js.map +0 -1
  96. package/dist/chunk-QDSOD7RC.js.map +0 -1
  97. package/dist/chunk-TVJQAYQM.js.map +0 -1
  98. package/dist/chunk-VSWBYWFK.js.map +0 -1
  99. package/dist/chunk-VYA2YEKA.js.map +0 -1
  100. package/dist/generator-YkAQrOoD.d.ts +0 -382
  101. package/dist/improve-BN3HyXIO.d.ts +0 -172
  102. package/dist/lifecycle.d.ts +0 -870
  103. package/dist/lifecycle.js +0 -981
  104. package/dist/lifecycle.js.map +0 -1
  105. package/dist/mcp-serve-verifier-DQQDbuyz.d.ts +0 -34
  106. package/skills/agent-runtime-adoption/SKILL.md +0 -246
  107. /package/dist/{chunk-U33YZ7B2.js.map → chunk-FXF2OL34.js.map} +0 -0
  108. /package/dist/{chunk-XP5KDM3R.js.map → chunk-SGQ4YIQW.js.map} +0 -0
  109. /package/dist/{chunk-33OG2NN3.js.map → chunk-WYC2XJF2.js.map} +0 -0
  110. /package/dist/{chunk-ZEYAT33L.js.map → chunk-Y3SRWZMP.js.map} +0 -0
@@ -1,870 +0,0 @@
1
- import { AgentProfile } from '@tangle-network/agent-interface';
2
- import { a as ProfileArtifact, A as ArtifactKind, b as ArtifactStatus, c as ArtifactInput, E as EvalRunner, d as EvalResult, G as GenerateContext, C as CandidateGenerator, e as PromotionVerdict, P as PromotionGate } from './generator-YkAQrOoD.js';
3
- export { f as ArtifactPayloads, H as HeldOutPromotionGateOptions, M as MarginalLift, g as MeasureMarginalLiftOptions, h as heldOutPromotionGate, m as measureMarginalLift, t as thresholdPromotionGate } from './generator-YkAQrOoD.js';
4
- import { LlmClientOptions, AnalystFinding } from '@tangle-network/agent-eval';
5
- import { M as McpServeSpec } from './mcp-serve-verifier-DQQDbuyz.js';
6
- import { L as LocalHarness } from './local-harness-ZqCx51u7.js';
7
- import './agentic-generator-DDMM45kZ.js';
8
- import '@tangle-network/agent-eval/campaign';
9
- import 'node:child_process';
10
-
11
- /**
12
- * `applyArtifact` — merge one `ProfileArtifact` onto an `AgentProfile`.
13
- *
14
- * This is the deterministic bridge between the artifact catalog and the §1.5
15
- * profile law: each `ArtifactKind` lands on exactly one profile field. The merge
16
- * is shallow-immutable — the input profile is never mutated; a new profile with
17
- * the artifact applied is returned. This is the ONE place that knows how a
18
- * `kind` maps to a profile field, so both the registry's `compose` and
19
- * `measureMarginalLift`'s with/without ablation share a single source of truth.
20
- */
21
-
22
- /**
23
- * Return a new profile with `artifact` merged onto `base`. Keyed kinds
24
- * (`tool`/`mcp`/`hook`/`subagent`) land under `artifact.key` (falling back to
25
- * `artifact.id`). `prompt` appends an instruction line; `skill` appends a
26
- * resource ref. Existing keys are overwritten by the artifact (the artifact is
27
- * the candidate being measured/promoted, so it wins on conflict).
28
- */
29
- declare function applyArtifact(base: AgentProfile, artifact: ProfileArtifact): AgentProfile;
30
- /** Apply many artifacts left-to-right; later artifacts win on key conflicts. */
31
- declare function applyArtifacts(base: AgentProfile, artifacts: readonly ProfileArtifact[]): AgentProfile;
32
-
33
- /**
34
- * `ArtifactRegistry` — a typed catalog of profile artifacts with stable ids.
35
- *
36
- * The registry is the lifecycle's source of truth for "what pieces could go into
37
- * this agent's profile". It holds `register` / `list` / `get` / `promote`, assigns
38
- * stable ids, and can `compose` a subset of artifacts onto a baseline profile via
39
- * the single `applyArtifact` bridge. It owns NO measurement, NO gate, NO LLM — it
40
- * is a pure in-memory store so callers can persist/snapshot it however they like.
41
- */
42
-
43
- /** Filter for `list`. Omit a field to leave that dimension unconstrained. */
44
- interface ArtifactQuery {
45
- kind?: ArtifactKind;
46
- status?: ArtifactStatus;
47
- }
48
- /**
49
- * The metadata key under which the registry stores an artifact's measured held-
50
- * back lift. This is the registry INVARIANT's anchor: an artifact is `active`
51
- * IFF this key holds a finite number — see `promoteWithLift` and `liftOf`. The
52
- * lifecycle never promotes by status flag alone; the lift score is the receipt.
53
- * `driftWatch` overwrites it with the latest re-measure so `liftOf` always
54
- * reflects the most recent evidence.
55
- */
56
- declare const liftMetadataKey = "measuredLift";
57
- /**
58
- * The metadata key under which the registry records WHY an artifact left the
59
- * active set — the human-readable reason a `demote` (→ decayed) or `retire`
60
- * (→ retired) carried. Kept so a demotion/retirement is auditable, not silent.
61
- */
62
- declare const lifecycleReasonKey = "lifecycleReason";
63
- /**
64
- * A typed, in-memory registry of `ProfileArtifact`s with stable ids.
65
- *
66
- * Ids are stable for the life of the registry: `register` assigns one (or honors
67
- * a caller-supplied id idempotently), and no later operation reassigns it.
68
- * Re-registering the same id REPLACES the artifact's mutable fields but preserves
69
- * the id, so a re-proposed candidate keeps its identity across generations.
70
- */
71
- declare class ArtifactRegistry {
72
- private readonly artifacts;
73
- private counter;
74
- /**
75
- * Register an artifact, returning the stored record (with its assigned id).
76
- * When `input.id` is set it is honored (idempotent re-registration replaces
77
- * the record under the same id); otherwise a stable id is minted as
78
- * `<kind>-<n>`. `status` defaults to `'candidate'`.
79
- */
80
- register<K extends ArtifactKind>(input: ArtifactInput<K>): ProfileArtifact<K>;
81
- /** Get an artifact by id, or `undefined` if it was never registered. */
82
- get(id: string): ProfileArtifact | undefined;
83
- /**
84
- * List artifacts, optionally filtered by `kind` and/or `status`. Returns a new
85
- * array in registration order; callers may safely sort/mutate the result.
86
- */
87
- list(query?: ArtifactQuery): ProfileArtifact[];
88
- /**
89
- * Mark an artifact `active`. Fails loud on an unknown id — promoting a
90
- * non-existent artifact is a caller bug, not a no-op. Returns the updated
91
- * record. Idempotent: promoting an already-active artifact is a no-op return.
92
- *
93
- * NOTE: the artifact-lifecycle INVARIANT (no measured lift ⇒ not active) is
94
- * enforced by `promoteWithLift`, the path the closed loop uses. This bare
95
- * `promote` exists for callers that gate elsewhere and just flip the flag; it
96
- * does NOT record a lift score, so `liftOf` returns `undefined` and a
97
- * lift-ranked `composeProfile` will skip it. Prefer `promoteWithLift`.
98
- */
99
- promote(id: string): ProfileArtifact;
100
- /**
101
- * Promote an artifact AND record the measured held-back lift that earned it.
102
- * This is the closed loop's promotion path and the enforcement point of the
103
- * lifecycle invariant: an artifact becomes `active` only WITH a finite lift
104
- * number stamped under `liftMetadataKey`. A non-finite `lift` (NaN/Infinity)
105
- * fails loud — promoting on a broken measurement is exactly the silent-zero the
106
- * doctrine forbids. Re-promotes a `decayed` artifact whose lift recovered.
107
- * Returns the updated record.
108
- */
109
- promoteWithLift(id: string, lift: number): ProfileArtifact;
110
- /**
111
- * Demote an `active` artifact to `decayed`: it was promoted, but a later
112
- * re-measure (`driftWatch`) found its held-back lift fell below the keep-bar.
113
- * Records the latest re-measured `lift` (so `liftOf` reflects current evidence)
114
- * and the `reason` (so the demotion is auditable). The artifact stays in the
115
- * registry — `decayed`, not deleted — so it can be re-promoted if a future
116
- * re-measure recovers the lift. Fails loud on an unknown id. Demoting a
117
- * non-`active` artifact fails loud too: only the active set decays.
118
- */
119
- demote(id: string, reason: string, lift?: number): ProfileArtifact;
120
- /**
121
- * Retire an artifact to the terminal `retired` state: it is permanently out of
122
- * the active set (`dedupeArtifacts` retires the weaker half of a non-stacking
123
- * pair). Records the `reason` for the audit trail. Unlike `demote`, this is
124
- * terminal — a retired artifact is never re-promoted by the loop. Idempotent on
125
- * an already-retired artifact; fails loud on an unknown id.
126
- */
127
- retire(id: string, reason: string): ProfileArtifact;
128
- /**
129
- * The measured held-back lift recorded at promotion time (and overwritten by
130
- * the latest `driftWatch` re-measure), or `undefined` when the artifact was
131
- * never promoted WITH a lift (a fresh candidate, or one promoted via the bare
132
- * `promote`). The lifecycle invariant in one accessor: `liftOf(id) ===
133
- * undefined` ⇒ the artifact has no measured lift ⇒ it is not eligible for a
134
- * lift-ranked compose. Note this returns the recorded lift regardless of status
135
- * — `composeProfile` separately filters to `active`, so a `decayed` artifact's
136
- * stale lift is visible for audit but never folded into a profile.
137
- */
138
- liftOf(id: string): number | undefined;
139
- /**
140
- * Compose a set of registered artifacts onto a baseline profile. With no ids
141
- * given, composes every `active` artifact (the "ship the passing set"
142
- * default). With explicit ids, composes exactly those (in id order given),
143
- * failing loud on any unknown id. The applied order is the order passed (or
144
- * registration order for the active-default), and later artifacts win on key
145
- * conflicts — same semantics as `applyArtifacts`.
146
- */
147
- compose(base: AgentProfile, ids?: readonly string[]): AgentProfile;
148
- /** Number of registered artifacts (any status). */
149
- get size(): number;
150
- private mintId;
151
- }
152
- /** Construct an empty `ArtifactRegistry`. */
153
- declare function createArtifactRegistry(): ArtifactRegistry;
154
-
155
- /**
156
- * `composeProfile` — fold the top-k active artifacts back into a profile.
157
- *
158
- * This is the "give it all the passing options at once" step. After the loop has
159
- * generated, measured, and promoted artifacts, `composeProfile` selects the
160
- * highest-lift ACTIVE ones (those promoted WITH a measured held-back lift) and
161
- * applies them onto a baseline profile via the single `applyArtifact` bridge.
162
- *
163
- * It differs from the registry's raw `compose` in two ways the lifecycle needs:
164
- * 1. It ranks by MEASURED LIFT and takes the top `k`, rather than applying
165
- * every promoted artifact in registration order. The best pieces go in
166
- * first; a `k` budget keeps the composed profile from accreting marginal
167
- * wins without bound.
168
- * 2. It enforces the lifecycle INVARIANT — only artifacts with a finite
169
- * `liftOf` are eligible. An artifact flipped to `promoted` without a lift
170
- * receipt is invisible here, by construction.
171
- */
172
-
173
- interface ComposeProfileOptions {
174
- /** Cap on how many artifacts to fold in. Default: all eligible. */
175
- k?: number;
176
- /** Restrict to one surface (e.g. only fold in skills). Default: all kinds. */
177
- kind?: ArtifactKind;
178
- }
179
- /**
180
- * Return a new profile with the top-`k` active artifacts (highest measured lift
181
- * first) applied onto `base`.
182
- *
183
- * "Active" = promoted WITH a finite measured lift (`registry.liftOf` returns a
184
- * number) — the lifecycle invariant. Ties in lift fall back to registration
185
- * order (stable). With no `k`, every eligible artifact is folded in.
186
- *
187
- * @param registry the catalog the loop populated
188
- * @param base the baseline profile to fold artifacts onto (the empty profile
189
- * on a cold start)
190
- * @param opts `k` (top-k budget) and an optional `kind` filter
191
- *
192
- * @example Cold start — fold the single best distilled skill back in:
193
- * const composed = composeProfile(registry, emptyProfile, { kind: 'skill', k: 1 })
194
- */
195
- declare function composeProfile(registry: ArtifactRegistry, base: AgentProfile, opts?: ComposeProfileOptions): AgentProfile;
196
-
197
- /**
198
- * `dedupeArtifacts` — retire the redundant half of a non-stacking pair.
199
- *
200
- * The lifecycle promotes each artifact on its OWN marginal lift, in isolation.
201
- * But two artifacts can each earn lift alone yet teach the agent the SAME thing —
202
- * a distilled "check state first" skill and a "verify before acting" prompt line,
203
- * say. Composed together they don't add up: the agent already learned the tactic
204
- * from the first, so the second buys little. Keeping both is wasted profile
205
- * surface (more context, more cost, more ways to conflict) for no extra score.
206
- *
207
- * `dedupeArtifacts` is the judge that catches this. For each pair of `active`
208
- * artifacts it measures whether their lifts STACK: it scores the baseline, each
209
- * artifact alone, and BOTH together, then compares the combined lift against the
210
- * sum of the individual lifts. When `combined < (a + b) − tolerance`, the pair
211
- * overlaps — they do not stack — and the weaker member (lower individual lift) is
212
- * `retire`d. Retirement is terminal: the kept member already delivers the shared
213
- * value, so the loop should not re-promote the redundant one.
214
- *
215
- * The "judge" is a MEASUREMENT, not an LLM verdict — it reuses the same ablation
216
- * machinery (`measureMarginalLift` + the caller's `EvalRunner`) the rest of the
217
- * lifecycle runs on, so the selector≠judge firewall holds and there is no new
218
- * execution model. It is surface-agnostic: any two artifacts (skill+prompt,
219
- * tool+tool, …) are compared identically, because stacking is measured on the
220
- * composed profile via the one `applyArtifacts` bridge.
221
- */
222
-
223
- interface DedupeOptions {
224
- /** The registry whose `active` artifacts are pairwise stack-tested. Mutated in
225
- * place: the weaker member of each non-stacking pair is retired. */
226
- registry: ArtifactRegistry;
227
- /** The baseline profile the stacking ablation runs on top of. The "without"
228
- * arm is scored once and shared across every pair. */
229
- baseline: AgentProfile;
230
- /** Scores a profile on the held-back split. Called for the baseline, each
231
- * artifact alone, and each candidate pair together. */
232
- evalRunner: EvalRunner;
233
- /**
234
- * The stacking tolerance. A pair is judged non-stacking (redundant) when the
235
- * combined lift falls SHORT of the sum of individual lifts by more than this:
236
- * `combined < a + b − tolerance`. A small positive tolerance absorbs eval
237
- * noise so only a real overlap retires an artifact. Default 0 — any shortfall
238
- * counts as non-stacking (use a positive value on noisy live evals).
239
- */
240
- tolerance?: number;
241
- /** Restrict dedupe to one surface (only compare skills against skills, …).
242
- * Default: every `active` artifact is a candidate, across kinds. */
243
- kind?: ProfileArtifact['kind'];
244
- /** A pre-computed baseline result, to skip the shared "without" run. */
245
- baselineResult?: EvalResult;
246
- /** Cooperative cancellation, forwarded to every `evalRunner` call. */
247
- signal?: AbortSignal;
248
- }
249
- /** The stacking verdict for one pair of active artifacts. */
250
- interface PairStackCheck {
251
- /** Ids of the two artifacts compared, in (a, b) order as examined. */
252
- pair: [string, string];
253
- /** Individual held-back lift of the first artifact alone (re-measured). */
254
- liftA: number;
255
- /** Individual held-back lift of the second artifact alone (re-measured). */
256
- liftB: number;
257
- /** Held-back lift of BOTH artifacts composed together. */
258
- combinedLift: number;
259
- /** `combinedLift − (liftA + liftB)`: ≥ −tolerance ⇒ they stack; below ⇒ they
260
- * overlap (redundant). */
261
- stackGap: number;
262
- /** Whether the pair was judged non-stacking (redundant). */
263
- redundant: boolean;
264
- /** The id retired when redundant (the weaker member), else `undefined`. */
265
- retiredId?: string;
266
- }
267
- interface DedupeResult {
268
- /** One verdict per examined pair, in iteration order. Pairs where one member
269
- * was already retired by an earlier pair this cycle are skipped. */
270
- checks: PairStackCheck[];
271
- /** Ids retired this cycle (the weaker member of each non-stacking pair). */
272
- retired: string[];
273
- /** The shared baseline eval (the "without" arm, measured once). */
274
- baselineResult: EvalResult;
275
- }
276
- /**
277
- * Pairwise stack-test the `active` artifacts and retire the redundant half of
278
- * each non-stacking pair.
279
- *
280
- * For every unordered pair of active artifacts (optionally within one `kind`),
281
- * the combined lift is compared against the sum of individual lifts. A pair is
282
- * non-stacking when `combinedLift < liftA + liftB − tolerance`; the lower-lift
283
- * member is retired (ties retire the second-examined). An artifact retired by one
284
- * pair is removed from the remaining comparisons that cycle, so a cluster of
285
- * three mutually-redundant artifacts collapses to its single strongest member.
286
- *
287
- * Cost is one shared baseline run, one re-measure per active artifact (cached
288
- * across the pairs it appears in), and one combined run per still-eligible pair.
289
- *
290
- * @example Retire skills that teach the same tactic as a stronger one:
291
- * const out = await dedupeArtifacts({ registry, baseline, evalRunner, kind: 'skill' })
292
- * if (out.retired.length) report(`retired ${out.retired.length} redundant skills`)
293
- */
294
- declare function dedupeArtifacts(opts: DedupeOptions): Promise<DedupeResult>;
295
-
296
- /**
297
- * `driftWatch` — the scheduled re-measure that DEMOTES decayed artifacts.
298
- *
299
- * Promotion is a one-time decision on the evidence available THEN; the world
300
- * moves on. A skill that earned its lift against last month's task mix, a tool
301
- * grant the underlying model has since internalized, a prompt line another active
302
- * artifact now subsumes — each can quietly stop pulling its weight. An artifact
303
- * that no longer earns its keep but stays `active` is dead weight in every
304
- * composed profile and a lie in the registry's lift receipt.
305
- *
306
- * `driftWatch` closes that gap. It re-runs the SAME `measureMarginalLift`
307
- * ablation the loop used to promote each `active` artifact — over the CURRENT
308
- * baseline and the CURRENT eval — and demotes any whose re-measured held-back
309
- * lift fell below the keep-bar. Demotion moves the artifact `active` → `decayed`
310
- * (reversible: a later re-measure that recovers the lift can re-promote it), so
311
- * it drops out of `composeProfile` immediately while staying in the registry as
312
- * an auditable record.
313
- *
314
- * This is NOT a new execution model — it is the existing ablation, run on a
315
- * schedule, against the active set. The schedule itself is the caller's cron;
316
- * this module is the one re-measure-and-demote step that cron invokes. It is
317
- * surface-agnostic (skills, tools, prompts, MCP all re-measure identically) — the
318
- * only per-surface logic lives, as everywhere in the lifecycle, behind the
319
- * artifact's own `applyArtifact` bridge, which `measureMarginalLift` already uses.
320
- */
321
-
322
- interface DriftWatchOptions {
323
- /** The registry whose `active` artifacts are re-measured. Mutated in place:
324
- * artifacts that fail the keep-bar are demoted to `decayed`. */
325
- registry: ArtifactRegistry;
326
- /** The baseline profile each artifact is re-measured ON TOP OF — the SAME
327
- * ablation shape as promotion. Pass the CURRENT baseline (the world the
328
- * artifact lives in now), which may differ from the one it was promoted on. */
329
- baseline: AgentProfile;
330
- /** Scores a profile on the held-back split. The shared baseline arm is run
331
- * once and reused across every artifact's ablation (the "without" arm). */
332
- evalRunner: EvalRunner;
333
- /**
334
- * The absolute keep-bar: an artifact stays `active` only while its re-measured
335
- * held-back lift is STRICTLY ABOVE this floor. Default 0 — an artifact that no
336
- * longer adds anything (or now subtracts) decays. Mirrors `thresholdPromotionGate`'s
337
- * `minDelta` so the keep-bar and the promote-bar can be set consistently.
338
- */
339
- minLift?: number;
340
- /**
341
- * The relative keep-bar: an artifact also decays if its re-measured lift fell
342
- * to below this FRACTION of the lift recorded at promotion (`registry.liftOf`).
343
- * E.g. `0.5` demotes an artifact that lost more than half its original lift,
344
- * even if it still clears `minLift`. Default: unset (no relative check — only
345
- * the absolute `minLift` floor applies). Ignored for an artifact with no
346
- * recorded prior lift (a bare `promote`), which is judged on `minLift` alone.
347
- */
348
- maxRelativeDecay?: number;
349
- /** Restrict the watch to one surface (e.g. re-measure only skills). Default:
350
- * every `active` artifact, regardless of kind. */
351
- kind?: ProfileArtifact['kind'];
352
- /** A pre-computed baseline result, to skip the shared "without" run when the
353
- * caller already scored the current baseline this cycle. */
354
- baselineResult?: EvalResult;
355
- /** Cooperative cancellation, forwarded to every `evalRunner` call. */
356
- signal?: AbortSignal;
357
- }
358
- /** Per-artifact record of what the re-measure found and decided. */
359
- interface DriftCheck {
360
- /** The re-measured artifact (status reflects the decision: still `active`, or
361
- * now `decayed`). */
362
- artifact: ProfileArtifact;
363
- /** The lift recorded at promotion time (`registry.liftOf` before the check),
364
- * or `undefined` if it was promoted without a lift receipt. */
365
- priorLift: number | undefined;
366
- /** The freshly re-measured held-back lift (with − without composite). */
367
- currentLift: number;
368
- /** Whether this re-measure demoted the artifact (`active` → `decayed`). */
369
- demoted: boolean;
370
- /** Human-readable reason the artifact was kept or demoted (the same string
371
- * recorded under `lifecycleReasonKey` on a demotion). */
372
- reason: string;
373
- }
374
- interface DriftWatchResult {
375
- /** One check per `active` artifact examined, in registry order. */
376
- checks: DriftCheck[];
377
- /** Ids of the artifacts demoted to `decayed` this cycle. */
378
- demoted: string[];
379
- /** The shared baseline eval (the "without" arm, measured once). */
380
- baselineResult: EvalResult;
381
- }
382
- /**
383
- * Re-measure every `active` artifact and demote those whose held-back lift
384
- * decayed below the keep-bar.
385
- *
386
- * For each `active` artifact (optionally filtered to one `kind`), this re-runs
387
- * `measureMarginalLift` over the supplied `baseline`, then applies the keep-bar:
388
- *
389
- * - ABSOLUTE: `currentLift > minLift` (default `> 0`), and
390
- * - RELATIVE (when `maxRelativeDecay` is set AND a prior lift exists):
391
- * `currentLift >= priorLift * (1 − maxRelativeDecay)`.
392
- *
393
- * An artifact failing EITHER bar is demoted to `decayed` (recording the
394
- * re-measured lift + reason) and drops out of `composeProfile`. An artifact that
395
- * passes both bars stays `active`, with its recorded lift refreshed to the latest
396
- * measurement so `liftOf` never reports stale evidence.
397
- *
398
- * The baseline "without" arm is scored ONCE and shared across all artifacts, so
399
- * the cost is `1 + (number of active artifacts)` eval runs.
400
- *
401
- * @example A nightly cron that demotes any active artifact that lost its lift:
402
- * const out = await driftWatch({ registry, baseline, evalRunner, minLift: 0 })
403
- * if (out.demoted.length) report(`demoted ${out.demoted.length} decayed artifacts`)
404
- */
405
- declare function driftWatch(opts: DriftWatchOptions): Promise<DriftWatchResult>;
406
-
407
- /**
408
- * `promptGenerator` — the `CandidateGenerator` for the PROMPT surface.
409
- *
410
- * It is to the prompt surface what `skillGenerator` is to skills: the thin
411
- * per-surface adapter that turns the agent's history into fresh, unmeasured
412
- * prompt-instruction candidates the surface-agnostic `runLifecycle` loop then
413
- * measures, gates, and promotes. Each candidate is a `prompt` artifact carrying
414
- * one `{ instruction }` line that `applyArtifact` appends to
415
- * `profile.prompt.instructions`.
416
- *
417
- * Two candidate sources, run together each generation:
418
- *
419
- * 1. REFINE (incumbent-grounded) — drive agent-eval's `gepaProposer`, the
420
- * reflective prompt-tier proposer. It reflects on the incumbent prompt +
421
- * the trace findings and proposes TARGETED rewrites. This is the exploit
422
- * arm: it polishes the framing the agent already has.
423
- *
424
- * 2. SEED (basin escape) — author N genuinely DIVERSE fresh instruction
425
- * lines from the TASK SPEC, NOT from the incumbent. A reflective proposer
426
- * only ever perturbs the current surface, so on its own it can polish a
427
- * local minimum forever — it never tries a fundamentally different framing.
428
- * The seed arm authors several rewrites that each take a DIFFERENT stance
429
- * (imperative vs. checklist vs. failure-mode-first, …) so the search can
430
- * JUMP basins instead of only descending the one it starts in. This is the
431
- * explore arm, and it is the piece `gepaProposer` structurally cannot do.
432
- *
433
- * Both seams are INJECTED (per the §1.5 law: the generator AUTHORS profile
434
- * pieces, it does not embed a specific LLM loop). `refine` wraps `gepaProposer`;
435
- * `authorDiverseSeeds` wraps an LLM author. A test injects pure stubs for both,
436
- * keeping the closed loop deterministically testable; production wires the real
437
- * router-backed engines (see `productionPromptGenerator`).
438
- */
439
-
440
- /** A proposed prompt instruction line plus the WHY behind it. The `rationale`
441
- * rides into the artifact metadata so a promotion decision is auditable. */
442
- interface PromptDraft {
443
- /** The instruction line appended to `profile.prompt.instructions`. */
444
- instruction: string;
445
- /** Short human label for review surfaces. */
446
- label: string;
447
- /** Why this line was proposed — which failure / framing it targets. */
448
- rationale: string;
449
- }
450
- /**
451
- * REFINE — incumbent-grounded rewrites. Given the lifecycle context, return
452
- * targeted edits OF the current prompt framing. The production implementation
453
- * drives `gepaProposer`; a test injects a pure function. Returns zero or more
454
- * drafts.
455
- */
456
- type RefinePrompt = (ctx: GenerateContext) => Promise<PromptDraft[]> | PromptDraft[];
457
- /**
458
- * SEED — author N genuinely DIVERSE fresh instruction lines from the task spec,
459
- * NOT mutations of the incumbent. This is the local-minimum escape: each seed
460
- * MUST take a different framing so the population spans multiple basins. The
461
- * production implementation makes one LLM call at a non-trivial temperature; a
462
- * test injects a pure function. Returns up to `count` drafts.
463
- */
464
- type AuthorDiverseSeeds = (ctx: GenerateContext, count: number) => Promise<PromptDraft[]> | PromptDraft[];
465
- interface PromptGeneratorOptions {
466
- /** OPTIONAL — the exploit arm (incumbent-grounded rewrites via `gepaProposer`).
467
- * Omit to run seeds-only. */
468
- refine?: RefinePrompt;
469
- /** OPTIONAL — the explore arm (diverse fresh seeds). Omit to run refine-only. */
470
- authorDiverseSeeds?: AuthorDiverseSeeds;
471
- /** How many diverse seeds the explore arm authors each generation. Default 3.
472
- * Zero disables seeding even when `authorDiverseSeeds` is set. */
473
- diverseSeedCount?: number;
474
- }
475
- /**
476
- * Build a `CandidateGenerator` for the prompt surface. Each generation it pools
477
- * the refine arm (incumbent rewrites) and the seed arm (diverse fresh framings),
478
- * de-duplicates by instruction text, and emits each as a `prompt` artifact.
479
- *
480
- * At least one of `refine` / `authorDiverseSeeds` MUST be provided — a generator
481
- * with neither has no way to produce a candidate and is a wiring bug, so it
482
- * throws at construction rather than silently returning `[]` every round.
483
- *
484
- * @example Production wiring (refine = gepaProposer, seed = LLM author):
485
- * promptGenerator({
486
- * refine: gepaRefine(llm, model),
487
- * authorDiverseSeeds: routerSeedAuthor(llm, model),
488
- * })
489
- */
490
- declare function promptGenerator(opts: PromptGeneratorOptions): CandidateGenerator<'prompt'>;
491
- interface ProductionPromptGeneratorOptions {
492
- /** Router transport (baseUrl/apiKey) for both the refine and seed arms. */
493
- llm: LlmClientOptions;
494
- /** Model that performs reflection + seed authoring. Default `deepseek-v4-flash`. */
495
- model?: string;
496
- /** Population size handed to `gepaProposer`'s `propose`. Default 3. */
497
- refinePopulation?: number;
498
- /** Diverse-seed count. Default 3. */
499
- diverseSeedCount?: number;
500
- /** Seed-authoring temperature — high on purpose so the seeds spread across
501
- * framings rather than collapsing onto one. Default 1.0. */
502
- seedTemperature?: number;
503
- }
504
- /**
505
- * Production `promptGenerator`: refine via `gepaProposer`, seed via a
506
- * router-backed diverse author. The one call a consumer makes to grow prompt
507
- * artifacts with the real engines.
508
- */
509
- declare function productionPromptGenerator(opts: ProductionPromptGeneratorOptions): CandidateGenerator<'prompt'>;
510
- /**
511
- * Wrap `gepaProposer` as a `RefinePrompt`. The proposer reflects on the
512
- * incumbent prompt (`ctx.baseline.prompt.systemPrompt`) and the trace findings
513
- * to propose `population` targeted rewrites. We feed it the generation-0
514
- * `ProposeContext` (no scored history yet inside one lifecycle round), so it
515
- * reflects on the current surface against its mutation primitives — exactly the
516
- * "polish the framing you already have" arm.
517
- */
518
- declare function gepaRefine(llm: LlmClientOptions, model: string, population: number): RefinePrompt;
519
- /**
520
- * A router-backed `AuthorDiverseSeeds`: one structured LLM call that authors
521
- * `count` instruction lines, each REQUIRED to take a distinct framing. The
522
- * prompt is grounded in the task spec (the incumbent system prompt as the spec
523
- * of WHAT the agent must do) plus the trace findings (what it gets wrong) — but
524
- * the model is told to author FRESH lines, not edits, so the output spans
525
- * basins the incumbent's neighborhood never reaches.
526
- */
527
- declare function routerSeedAuthor(llm: LlmClientOptions, model: string, temperature: number): AuthorDiverseSeeds;
528
-
529
- /**
530
- * `runLifecycle` — the ONE closed-loop orchestrator: generate → measure →
531
- * promote → store.
532
- *
533
- * It is surface-agnostic: it knows nothing about skills vs tools vs prompts. All
534
- * per-surface logic lives behind the `CandidateGenerator` seam. The loop:
535
- *
536
- * 1. GENERATE — ask each generator for candidate artifacts from the agent's
537
- * history (traces/findings). Register them as `candidate`s.
538
- * 2. MEASURE — for each candidate, run `measureMarginalLift` on the held-
539
- * back split (the with/without ablation, baseline shared across
540
- * candidates so the "without" arm runs once).
541
- * 3. PROMOTE — run the `PromotionGate` (default: the held-back exam). On a
542
- * pass, `promoteWithLift` records the measured lift — the
543
- * registry invariant: no measured lift ⇒ never active.
544
- * 4. STORE — every candidate (promoted or not) lands in the registry with
545
- * provenance (domain, generation, generator kind, gate verdict)
546
- * + its lift in metadata, so the decision is auditable.
547
- *
548
- * Compose is intentionally NOT part of this loop — folding the promoted set back
549
- * into a profile is a separate, explicit `composeProfile` call the caller makes
550
- * when it wants a deployable profile. Keeping them separate means a caller can
551
- * run the loop to grow the catalog without committing a profile change.
552
- */
553
-
554
- interface RunLifecycleOptions {
555
- /** The baseline profile candidates are proposed and measured on top of. On a
556
- * cold start this is the empty (or near-empty) profile. */
557
- baseline: AgentProfile;
558
- /** The agent/domain id — namespaces provenance + scopes generators. */
559
- domain: string;
560
- /** The per-surface candidate generators. One per surface the loop grows; the
561
- * loop runs them in order and pools their candidates. */
562
- generators: ReadonlyArray<CandidateGenerator>;
563
- /** Scores a profile on the HELD-BACK split. Run by `measureMarginalLift` —
564
- * once for the shared baseline, once per candidate. */
565
- evalRunner: EvalRunner;
566
- /** The promotion gate (the held-back exam). Default-free on purpose: the
567
- * caller chooses the policy (`thresholdPromotionGate` / `heldOutPromotionGate`). */
568
- gate: PromotionGate;
569
- /** Trace-analyst findings the generators distill/propose from. */
570
- findings?: ReadonlyArray<AnalystFinding>;
571
- /** Raw captured traces for generators (e.g. a skill `distill`). Opaque. */
572
- traces?: unknown;
573
- /** Generation counter for provenance. Default 0. */
574
- generation?: number;
575
- /** An existing registry to grow across generations. Default: a fresh one. */
576
- registry?: ArtifactRegistry;
577
- /** Cooperative cancellation. */
578
- signal?: AbortSignal;
579
- }
580
- /** The per-candidate record of what the loop decided and why. */
581
- interface CandidateOutcome {
582
- /** The stored artifact (status reflects the gate verdict). */
583
- artifact: ProfileArtifact;
584
- /** The surface this candidate targeted. */
585
- kind: ArtifactKind;
586
- /** Measured held-back lift (with − without composite). */
587
- scoreDelta: number;
588
- /** Measured extra USD cost the artifact adds. */
589
- costDelta: number;
590
- /** The gate's verdict. */
591
- verdict: PromotionVerdict;
592
- /** Whether the candidate was promoted into the registry as active. */
593
- promoted: boolean;
594
- }
595
- interface RunLifecycleResult {
596
- /** The registry, grown with this generation's candidates + promotions. */
597
- registry: ArtifactRegistry;
598
- /** One outcome per candidate the generators produced, in generation order. */
599
- outcomes: CandidateOutcome[];
600
- /** Ids of the artifacts promoted this generation. */
601
- promoted: string[];
602
- /** The shared baseline eval (the "without" arm, measured once). */
603
- baselineResult: EvalResult;
604
- }
605
- /**
606
- * Run ONE generation of the artifact lifecycle.
607
- *
608
- * @example Cold start on a fixture domain (the closed loop in one call):
609
- * const out = await runLifecycle({
610
- * baseline: emptyProfile,
611
- * domain: 'support-bot',
612
- * generators: [skillGenerator({ distill, refine })],
613
- * evalRunner: scoreOnHeldBackSplit,
614
- * gate: thresholdPromotionGate(),
615
- * traces: seededTraces,
616
- * })
617
- * const composed = composeProfile(out.registry, emptyProfile, { kind: 'skill' })
618
- */
619
- declare function runLifecycle(opts: RunLifecycleOptions): Promise<RunLifecycleResult>;
620
-
621
- /**
622
- * `skillGenerator` — the reference `CandidateGenerator` for the SKILL surface,
623
- * and the literal answer to "an empty profile has no skills, so what creates
624
- * one?".
625
- *
626
- * It is a two-step pipeline:
627
- *
628
- * 1. DISTILL — read the agent's traces/findings and WRITE a new skill document
629
- * (a reusable how-to note: "check state before acting", "verify after every
630
- * edit"). This is the CREATE step. It is the piece `skillOpt` cannot do —
631
- * an optimizer refines an existing skill, it cannot conjure one from
632
- * nothing. The production `distill` is `reflectiveGenerator`-style: an LLM
633
- * reflection over the trace produces the first draft.
634
- *
635
- * 2. REFINE — take the distilled draft and improve its wording/structure. The
636
- * production `refine` drives agent-eval's `skillOptProposer` (the uniform
637
- * skill-surface proposer factory from `@tangle-network/agent-eval/campaign`,
638
- * the same source as `gepaProposer` for the prompt surface). Refinement is
639
- * optional: with no `refine`, the distilled draft IS the candidate.
640
- *
641
- * Both steps are INJECTED seams, not hardcoded engines — per the §1.5 law, the
642
- * generator AUTHORS a profile piece; it does not embed a specific LLM loop. That
643
- * keeps the closed loop deterministically testable (inject pure stubs) while
644
- * production wires the real distill + skillOpt. The generator emits a `skill`
645
- * artifact (an inline `SKILL.md` resource ref) the orchestrator measures + gates.
646
- */
647
-
648
- /** A distilled skill draft: a name + the `SKILL.md` body. */
649
- interface SkillDraft {
650
- /** Skill name — becomes the inline resource ref name + the artifact name. */
651
- name: string;
652
- /** The `SKILL.md` document body (markdown). */
653
- content: string;
654
- /** Optional one-line description for review surfaces. */
655
- description?: string;
656
- }
657
- /**
658
- * DISTILL — create new skill drafts from the agent's history. Returns zero or
659
- * more drafts (zero is valid: nothing worth distilling this round). The
660
- * production implementation reflects over `ctx.traces` / `ctx.findings` with an
661
- * LLM; a test injects a pure function.
662
- */
663
- type DistillSkills = (ctx: GenerateContext) => Promise<SkillDraft[]> | SkillDraft[];
664
- /**
665
- * REFINE — improve ONE distilled draft (wording, structure, examples). The
666
- * production implementation drives `skillOptProposer`. Returns the refined draft;
667
- * when omitted from `skillGenerator`, the distilled draft is used as-is.
668
- */
669
- type RefineSkill = (draft: SkillDraft) => Promise<SkillDraft> | SkillDraft;
670
- interface SkillGeneratorOptions {
671
- /** REQUIRED — the create step. Without it there is no skill to optimize. */
672
- distill: DistillSkills;
673
- /** OPTIONAL — the optimize step. Omit to ship distilled drafts unrefined. */
674
- refine?: RefineSkill;
675
- }
676
- /**
677
- * Build a `CandidateGenerator` for the skill surface that distills new skills
678
- * from history, then (optionally) refines them, and emits each as a `skill`
679
- * artifact carrying an inline `SKILL.md` resource ref.
680
- *
681
- * @example Production wiring (distill = LLM reflection, refine = skillOptProposer):
682
- * skillGenerator({
683
- * distill: reflectiveDistill, // creates the draft from traces
684
- * refine: skillOptRefine, // optimizes the draft via skillOptProposer
685
- * })
686
- */
687
- declare function skillGenerator(opts: SkillGeneratorOptions): CandidateGenerator<'skill'>;
688
-
689
- /**
690
- * `buildableGenerator` — the `CandidateGenerator` for the BUILDABLE surfaces
691
- * (`tool` and `mcp`), and the answer to "a profile can't *write* a new tool from
692
- * a prompt rewrite — who builds it?".
693
- *
694
- * Unlike the prompt and skill surfaces, where a candidate is a piece of TEXT an
695
- * LLM authors in one call, a tool or an MCP server is CODE that must compile,
696
- * pass its tests, and — for an MCP server — actually boot and serve. You cannot
697
- * one-shot that reliably. So this generator is a SUPERVISOR DISPATCH, not a
698
- * single author:
699
- *
700
- * 1. FAN OUT — spawn N parallel candidate implementations, each built in its
701
- * OWN git worktree by a real coding harness (research → implement
702
- * → test → prove it compiles / actually serves). The per-candidate
703
- * build is the `buildCandidate` seam; production wires it to the
704
- * shipped `improvementDriver` + `agenticGenerator` + a verifier
705
- * (`commandVerifier` for a tool, `mcpServeVerifier` for an MCP).
706
- * 2. FILTER — keep only the VERIFIED builds. An unverified worktree is never a
707
- * candidate (the verifier is the gate — same valid-only discipline
708
- * as `worktreeFanout`'s `selectValidWinner`). A build that never
709
- * compiles/serves is discarded, never ranked.
710
- * 3. RANK — score each verified survivor by `measureMarginalLift` against
711
- * `ctx.baseline` (the with/without held-back ablation — the SAME
712
- * selector the rest of the lifecycle uses; never a judge).
713
- * 4. EMIT — return the single best survivor as ONE `tool` / `mcp` artifact,
714
- * carrying the measured lift + the winning worktree ref as
715
- * auditable provenance in metadata.
716
- *
717
- * Steps 2–4 (filter, lift-rank, emit) are surface-agnostic and live here; only
718
- * `buildCandidate` (the per-candidate worktree build) varies, and it is INJECTED
719
- * — production wires the real harness fan-out, a test injects pure stubs so the
720
- * dispatch is deterministically exercisable without spawning processes or models.
721
- *
722
- * NB the lifecycle orchestrator (`runLifecycle`) STILL measures + gates whatever
723
- * this returns. The rank here is an INTERNAL selection — "which of the N parallel
724
- * builds is the best candidate to put forward" — not the promotion gate. A
725
- * generator that puts forward a measured-best candidate and an orchestrator that
726
- * re-measures it on the held-back exam are not redundant: the first picks among
727
- * siblings, the second decides whether the winner clears the bar to ship.
728
- */
729
-
730
- /** The buildable surfaces — the kinds whose candidate IS code that must compile
731
- * / serve, so building one is a fan-out-and-verify dispatch, not a one-shot. */
732
- type BuildableKind = 'tool' | 'mcp';
733
- /**
734
- * The result of building ONE candidate in its own worktree. A build either
735
- * verified (compiled + tests passed, or — for an MCP — booted and served) or it
736
- * did not; an unverified build is dropped before ranking, so `verified:false`
737
- * carries the reason for the audit trail but never becomes a candidate.
738
- */
739
- interface BuiltCandidate {
740
- /** A short label for this candidate (worktree branch / trace node). */
741
- label: string;
742
- /** Did the build compile + pass its verifier? Only verified builds rank. */
743
- verified: boolean;
744
- /** The worktree path / git ref holding the built change (provenance). */
745
- worktreeRef: string;
746
- /**
747
- * For an `mcp` candidate: how to START the built server (stdio transport).
748
- * Becomes the `AgentProfileMcpServer` the artifact carries. REQUIRED for an
749
- * `mcp` build; ignored for a `tool` build.
750
- */
751
- serve?: {
752
- command: string;
753
- args?: string[];
754
- cwd?: string;
755
- env?: Record<string, string>;
756
- };
757
- /**
758
- * For a `tool` candidate: the tool name the grant lands under (the profile
759
- * `tools` key). REQUIRED for a `tool` build; ignored for an `mcp` build.
760
- */
761
- toolName?: string;
762
- /** Why the build failed verification, when `verified` is false (audit only). */
763
- failureReason?: string;
764
- /** Free-form provenance to ride into the artifact metadata (build summary, …). */
765
- metadata?: Record<string, unknown>;
766
- }
767
- /**
768
- * BUILD ONE candidate. Given the lifecycle context and the index in the fan-out,
769
- * produce a fresh worktree implementation and report whether it verified. The
770
- * production implementation drives a real coding harness in a fresh worktree
771
- * (`worktreeBuildCandidate`); a test injects a pure function.
772
- *
773
- * MUST NOT measure lift, gate, or register — that is the dispatch's / the
774
- * orchestrator's job. It only builds + verifies ONE sibling.
775
- */
776
- type BuildCandidate = (ctx: GenerateContext, index: number, signal: AbortSignal) => Promise<BuiltCandidate>;
777
- interface BuildableGeneratorOptions {
778
- /** The buildable surface this generator targets (`tool` or `mcp`). */
779
- kind: BuildableKind;
780
- /** The per-candidate build seam (the fan-out leaf). REQUIRED. */
781
- buildCandidate: BuildCandidate;
782
- /** How many candidate implementations to build in parallel each generation.
783
- * Default 3. Must be >= 1. The conserved-budget / live-worker caps that bound
784
- * the real fan-out live in the production `buildCandidate` wiring. */
785
- fanout?: number;
786
- /** Scores a profile on the held-back split — used to RANK the verified
787
- * siblings by `measureMarginalLift`. REQUIRED: without it there is no way to
788
- * pick the best of N, and "best" is the whole point of a fan-out. */
789
- evalRunner: EvalRunner;
790
- }
791
- /**
792
- * Build a `CandidateGenerator` for a buildable surface (`tool` / `mcp`). Each
793
- * generation it fans out `fanout` parallel worktree builds, keeps the verified
794
- * ones, ranks them by held-back marginal lift, and emits the single best as one
795
- * artifact. Returns `[]` when no build verifies — the surface had nothing
796
- * shippable to contribute this round (a valid, common outcome for hard builds).
797
- *
798
- * @example Production wiring (build = real harness fan-out, rank = held-back eval):
799
- * buildableGenerator({
800
- * kind: 'mcp',
801
- * buildCandidate: worktreeBuildCandidate({ repoRoot, harness: 'claude' }),
802
- * evalRunner: scoreOnHeldBackSplit,
803
- * fanout: 4,
804
- * })
805
- */
806
- declare function buildableGenerator(opts: BuildableGeneratorOptions): CandidateGenerator<BuildableKind>;
807
-
808
- /**
809
- * `worktreeBuildCandidate` — the PRODUCTION `BuildCandidate`: one fan-out leaf
810
- * that builds a real tool / MCP server in a fresh git worktree with a real
811
- * coding harness, and verifies it by the surface's intrinsic check.
812
- *
813
- * It is the wiring that makes `buildableGenerator`'s dispatch real, composed
814
- * entirely from shipped engines — NO new execution model:
815
- *
816
- * - `gitWorktreeAdapter` (agent-eval/campaign) cuts the isolated worktree.
817
- * - `improvementDriver` (this repo) owns the worktree lifecycle (create →
818
- * generate → finalize/discard) and drives ONE candidate.
819
- * - `agenticGenerator` (this repo) runs the coding harness IN the worktree
820
- * with the surface's build-prompt (`toolBuildPrompt` / `mcpBuildPrompt`):
821
- * research → implement → test, multi-shot resume-on-failure.
822
- * - the surface's verifier proves the result: `commandVerifier('pnpm', ['test'])`
823
- * for a tool (compiles + tests pass), `mcpServeVerifier(serve)` for an MCP
824
- * (boots over stdio + answers `tools/list`). A build that never verifies is
825
- * dropped by `agenticGenerator` (no `CodeSurface` finalized) → `verified:false`.
826
- *
827
- * Kept out of `tool-generator.ts` so the dispatch core stays process-free and
828
- * unit-testable; this file is the one call a consumer makes to wire the real
829
- * harness fan-out.
830
- */
831
-
832
- interface WorktreeBuildOptions {
833
- /** The buildable surface to build (`tool` / `mcp`). */
834
- kind: BuildableKind;
835
- /** Absolute path to the git checkout each candidate worktree is cut from. */
836
- repoRoot: string;
837
- /** Which local coding harness drives the build. Default `claude`. */
838
- harness?: LocalHarness;
839
- /** Base ref each worktree forks from. Default `main`. */
840
- baseRef?: string;
841
- /** Max harness shots per candidate (the depth dial — resume-on-failure). Default 3. */
842
- maxShots?: number;
843
- /** Per-shot wall-clock timeout (ms). Forwarded to `agenticGenerator`. */
844
- timeoutMs?: number;
845
- /**
846
- * For a `tool` build: the verify command (run in the worktree, exit 0 = pass)
847
- * and the tool name the resulting grant lands under. Default command
848
- * `pnpm test`.
849
- */
850
- tool?: {
851
- verifyCommand?: string;
852
- verifyArgs?: string[];
853
- toolName: string;
854
- };
855
- /**
856
- * For an `mcp` build: how to BOOT the built server (the boot-and-probe spec)
857
- * — used both to verify it serves AND as the artifact's start command. The
858
- * `cwd` defaults to the candidate worktree.
859
- */
860
- mcp?: McpServeSpec;
861
- }
862
- /**
863
- * Build the production per-candidate seam for `buildableGenerator`. Each call to
864
- * the returned `BuildCandidate` cuts a fresh worktree, drives the harness to
865
- * implement + verify the surface, and reports the verified worktree (or
866
- * `verified:false` with the reason) back to the dispatch for ranking.
867
- */
868
- declare function worktreeBuildCandidate(opts: WorktreeBuildOptions): BuildCandidate;
869
-
870
- export { ArtifactInput, ArtifactKind, type ArtifactQuery, ArtifactRegistry, ArtifactStatus, type AuthorDiverseSeeds, type BuildCandidate, type BuildableGeneratorOptions, type BuildableKind, type BuiltCandidate, CandidateGenerator, type CandidateOutcome, type ComposeProfileOptions, type DedupeOptions, type DedupeResult, type DistillSkills, type DriftCheck, type DriftWatchOptions, type DriftWatchResult, EvalResult, EvalRunner, GenerateContext, type PairStackCheck, type ProductionPromptGeneratorOptions, ProfileArtifact, PromotionGate, PromotionVerdict, type PromptDraft, type PromptGeneratorOptions, type RefinePrompt, type RefineSkill, type RunLifecycleOptions, type RunLifecycleResult, type SkillDraft, type SkillGeneratorOptions, type WorktreeBuildOptions, applyArtifact, applyArtifacts, buildableGenerator, composeProfile, createArtifactRegistry, dedupeArtifacts, driftWatch, gepaRefine, lifecycleReasonKey, liftMetadataKey, productionPromptGenerator, promptGenerator, routerSeedAuthor, runLifecycle, skillGenerator, worktreeBuildCandidate };