@tangle-network/agent-runtime 0.91.0 → 0.92.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/README.md +4 -2
  2. package/dist/agent.d.ts +3 -3
  3. package/dist/agent.js +106 -25
  4. package/dist/agent.js.map +1 -1
  5. package/dist/{mcp-serve-verifier-XsX8rkB9.d.ts → agentic-generator-B8oeE2Yv.d.ts} +6 -33
  6. package/dist/analyst-loop.d.ts +1 -1
  7. package/dist/candidate-execution/index.d.ts +104 -0
  8. package/dist/candidate-execution/index.js +34 -0
  9. package/dist/candidate-execution/index.js.map +1 -0
  10. package/dist/chunk-3D2RHC4K.js +73 -0
  11. package/dist/chunk-3D2RHC4K.js.map +1 -0
  12. package/dist/chunk-3MDZX7YU.js +125 -0
  13. package/dist/chunk-3MDZX7YU.js.map +1 -0
  14. package/dist/chunk-4FPXIMSI.js +659 -0
  15. package/dist/chunk-4FPXIMSI.js.map +1 -0
  16. package/dist/chunk-6O73TRHW.js +142 -0
  17. package/dist/chunk-6O73TRHW.js.map +1 -0
  18. package/dist/chunk-A62TP7SK.js +4784 -0
  19. package/dist/chunk-A62TP7SK.js.map +1 -0
  20. package/dist/{chunk-AD7JW4QG.js → chunk-BVVRQ4YC.js} +23 -1231
  21. package/dist/chunk-BVVRQ4YC.js.map +1 -0
  22. package/dist/chunk-FDJ7AHXG.js +1229 -0
  23. package/dist/chunk-FDJ7AHXG.js.map +1 -0
  24. package/dist/{chunk-FF77IBQM.js → chunk-FRBHUNQ7.js} +2 -141
  25. package/dist/chunk-FRBHUNQ7.js.map +1 -0
  26. package/dist/chunk-J2K6WIG6.js +2172 -0
  27. package/dist/chunk-J2K6WIG6.js.map +1 -0
  28. package/dist/{chunk-JRS3YSRZ.js → chunk-LQQPGRKT.js} +2 -2
  29. package/dist/{chunk-IOUUITQA.js → chunk-P6B3Z7PR.js} +3 -3
  30. package/dist/chunk-PH65PR4F.js +860 -0
  31. package/dist/chunk-PH65PR4F.js.map +1 -0
  32. package/dist/{chunk-7ON74BQO.js → chunk-R5GWDTM3.js} +2 -2
  33. package/dist/{chunk-DWWII6N2.js → chunk-TKJ3Q6CQ.js} +2 -2
  34. package/dist/{chunk-NC66AM3S.js → chunk-UO2L5VTP.js} +32 -1413
  35. package/dist/chunk-UO2L5VTP.js.map +1 -0
  36. package/dist/{chunk-BZF3KQ6G.js → chunk-VSWBYWFK.js} +4 -122
  37. package/dist/chunk-VSWBYWFK.js.map +1 -0
  38. package/dist/{completion-gate-DkAnUmpb.d.ts → completion-gate-BLaiN0-X.d.ts} +1 -1
  39. package/dist/{coordination-rRj5hjJK.d.ts → coordination-DxJ83oZA.d.ts} +12 -5
  40. package/dist/environment-provider.d.ts +2 -2
  41. package/dist/environment-provider.js +2 -1
  42. package/dist/improve-CUVCq7xg.d.ts +152 -0
  43. package/dist/{improvement-adapter-CDR8QNVM.d.ts → improvement-adapter-BieWeK5J.d.ts} +16 -0
  44. package/dist/index.d.ts +27 -1092
  45. package/dist/index.js +108 -6491
  46. package/dist/index.js.map +1 -1
  47. package/dist/intelligence.d.ts +160 -13
  48. package/dist/intelligence.js +534 -59
  49. package/dist/intelligence.js.map +1 -1
  50. package/dist/knowledge.d.ts +6 -6
  51. package/dist/knowledge.js +6 -4
  52. package/dist/lifecycle.d.ts +2 -1
  53. package/dist/lifecycle.js +5 -3
  54. package/dist/lifecycle.js.map +1 -1
  55. package/dist/{loop-runner-bin-DTbZVGfM.d.ts → loop-runner-bin-kKUNGLyV.d.ts} +2 -2
  56. package/dist/loop-runner-bin.d.ts +5 -5
  57. package/dist/loop-runner-bin.js +8 -5
  58. package/dist/loops.d.ts +16 -16
  59. package/dist/loops.js +47 -41
  60. package/dist/mcp/bin.js +6 -4
  61. package/dist/mcp/bin.js.map +1 -1
  62. package/dist/mcp/index.d.ts +8 -8
  63. package/dist/mcp/index.js +9 -6
  64. package/dist/mcp/index.js.map +1 -1
  65. package/dist/mcp-serve-verifier-Bg4C3p5S.d.ts +34 -0
  66. package/dist/{openai-tools-C4ZfUD4L.d.ts → openai-tools-E3woykz9.d.ts} +1 -1
  67. package/dist/prepare-Z08a4heC.d.ts +713 -0
  68. package/dist/profiles.d.ts +1 -1
  69. package/dist/{router-client-DJImUDlm.d.ts → sanitize-C9go6tXj.d.ts} +113 -1
  70. package/dist/{structural-rollout-MwlpgQ-6.d.ts → structural-rollout-DHGDbhvR.d.ts} +3 -3
  71. package/dist/{supervise-DPmYPk0j.d.ts → supervise-T2pazU3G.d.ts} +4 -4
  72. package/dist/{types-SyuwunY_.d.ts → types-B00NtbCs.d.ts} +1 -1
  73. package/dist/{types-eMNgWgFi.d.ts → types-DAdIm4AC.d.ts} +1 -1
  74. package/dist/{worktree-fanout-BDFQIO-Y.d.ts → worktree-fanout-BUb2Ag02.d.ts} +3 -3
  75. package/package.json +11 -6
  76. package/skills/build-with-agent-runtime/SKILL.md +1 -1
  77. package/dist/chunk-AD7JW4QG.js.map +0 -1
  78. package/dist/chunk-BZF3KQ6G.js.map +0 -1
  79. package/dist/chunk-FF77IBQM.js.map +0 -1
  80. package/dist/chunk-IVGYLCFH.js +0 -381
  81. package/dist/chunk-IVGYLCFH.js.map +0 -1
  82. package/dist/chunk-NC66AM3S.js.map +0 -1
  83. /package/dist/{chunk-JRS3YSRZ.js.map → chunk-LQQPGRKT.js.map} +0 -0
  84. /package/dist/{chunk-IOUUITQA.js.map → chunk-P6B3Z7PR.js.map} +0 -0
  85. /package/dist/{chunk-7ON74BQO.js.map → chunk-R5GWDTM3.js.map} +0 -0
  86. /package/dist/{chunk-DWWII6N2.js.map → chunk-TKJ3Q6CQ.js.map} +0 -0
@@ -0,0 +1,659 @@
1
+ import {
2
+ defaultStructuralRolloutPolicy
3
+ } from "./chunk-FDJ7AHXG.js";
4
+ import {
5
+ agenticGenerator
6
+ } from "./chunk-FRBHUNQ7.js";
7
+ import {
8
+ assertModelAllowed
9
+ } from "./chunk-J2K6WIG6.js";
10
+ import {
11
+ ConfigError
12
+ } from "./chunk-YEJR7IXO.js";
13
+
14
+ // src/improvement/improvement-driver.ts
15
+ function improvementDriver(opts) {
16
+ const baseRef = opts.baseRef ?? "main";
17
+ const finalized = /* @__PURE__ */ new Map();
18
+ return {
19
+ kind: `improvement:${opts.generator.kind}`,
20
+ async propose(ctx) {
21
+ const findings = resolveFindings(ctx);
22
+ if (findings.length === 0 && ctx.report === void 0 && !opts.generator.proposesWithoutFindings) {
23
+ return [];
24
+ }
25
+ const surfaces = [];
26
+ for (let i = 0; i < ctx.populationSize; i++) {
27
+ if (ctx.signal.aborted) break;
28
+ const wt = await opts.worktree.create({
29
+ baseRef,
30
+ label: `${opts.generator.kind}-gen${ctx.generation}-cand${i}`
31
+ });
32
+ try {
33
+ const { applied, summary } = await opts.generator.generate({
34
+ worktreePath: wt.path,
35
+ report: ctx.report,
36
+ findings,
37
+ dataset: ctx.dataset,
38
+ maxShots: ctx.maxImprovementShots ?? 1,
39
+ signal: ctx.signal
40
+ });
41
+ if (!applied) {
42
+ await opts.worktree.discard(wt);
43
+ continue;
44
+ }
45
+ const surface = await opts.worktree.finalize(wt, summary);
46
+ surfaces.push(surface);
47
+ finalized.set(surface.worktreeRef, wt);
48
+ } catch (err) {
49
+ await opts.worktree.discard(wt).catch(() => {
50
+ });
51
+ throw err;
52
+ }
53
+ }
54
+ return surfaces;
55
+ },
56
+ async cleanup(retainWorktreeRefs = []) {
57
+ const retained = new Set(retainWorktreeRefs);
58
+ const errors = [];
59
+ for (const [worktreeRef, worktree] of finalized) {
60
+ if (retained.has(worktreeRef)) continue;
61
+ try {
62
+ await opts.worktree.discard(worktree);
63
+ finalized.delete(worktreeRef);
64
+ } catch (cause) {
65
+ errors.push(cause);
66
+ }
67
+ }
68
+ if (errors.length > 0) {
69
+ throw new AggregateError(errors, "improvementDriver: failed to discard candidate worktrees");
70
+ }
71
+ }
72
+ };
73
+ }
74
+ function resolveFindings(ctx) {
75
+ const report = ctx.report;
76
+ if (report && typeof report === "object" && "findings" in report) {
77
+ const f = report.findings;
78
+ if (Array.isArray(f) && f.length > 0) return f;
79
+ }
80
+ return ctx.findings;
81
+ }
82
+
83
+ // src/improvement/raw-trace-distiller.ts
84
+ import { existsSync, readdirSync } from "fs";
85
+ import { basename, join, resolve } from "path";
86
+ import { makeFinding } from "@tangle-network/agent-eval";
87
+ var ANALYST_ID = "raw-trace-distiller";
88
+ var PASS_THRESHOLD = 0.999;
89
+ function rawTraceDistiller(options = {}) {
90
+ const maxCandidates = options.maxCandidates ?? 12;
91
+ const maxCellsPerCandidate = options.maxCellsPerCandidate ?? 8;
92
+ const maxFilesPerCell = options.maxFilesPerCell ?? 24;
93
+ return async (input) => {
94
+ const genRoot = absoluteRunDir(options.runDir ?? input.runDir);
95
+ const durable = isDurable(genRoot);
96
+ const ranked = [...input.candidates].map((c) => ({
97
+ surfaceHash: c.surfaceHash,
98
+ composite: c.composite,
99
+ campaignDir: absoluteRunDir(c.campaign.runDir),
100
+ cells: failingCells(c.campaign, maxCellsPerCandidate, maxFilesPerCell)
101
+ })).sort((a, b) => a.composite - b.composite).slice(0, maxCandidates);
102
+ const totalFailingCells = ranked.reduce((n, c) => n + c.cells.length, 0);
103
+ if (totalFailingCells === 0) {
104
+ if (options.fallbackFindings && options.fallbackFindings.length > 0) {
105
+ return options.fallbackFindings;
106
+ }
107
+ return [
108
+ makeFinding({
109
+ analyst_id: ANALYST_ID,
110
+ severity: "info",
111
+ area: "raw-trace-context",
112
+ confidence: 1,
113
+ claim: `Generation ${input.generation} had no failing cells. The full raw run traces are on disk under ${genRoot}.`,
114
+ recommended_action: `To keep improving, grep/cat the raw traces under ${genRoot} (per-cell spans.jsonl + cached-result.json) to find the weakest passing runs, then make a targeted harness-code edit.`,
115
+ evidence_refs: [{ kind: "artifact", uri: genRoot }],
116
+ metadata: { generation: input.generation, runDir: genRoot, failingCells: 0 }
117
+ })
118
+ ];
119
+ }
120
+ const findings = [];
121
+ findings.push(
122
+ makeFinding({
123
+ analyst_id: ANALYST_ID,
124
+ severity: "high",
125
+ area: "raw-trace-context",
126
+ confidence: 1,
127
+ claim: `Generation ${input.generation} produced ${totalFailingCells} failing/low-scoring cell(s) across ${ranked.length} candidate(s). Their FULL RAW run traces are on disk under ${genRoot} \u2014 the actual event logs (spans.jsonl), scores (cached-result.json), and artifacts, not a summary.${durable ? "" : " (WARNING: this run root does not exist on disk \u2014 it looks like an in-memory run; pass a real runDir to improve() to get raw-trace context.)"}`,
128
+ recommended_action: `Do NOT rely on a pre-summarized finding. Before editing, DIAGNOSE from the raw traces: run \`grep\`/\`cat\`/\`ls\` over the trace files and directories named in the following findings to see exactly what each failing run did and why it scored low, then make the smallest harness-code edit that fixes the dominant failure. Start with \`grep -rIn "error" ${genRoot}\` then \`cat\` the spans.jsonl of the worst cell.`,
129
+ evidence_refs: [{ kind: "artifact", uri: genRoot }],
130
+ metadata: {
131
+ generation: input.generation,
132
+ runDir: genRoot,
133
+ failingCells: totalFailingCells,
134
+ candidates: ranked.length
135
+ }
136
+ })
137
+ );
138
+ for (const cand of ranked) {
139
+ if (cand.cells.length === 0) continue;
140
+ const scenarioList = cand.cells.map((c) => c.scenarioId).join(", ");
141
+ const fileLines = cand.cells.map((c) => {
142
+ const header = ` cell ${c.scenarioId} (composite ${c.composite.toFixed(3)}${c.error ? `, error: ${truncate(c.error, 160)}` : ""}) \u2014 dir ${c.cellDir}`;
143
+ const files = c.files.map((f) => ` - ${f}`).join("\n");
144
+ const more = c.truncatedFiles ? `
145
+ - \u2026(ls ${c.cellDir} for the rest)` : "";
146
+ return c.files.length > 0 ? `${header}
147
+ ${files}${more}` : header;
148
+ }).join("\n");
149
+ findings.push(
150
+ makeFinding({
151
+ analyst_id: ANALYST_ID,
152
+ severity: cand.composite < 0.5 ? "critical" : "high",
153
+ area: "raw-trace-context",
154
+ confidence: 1,
155
+ subject: cand.surfaceHash,
156
+ claim: `Candidate ${cand.surfaceHash} scored composite ${cand.composite.toFixed(3)} with ${cand.cells.length} failing cell(s) [${scenarioList}]. Its raw traces are under ${cand.campaignDir}.`,
157
+ recommended_action: `grep/cat these raw trace files to diagnose WHY this candidate failed before editing:
158
+ ${fileLines}
159
+ Or scan the whole candidate at once: \`grep -rIn . ${cand.campaignDir}\` and \`ls -R ${cand.campaignDir}\`.`,
160
+ evidence_refs: [
161
+ { kind: "artifact", uri: cand.campaignDir },
162
+ ...cand.cells.flatMap(
163
+ (c) => c.files.map((f) => ({ kind: "artifact", uri: f }))
164
+ )
165
+ ],
166
+ metadata: {
167
+ surfaceHash: cand.surfaceHash,
168
+ composite: cand.composite,
169
+ campaignDir: cand.campaignDir,
170
+ cells: cand.cells.map((c) => ({
171
+ scenarioId: c.scenarioId,
172
+ composite: c.composite,
173
+ cellDir: c.cellDir,
174
+ files: c.files,
175
+ ...c.error ? { error: c.error } : {}
176
+ }))
177
+ }
178
+ })
179
+ );
180
+ }
181
+ return findings;
182
+ };
183
+ }
184
+ function failingCells(campaign, maxCells, maxFiles) {
185
+ const campaignDir = absoluteRunDir(campaign.runDir);
186
+ const durable = isDurable(campaignDir);
187
+ const out = [];
188
+ for (const cell of campaign.cells) {
189
+ const scores = Object.values(cell.judgeScores ?? {});
190
+ const composite = scores.length === 0 ? 0 : scores.reduce((sum, s) => sum + (s.composite ?? 0), 0) / scores.length;
191
+ if (!cell.error && composite >= PASS_THRESHOLD) continue;
192
+ const cellDir = join(campaignDir, sanitizeCellId(cell.cellId));
193
+ const artifactPaths = artifactPathsForCell(campaign.artifactsByPath, cell.cellId);
194
+ const discovered = durable ? listTraceFiles(cellDir) : [];
195
+ const canonical = [join(cellDir, "spans.jsonl"), join(cellDir, "cached-result.json")];
196
+ const files = dedupeSorted([...discovered, ...artifactPaths, ...canonical]);
197
+ out.push({
198
+ scenarioId: cell.scenarioId,
199
+ composite: Number(composite.toFixed(3)),
200
+ ...cell.error ? { error: cell.error } : {},
201
+ cellDir,
202
+ files: files.slice(0, maxFiles),
203
+ truncatedFiles: files.length > maxFiles
204
+ });
205
+ if (out.length >= maxCells) break;
206
+ }
207
+ return out;
208
+ }
209
+ function artifactPathsForCell(artifactsByPath, cellId) {
210
+ if (!artifactsByPath) return [];
211
+ const prefix = `${cellId}/`;
212
+ return Object.entries(artifactsByPath).filter(([key]) => key.startsWith(prefix)).map(([, absPath]) => resolve(absPath));
213
+ }
214
+ function listTraceFiles(dir) {
215
+ const out = [];
216
+ for (const entry of safeReadDir(dir)) {
217
+ const full = join(dir, entry.name);
218
+ if (entry.isFile()) {
219
+ out.push(full);
220
+ } else if (!entry.isSymbolicLink() && entry.isDirectory()) {
221
+ for (const sub of safeReadDir(full)) {
222
+ if (sub.isFile()) out.push(join(full, sub.name));
223
+ }
224
+ }
225
+ }
226
+ return out;
227
+ }
228
+ function safeReadDir(dir) {
229
+ try {
230
+ return readdirSync(dir, { withFileTypes: true });
231
+ } catch {
232
+ return [];
233
+ }
234
+ }
235
+ function sanitizeCellId(cellId) {
236
+ return cellId.replace(/[^a-zA-Z0-9_-]/g, "_");
237
+ }
238
+ function isDurable(runDir) {
239
+ return !runDir.startsWith("mem://") && existsSync(runDir);
240
+ }
241
+ function absoluteRunDir(runDir) {
242
+ return runDir.startsWith("mem://") ? runDir : resolve(runDir);
243
+ }
244
+ function dedupeSorted(paths) {
245
+ return [...new Set(paths)].sort((a, b) => {
246
+ const da = a.slice(0, a.length - basename(a).length);
247
+ const db = b.slice(0, b.length - basename(b).length);
248
+ return da === db ? basename(a).localeCompare(basename(b)) : da.localeCompare(db);
249
+ });
250
+ }
251
+ function truncate(s, n) {
252
+ return s.length <= n ? s : `${s.slice(0, n - 1)}\u2026`;
253
+ }
254
+
255
+ // src/improvement/rollout-policy.ts
256
+ var ROLLOUT_POLICY_EXTENSION = "structural-rollout";
257
+ var ROLLOUT_POLICY_BOUNDS = {
258
+ k: { min: 1, max: 10, step: 2 },
259
+ repairRounds: { min: 0, max: 3, step: 1 },
260
+ testgen: { min: 0, max: 10, step: 3 }
261
+ };
262
+ var MAX_CANDIDATES_PER_GENERATION = 4;
263
+ var clamp = (v, min, max) => Math.min(max, Math.max(min, v));
264
+ var isBoundedInt = (v, min) => typeof v === "number" && Number.isInteger(v) && v >= min;
265
+ function parseRolloutPolicy(surface) {
266
+ if (typeof surface !== "string" || surface.trim().length === 0) return void 0;
267
+ let raw;
268
+ try {
269
+ raw = JSON.parse(surface);
270
+ } catch {
271
+ return void 0;
272
+ }
273
+ return normalizeRolloutPolicy(raw);
274
+ }
275
+ function normalizeRolloutPolicy(raw) {
276
+ if (typeof raw !== "object" || raw === null || Array.isArray(raw)) return void 0;
277
+ const bag = raw;
278
+ const k = bag.k ?? defaultStructuralRolloutPolicy.k;
279
+ const repairRounds = bag.repairRounds ?? defaultStructuralRolloutPolicy.repairRounds;
280
+ const testgen = bag.testgen ?? defaultStructuralRolloutPolicy.testgen;
281
+ if (!isBoundedInt(k, 1) || !isBoundedInt(repairRounds, 0) || !isBoundedInt(testgen, 0)) {
282
+ return void 0;
283
+ }
284
+ return {
285
+ k,
286
+ repairRounds,
287
+ testgen,
288
+ ...typeof bag.diverse === "boolean" ? { diverse: bag.diverse } : {},
289
+ ...typeof bag.temperature === "number" ? { temperature: bag.temperature } : {}
290
+ };
291
+ }
292
+ function serializeRolloutPolicy(policy) {
293
+ return JSON.stringify({
294
+ k: policy.k,
295
+ repairRounds: policy.repairRounds,
296
+ testgen: policy.testgen,
297
+ ...policy.diverse !== void 0 ? { diverse: policy.diverse } : {},
298
+ ...policy.temperature !== void 0 ? { temperature: policy.temperature } : {}
299
+ });
300
+ }
301
+ function structuralRolloutPolicyFromProfile(profile) {
302
+ const bag = profile.extensions?.[ROLLOUT_POLICY_EXTENSION];
303
+ if (bag === void 0) return void 0;
304
+ return normalizeRolloutPolicy(bag);
305
+ }
306
+ function applyRolloutPolicyToProfile(profile, policy) {
307
+ const bag = {
308
+ k: policy.k,
309
+ repairRounds: policy.repairRounds,
310
+ testgen: policy.testgen,
311
+ ...policy.diverse !== void 0 ? { diverse: policy.diverse } : {},
312
+ ...policy.temperature !== void 0 ? { temperature: policy.temperature } : {}
313
+ };
314
+ return {
315
+ ...profile,
316
+ extensions: { ...profile.extensions, [ROLLOUT_POLICY_EXTENSION]: bag }
317
+ };
318
+ }
319
+ function enumerateNeighborPolicies(policy) {
320
+ const moves = [
321
+ { dial: "k", delta: 1 },
322
+ { dial: "k", delta: -1 },
323
+ { dial: "repairRounds", delta: 1 },
324
+ { dial: "repairRounds", delta: -1 },
325
+ { dial: "testgen", delta: 1 },
326
+ { dial: "testgen", delta: -1 }
327
+ ];
328
+ const seen = /* @__PURE__ */ new Set([serializeRolloutPolicy(policy)]);
329
+ const neighbors = [];
330
+ for (const move of moves) {
331
+ const bounds = ROLLOUT_POLICY_BOUNDS[move.dial];
332
+ const next = clamp(policy[move.dial] + move.delta * bounds.step, bounds.min, bounds.max);
333
+ const candidate = { ...policy, [move.dial]: next };
334
+ const key = serializeRolloutPolicy(candidate);
335
+ if (seen.has(key)) continue;
336
+ seen.add(key);
337
+ neighbors.push(candidate);
338
+ }
339
+ return neighbors;
340
+ }
341
+ function candidateLabel(base, next) {
342
+ for (const dial of ["k", "repairRounds", "testgen"]) {
343
+ if (next[dial] !== base[dial]) return `${dial} ${base[dial]}\u2192${next[dial]}`;
344
+ }
345
+ return "unchanged";
346
+ }
347
+ function rolloutPolicyProposer() {
348
+ return {
349
+ kind: "rollout-policy",
350
+ async propose(ctx) {
351
+ const policy = parseRolloutPolicy(ctx.currentSurface);
352
+ if (!policy) return [];
353
+ const neighbors = enumerateNeighborPolicies(policy);
354
+ if (neighbors.length === 0) return [];
355
+ const cap = Math.max(1, Math.min(ctx.populationSize, MAX_CANDIDATES_PER_GENERATION));
356
+ const start = ctx.generation * cap % neighbors.length;
357
+ const window = [];
358
+ for (let i = 0; i < Math.min(cap, neighbors.length); i += 1) {
359
+ window.push(neighbors[(start + i) % neighbors.length]);
360
+ }
361
+ return window.map((candidate) => ({
362
+ surface: serializeRolloutPolicy(candidate),
363
+ label: candidateLabel(policy, candidate),
364
+ rationale: "bounded single-dial neighbor of the current structuralRollout policy; the held-out gate decides (deterministic enumeration \u2014 the dial space is tiny and prompt-style reflective proposals are a measured zero here)"
365
+ }));
366
+ }
367
+ };
368
+ }
369
+
370
+ // src/improvement/improve.ts
371
+ import {
372
+ gepaProposer,
373
+ gitWorktreeAdapter,
374
+ skillOptProposer
375
+ } from "@tangle-network/agent-eval/campaign";
376
+ import {
377
+ selfImprove
378
+ } from "@tangle-network/agent-eval/contract";
379
+ import { agentProfileSchema } from "@tangle-network/agent-interface";
380
+ var defaultReflectionModel = "deepseek-v4-flash";
381
+ var workflowExtension = "tangle.workflow";
382
+ function llmClientOptions(llm) {
383
+ return { baseUrl: llm?.baseUrl, apiKey: llm?.apiKey };
384
+ }
385
+ function defaultGeneratorFor(surface, llm) {
386
+ const model = llm?.model ?? defaultReflectionModel;
387
+ switch (surface) {
388
+ case "prompt":
389
+ return gepaProposer({ llm: llmClientOptions(llm), model, target: "agent system prompt" });
390
+ case "skills":
391
+ return skillOptProposer({ llm: llmClientOptions(llm), model, target: "agent skill document" });
392
+ case "rollout-policy":
393
+ return rolloutPolicyProposer();
394
+ default:
395
+ return void 0;
396
+ }
397
+ }
398
+ function baselineSurfaceFor(profile, surface, skills) {
399
+ switch (surface) {
400
+ case "prompt":
401
+ return profile.prompt?.systemPrompt ?? "";
402
+ case "skills":
403
+ return skills?.document ?? JSON.stringify(profile.resources?.skills ?? []);
404
+ case "tools":
405
+ return JSON.stringify(profile.tools ?? {});
406
+ case "mcp":
407
+ return JSON.stringify(profile.mcp ?? {});
408
+ case "hooks":
409
+ return JSON.stringify(profile.hooks ?? {});
410
+ case "subagents":
411
+ return JSON.stringify(profile.subagents ?? {});
412
+ case "workflow":
413
+ return JSON.stringify(profile.extensions?.[workflowExtension] ?? {});
414
+ case "agent-profile":
415
+ return JSON.stringify(profile);
416
+ case "rollout-policy": {
417
+ const policy = structuralRolloutPolicyFromProfile(profile);
418
+ return policy ? serializeRolloutPolicy(policy) : "";
419
+ }
420
+ case "code":
421
+ throw new ConfigError(
422
+ "improve(): code requires the isolated baseline created from opts.code.repoRoot"
423
+ );
424
+ }
425
+ }
426
+ function generationFailureDistiller(staticFindings) {
427
+ const CAP = 12;
428
+ return async (input) => {
429
+ const failures = [];
430
+ for (const candidate of input.candidates) {
431
+ for (const rawCell of candidate.campaign.cells) {
432
+ const cell = rawCell;
433
+ const scenario = String(cell.scenarioId ?? "unknown");
434
+ const error = typeof cell.error === "string" ? cell.error : void 0;
435
+ const judgeScores = cell.judgeScores && typeof cell.judgeScores === "object" ? Object.values(
436
+ cell.judgeScores
437
+ ) : [];
438
+ const composite = judgeScores.length === 0 ? 0 : judgeScores.reduce((sum, j) => sum + (j.composite ?? 0), 0) / judgeScores.length;
439
+ if (!error && composite >= 0.999) continue;
440
+ const notes = judgeScores.map((j) => j.notes).filter((n) => typeof n === "string" && n.length > 0).join("; ").slice(0, 400);
441
+ failures.push({
442
+ scenario,
443
+ composite: Number(composite.toFixed(3)),
444
+ notes,
445
+ ...error ? { error: error.slice(0, 200) } : {}
446
+ });
447
+ }
448
+ }
449
+ if (failures.length === 0) return staticFindings;
450
+ failures.sort((a, b) => a.composite - b.composite);
451
+ return failures.slice(0, CAP);
452
+ };
453
+ }
454
+ function isCodeSurface(surface) {
455
+ return typeof surface === "object" && surface !== null && surface.kind === "code";
456
+ }
457
+ async function prepareCodeRun(code, proposerOverride) {
458
+ const baseRef = code.baseRef ?? "main";
459
+ const worktree = gitWorktreeAdapter({
460
+ repoRoot: code.repoRoot,
461
+ ...code.worktreeDir ? { worktreeDir: code.worktreeDir } : {}
462
+ });
463
+ const baselineWorktree = await worktree.create({ baseRef, label: "incumbent-baseline" });
464
+ const baseline = await worktree.finalize(baselineWorktree, "Incumbent code checkout");
465
+ let managed;
466
+ if (!proposerOverride) {
467
+ const generator = code.generator ?? agenticGenerator({
468
+ ...code.harness ? { harness: code.harness } : {},
469
+ ...code.verify ? { verify: code.verify } : {},
470
+ ...code.timeoutMs ? { timeoutMs: code.timeoutMs } : {}
471
+ });
472
+ managed = improvementDriver({ worktree, generator, baseRef });
473
+ }
474
+ const proposer = proposerOverride ?? managed;
475
+ if (!proposer) {
476
+ throw new ConfigError("improve(): code candidate generator could not be constructed");
477
+ }
478
+ return {
479
+ baseline,
480
+ proposer,
481
+ async cleanup(retainedWinner) {
482
+ const errors = [];
483
+ try {
484
+ await managed?.cleanup(isCodeSurface(retainedWinner) ? [retainedWinner.worktreeRef] : []);
485
+ } catch (cause) {
486
+ errors.push(cause);
487
+ }
488
+ try {
489
+ await worktree.discard(baselineWorktree);
490
+ } catch (cause) {
491
+ errors.push(cause);
492
+ }
493
+ if (errors.length > 0) {
494
+ throw new AggregateError(errors, "improve(): failed to clean code improvement worktrees");
495
+ }
496
+ }
497
+ };
498
+ }
499
+ function parseWinnerJson(winner, surface) {
500
+ try {
501
+ return JSON.parse(winner);
502
+ } catch (cause) {
503
+ throw new ConfigError(
504
+ `improve(): the shipped '${surface}' winner is not valid JSON, so it cannot be applied back to the profile: ${cause.message}`
505
+ );
506
+ }
507
+ }
508
+ function applyWinnerToProfile(profile, surface, winner) {
509
+ if (typeof winner !== "string") return profile;
510
+ let candidate;
511
+ switch (surface) {
512
+ case "prompt":
513
+ candidate = { ...profile, prompt: { ...profile.prompt, systemPrompt: winner } };
514
+ break;
515
+ case "skills":
516
+ candidate = {
517
+ ...profile,
518
+ resources: { ...profile.resources, skills: parseWinnerJson(winner, surface) }
519
+ };
520
+ break;
521
+ case "tools":
522
+ candidate = { ...profile, tools: parseWinnerJson(winner, surface) };
523
+ break;
524
+ case "mcp":
525
+ candidate = { ...profile, mcp: parseWinnerJson(winner, surface) };
526
+ break;
527
+ case "hooks":
528
+ candidate = { ...profile, hooks: parseWinnerJson(winner, surface) };
529
+ break;
530
+ case "subagents":
531
+ candidate = { ...profile, subagents: parseWinnerJson(winner, surface) };
532
+ break;
533
+ case "workflow":
534
+ candidate = {
535
+ ...profile,
536
+ extensions: {
537
+ ...profile.extensions,
538
+ [workflowExtension]: parseWinnerJson(winner, surface)
539
+ }
540
+ };
541
+ break;
542
+ case "agent-profile":
543
+ candidate = parseWinnerJson(winner, surface);
544
+ break;
545
+ case "rollout-policy": {
546
+ const policy = normalizeRolloutPolicy(parseWinnerJson(winner, surface));
547
+ if (!policy) {
548
+ throw new ConfigError(
549
+ `improve(): the shipped 'rollout-policy' winner is not a valid StructuralRolloutPolicy (integer k >= 1, repairRounds >= 0, testgen >= 0), so it cannot be applied: ${winner}`
550
+ );
551
+ }
552
+ candidate = applyRolloutPolicyToProfile(profile, policy);
553
+ break;
554
+ }
555
+ case "code":
556
+ return profile;
557
+ }
558
+ const parsed = agentProfileSchema.safeParse(candidate);
559
+ if (!parsed.success) {
560
+ throw new ConfigError(
561
+ `improve(): the shipped '${surface}' winner does not produce a valid AgentProfile: ${parsed.error.message}`
562
+ );
563
+ }
564
+ return parsed.data;
565
+ }
566
+ async function improve(profile, findings, opts) {
567
+ const {
568
+ surface = "prompt",
569
+ gate = "holdout",
570
+ generator,
571
+ allowedModels,
572
+ rawTraceContext,
573
+ code,
574
+ skills,
575
+ promotionGate,
576
+ analyzeGeneration,
577
+ ...sharedOptions
578
+ } = opts;
579
+ const parsedProfile = agentProfileSchema.safeParse(profile);
580
+ if (!parsedProfile.success) {
581
+ throw new ConfigError(
582
+ `improve(): input is not a valid AgentProfile: ${parsedProfile.error.message}`
583
+ );
584
+ }
585
+ if (surface === "skills" && !generator && !skills) {
586
+ throw new ConfigError(
587
+ "improve(): the default skills optimizer requires opts.skills.document; pass the skill text or an explicit generator that understands resource refs"
588
+ );
589
+ }
590
+ const usesReflectionModel = !generator && (surface === "prompt" || surface === "skills");
591
+ if (usesReflectionModel) {
592
+ assertModelAllowed(sharedOptions.llm?.model ?? defaultReflectionModel, allowedModels);
593
+ }
594
+ let preparedCode;
595
+ if (surface === "code") {
596
+ if (!code) {
597
+ throw new ConfigError(
598
+ "improve(): surface 'code' requires opts.code.repoRoot so the incumbent can run from an isolated checkout"
599
+ );
600
+ }
601
+ preparedCode = await prepareCodeRun(code, generator);
602
+ }
603
+ const proposer = preparedCode?.proposer ?? generator ?? defaultGeneratorFor(surface, sharedOptions.llm);
604
+ if (!proposer) {
605
+ throw new ConfigError(
606
+ `improve(): surface '${surface}' has no default generator \u2014 pass opts.generator (a SurfaceProposer) explicitly`
607
+ );
608
+ }
609
+ const budget = gate === "none" ? { ...sharedOptions.budget, generations: 0 } : { ...sharedOptions.budget };
610
+ let raw;
611
+ try {
612
+ raw = await selfImprove({
613
+ ...sharedOptions,
614
+ baselineSurface: preparedCode?.baseline ?? baselineSurfaceFor(profile, surface, skills),
615
+ proposer,
616
+ budget,
617
+ findings,
618
+ ...promotionGate !== void 0 ? { gate: promotionGate } : {},
619
+ ...analyzeGeneration === null ? {} : {
620
+ analyzeGeneration: analyzeGeneration ?? (rawTraceContext ? rawTraceDistiller({ fallbackFindings: findings }) : generationFailureDistiller(findings))
621
+ }
622
+ });
623
+ } catch (cause) {
624
+ if (!preparedCode) throw cause;
625
+ try {
626
+ await preparedCode.cleanup();
627
+ } catch (cleanupCause) {
628
+ throw new AggregateError(
629
+ [cause, cleanupCause],
630
+ "improve(): code improvement failed and its worktrees could not be cleaned"
631
+ );
632
+ }
633
+ throw cause;
634
+ }
635
+ const shipped = raw.gateDecision === "ship";
636
+ await preparedCode?.cleanup(shipped ? raw.winner.surface : void 0);
637
+ const usedSkillDocument = surface === "skills" && skills !== void 0;
638
+ if (shipped && usedSkillDocument && typeof raw.winner.surface === "string") {
639
+ skills?.writeBack?.(raw.winner.surface);
640
+ }
641
+ const nextProfile = shipped && !usedSkillDocument ? applyWinnerToProfile(profile, surface, raw.winner.surface) : profile;
642
+ return { profile: nextProfile, shipped, lift: raw.lift, gateDecision: raw.gateDecision, raw };
643
+ }
644
+
645
+ export {
646
+ improvementDriver,
647
+ rawTraceDistiller,
648
+ ROLLOUT_POLICY_EXTENSION,
649
+ ROLLOUT_POLICY_BOUNDS,
650
+ parseRolloutPolicy,
651
+ normalizeRolloutPolicy,
652
+ serializeRolloutPolicy,
653
+ structuralRolloutPolicyFromProfile,
654
+ applyRolloutPolicyToProfile,
655
+ enumerateNeighborPolicies,
656
+ rolloutPolicyProposer,
657
+ improve
658
+ };
659
+ //# sourceMappingURL=chunk-4FPXIMSI.js.map