@tangle-network/agent-runtime 0.95.0 → 0.96.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +64 -15
- package/dist/activation-B0ZD7nfX.d.ts +63 -0
- package/dist/agent.d.ts +5 -169
- package/dist/agent.js +8 -229
- package/dist/agent.js.map +1 -1
- package/dist/analyst-loop.d.ts +6 -9
- package/dist/analyst-loop.js +1 -2
- package/dist/candidate-execution/index.js +6 -7
- package/dist/chunk-3XKSBI2U.js +474 -0
- package/dist/chunk-3XKSBI2U.js.map +1 -0
- package/dist/{chunk-6YBA64Z2.js → chunk-6XKPVJAZ.js} +5 -18
- package/dist/chunk-6XKPVJAZ.js.map +1 -0
- package/dist/{chunk-MKGRLDWB.js → chunk-BLQIYRVR.js} +17 -2
- package/dist/chunk-BLQIYRVR.js.map +1 -0
- package/dist/{chunk-QDSOD7RC.js → chunk-FD2MBMOH.js} +13 -101
- package/dist/chunk-FD2MBMOH.js.map +1 -0
- package/dist/{chunk-YLUOTX6U.js → chunk-FXF2OL34.js} +7 -7
- package/dist/{chunk-EP6RVHMX.js → chunk-HZDEXTSL.js} +848 -2
- package/dist/chunk-HZDEXTSL.js.map +1 -0
- package/dist/chunk-PSOCBNM3.js +2069 -0
- package/dist/chunk-PSOCBNM3.js.map +1 -0
- package/dist/{chunk-BPGXIKK7.js → chunk-SGQ4YIQW.js} +4 -4
- package/dist/{chunk-IADLKE7I.js → chunk-UQ6PNNXM.js} +5 -7
- package/dist/{chunk-IADLKE7I.js.map → chunk-UQ6PNNXM.js.map} +1 -1
- package/dist/{chunk-Z5I642SY.js → chunk-WYC2XJF2.js} +2 -2
- package/dist/{chunk-ZEYAT33L.js → chunk-Y3SRWZMP.js} +2 -2
- package/dist/{chunk-WTZ37EQY.js → chunk-YOLKCWRV.js} +197 -90
- package/dist/chunk-YOLKCWRV.js.map +1 -0
- package/dist/conversation.js +0 -1
- package/dist/environment-provider.js +0 -1
- package/dist/{agentic-generator-hCaQRAes.d.ts → improve-g75IE2Cx.d.ts} +152 -3
- package/dist/{improvement-adapter-BieWeK5J.d.ts → improvement-adapter-HAZz-7vK.d.ts} +8 -31
- package/dist/index.d.ts +41 -11
- package/dist/index.js +180 -51
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +16 -9
- package/dist/intelligence.js +13 -8
- package/dist/intelligence.js.map +1 -1
- package/dist/knowledge.d.ts +30 -13
- package/dist/knowledge.js +11 -10
- package/dist/{loop-runner-bin-BIQldFS8.d.ts → loop-runner-bin-Cn1N2rRo.d.ts} +1 -1
- package/dist/loop-runner-bin.d.ts +2 -2
- package/dist/loop-runner-bin.js +6 -8
- package/dist/loops.d.ts +1 -1
- package/dist/loops.js +4 -6
- package/dist/mcp/bin.js +3 -5
- package/dist/mcp/bin.js.map +1 -1
- package/dist/mcp/index.js +10 -12
- package/dist/mcp/index.js.map +1 -1
- package/dist/platform.js +0 -2
- package/dist/platform.js.map +1 -1
- package/dist/primeintellect/index.js +0 -1
- package/dist/primeintellect/index.js.map +1 -1
- package/dist/profiles.js +0 -1
- package/dist/profiles.js.map +1 -1
- package/dist/{types-BC3bZpH0.d.ts → types-CmYCMbFT.d.ts} +12 -54
- package/package.json +7 -12
- package/skills/build-with-agent-runtime/SKILL.md +122 -213
- package/dist/chunk-6O73TRHW.js +0 -142
- package/dist/chunk-6O73TRHW.js.map +0 -1
- package/dist/chunk-6YBA64Z2.js.map +0 -1
- package/dist/chunk-AP7CPGMZ.js +0 -334
- package/dist/chunk-AP7CPGMZ.js.map +0 -1
- package/dist/chunk-DGUM43GV.js +0 -11
- package/dist/chunk-DGUM43GV.js.map +0 -1
- package/dist/chunk-DHCHL6OG.js +0 -625
- package/dist/chunk-DHCHL6OG.js.map +0 -1
- package/dist/chunk-EP6RVHMX.js.map +0 -1
- package/dist/chunk-G55QE4IQ.js +0 -1137
- package/dist/chunk-G55QE4IQ.js.map +0 -1
- package/dist/chunk-ISTDY47H.js +0 -849
- package/dist/chunk-ISTDY47H.js.map +0 -1
- package/dist/chunk-MKGRLDWB.js.map +0 -1
- package/dist/chunk-QDSOD7RC.js.map +0 -1
- package/dist/chunk-WTZ37EQY.js.map +0 -1
- package/dist/generator-YkAQrOoD.d.ts +0 -382
- package/dist/improve-B-UYaEH5.d.ts +0 -172
- package/dist/lifecycle.d.ts +0 -870
- package/dist/lifecycle.js +0 -981
- package/dist/lifecycle.js.map +0 -1
- package/dist/mcp-serve-verifier-Bs_n0xPc.d.ts +0 -34
- package/skills/agent-runtime-adoption/SKILL.md +0 -246
- /package/dist/{chunk-YLUOTX6U.js.map → chunk-FXF2OL34.js.map} +0 -0
- /package/dist/{chunk-BPGXIKK7.js.map → chunk-SGQ4YIQW.js.map} +0 -0
- /package/dist/{chunk-Z5I642SY.js.map → chunk-WYC2XJF2.js.map} +0 -0
- /package/dist/{chunk-ZEYAT33L.js.map → chunk-Y3SRWZMP.js.map} +0 -0
package/dist/chunk-G55QE4IQ.js
DELETED
|
@@ -1,1137 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
agentCandidateProfileAsAgentProfile,
|
|
3
|
-
candidateMaterializerHarness,
|
|
4
|
-
canonicalCandidateBytes,
|
|
5
|
-
canonicalCandidateDigest,
|
|
6
|
-
canonicalCandidateDocument,
|
|
7
|
-
createAgentCandidateProfileActivation,
|
|
8
|
-
executePreparedAgentCandidate,
|
|
9
|
-
immutableCandidateValue,
|
|
10
|
-
omitTopLevelDigest,
|
|
11
|
-
parseAgentCandidateProfileActivation,
|
|
12
|
-
prepareAgentCandidateExecution,
|
|
13
|
-
verifiedResourceTextByDigest,
|
|
14
|
-
verifyAgentCandidateBundle,
|
|
15
|
-
verifyCanonicalCandidateDocument
|
|
16
|
-
} from "./chunk-WTZ37EQY.js";
|
|
17
|
-
import {
|
|
18
|
-
assertModelAllowed
|
|
19
|
-
} from "./chunk-ISTDY47H.js";
|
|
20
|
-
import {
|
|
21
|
-
runAnalystLoop
|
|
22
|
-
} from "./chunk-QDSOD7RC.js";
|
|
23
|
-
import {
|
|
24
|
-
agenticGenerator
|
|
25
|
-
} from "./chunk-DHCHL6OG.js";
|
|
26
|
-
import {
|
|
27
|
-
ConfigError
|
|
28
|
-
} from "./chunk-YEJR7IXO.js";
|
|
29
|
-
|
|
30
|
-
// src/improvement/improvement-driver.ts
|
|
31
|
-
import { spawnSync } from "child_process";
|
|
32
|
-
import {
|
|
33
|
-
verifyCodeSurface
|
|
34
|
-
} from "@tangle-network/agent-eval/campaign";
|
|
35
|
-
|
|
36
|
-
// src/improvement/cleanup.ts
|
|
37
|
-
async function rethrowAfterCleanup(cause, cleanup, context) {
|
|
38
|
-
const cleanupErrors = [];
|
|
39
|
-
for (let attempt = 0; attempt < 2; attempt += 1) {
|
|
40
|
-
try {
|
|
41
|
-
await cleanup();
|
|
42
|
-
} catch (cleanupCause) {
|
|
43
|
-
cleanupErrors.push(cleanupCause);
|
|
44
|
-
continue;
|
|
45
|
-
}
|
|
46
|
-
if (cleanupErrors.length === 0) throw cause;
|
|
47
|
-
throw new AggregateError([cause, ...cleanupErrors], `${context}; cleanup retry succeeded`);
|
|
48
|
-
}
|
|
49
|
-
throw new AggregateError([cause, ...cleanupErrors], `${context}; cleanup failed`);
|
|
50
|
-
}
|
|
51
|
-
|
|
52
|
-
// src/improvement/improvement-driver.ts
|
|
53
|
-
function improvementDriver(opts) {
|
|
54
|
-
const baseRef = opts.baseRef ?? "main";
|
|
55
|
-
const owned = /* @__PURE__ */ new Map();
|
|
56
|
-
return {
|
|
57
|
-
kind: `improvement:${opts.generator.kind}`,
|
|
58
|
-
async propose(ctx) {
|
|
59
|
-
const findings = resolveFindings(ctx);
|
|
60
|
-
if (findings.length === 0 && ctx.report === void 0 && !opts.generator.proposesWithoutFindings) {
|
|
61
|
-
return [];
|
|
62
|
-
}
|
|
63
|
-
const surfaces = [];
|
|
64
|
-
const incumbent = verifiedCodeIncumbent(ctx.currentSurface);
|
|
65
|
-
const proposalBaseRef = incumbent?.baseCommit ?? baseRef;
|
|
66
|
-
for (let i = 0; i < ctx.populationSize; i++) {
|
|
67
|
-
if (ctx.signal.aborted) break;
|
|
68
|
-
const wt = await opts.worktree.create({
|
|
69
|
-
baseRef: proposalBaseRef,
|
|
70
|
-
label: `${opts.generator.kind}-gen${ctx.generation}-cand${i}`
|
|
71
|
-
});
|
|
72
|
-
owned.set(wt.path, wt);
|
|
73
|
-
try {
|
|
74
|
-
if (incumbent) advanceToIncumbent(wt, incumbent);
|
|
75
|
-
const { applied, summary } = await opts.generator.generate({
|
|
76
|
-
worktreePath: wt.path,
|
|
77
|
-
report: ctx.report,
|
|
78
|
-
findings,
|
|
79
|
-
dataset: ctx.dataset,
|
|
80
|
-
maxShots: ctx.maxImprovementShots ?? 1,
|
|
81
|
-
signal: ctx.signal,
|
|
82
|
-
generation: ctx.generation,
|
|
83
|
-
candidateIndex: i,
|
|
84
|
-
...ctx.costLedger ? { costLedger: ctx.costLedger } : {},
|
|
85
|
-
...ctx.costPhase ? { costPhase: ctx.costPhase } : {}
|
|
86
|
-
});
|
|
87
|
-
if (!applied) {
|
|
88
|
-
await opts.worktree.discard(wt);
|
|
89
|
-
owned.delete(wt.path);
|
|
90
|
-
continue;
|
|
91
|
-
}
|
|
92
|
-
const surface = await opts.worktree.finalize(wt, summary);
|
|
93
|
-
surfaces.push(surface);
|
|
94
|
-
owned.delete(wt.path);
|
|
95
|
-
owned.set(surface.worktreeRef, wt);
|
|
96
|
-
} catch (err) {
|
|
97
|
-
const failure = err instanceof Error ? err.message : String(err);
|
|
98
|
-
return rethrowAfterCleanup(
|
|
99
|
-
err,
|
|
100
|
-
async () => {
|
|
101
|
-
await opts.worktree.discard(wt);
|
|
102
|
-
owned.delete(wt.path);
|
|
103
|
-
},
|
|
104
|
-
`improvementDriver: ${failure}`
|
|
105
|
-
);
|
|
106
|
-
}
|
|
107
|
-
}
|
|
108
|
-
return surfaces;
|
|
109
|
-
},
|
|
110
|
-
async cleanup(retainWorktreeRefs = []) {
|
|
111
|
-
const retained = new Set(retainWorktreeRefs);
|
|
112
|
-
const errors = [];
|
|
113
|
-
for (const [worktreeRef, worktree] of owned) {
|
|
114
|
-
if (retained.has(worktreeRef)) continue;
|
|
115
|
-
try {
|
|
116
|
-
await opts.worktree.discard(worktree);
|
|
117
|
-
owned.delete(worktreeRef);
|
|
118
|
-
} catch (cause) {
|
|
119
|
-
errors.push(cause);
|
|
120
|
-
}
|
|
121
|
-
}
|
|
122
|
-
if (errors.length > 0) {
|
|
123
|
-
throw new AggregateError(errors, "improvementDriver: failed to discard candidate worktrees");
|
|
124
|
-
}
|
|
125
|
-
}
|
|
126
|
-
};
|
|
127
|
-
}
|
|
128
|
-
function verifiedCodeIncumbent(surface) {
|
|
129
|
-
if (typeof surface !== "object" || surface.kind !== "code") return void 0;
|
|
130
|
-
verifyCodeSurface(surface);
|
|
131
|
-
return surface;
|
|
132
|
-
}
|
|
133
|
-
function advanceToIncumbent(worktree, incumbent) {
|
|
134
|
-
if (worktree.baseCommit !== incumbent.baseCommit || worktree.baseTree !== incumbent.baseTree) {
|
|
135
|
-
throw new Error("improvementDriver: candidate worktree does not match incumbent base identity");
|
|
136
|
-
}
|
|
137
|
-
if (worktree.baseCommit === incumbent.candidateCommit) return;
|
|
138
|
-
const merge = spawnSync("git", ["merge", "--ff-only", incumbent.candidateCommit], {
|
|
139
|
-
cwd: worktree.path,
|
|
140
|
-
encoding: "utf8"
|
|
141
|
-
});
|
|
142
|
-
if (merge.error) {
|
|
143
|
-
throw new Error(
|
|
144
|
-
`improvementDriver: failed to start candidate from incumbent: ${merge.error.message}`
|
|
145
|
-
);
|
|
146
|
-
}
|
|
147
|
-
if (merge.status !== 0) {
|
|
148
|
-
throw new Error(
|
|
149
|
-
`improvementDriver: could not fast-forward candidate to incumbent ${incumbent.candidateCommit}: ${merge.stderr.trim()}`
|
|
150
|
-
);
|
|
151
|
-
}
|
|
152
|
-
const head = spawnSync("git", ["rev-parse", "--verify", "HEAD"], {
|
|
153
|
-
cwd: worktree.path,
|
|
154
|
-
encoding: "utf8"
|
|
155
|
-
});
|
|
156
|
-
if (head.error || head.status !== 0 || head.stdout.trim() !== incumbent.candidateCommit) {
|
|
157
|
-
throw new Error("improvementDriver: candidate worktree did not reach the incumbent commit");
|
|
158
|
-
}
|
|
159
|
-
}
|
|
160
|
-
function resolveFindings(ctx) {
|
|
161
|
-
const report = ctx.report;
|
|
162
|
-
if (report && typeof report === "object" && "findings" in report) {
|
|
163
|
-
const f = report.findings;
|
|
164
|
-
if (Array.isArray(f) && f.length > 0) return f;
|
|
165
|
-
}
|
|
166
|
-
return ctx.findings;
|
|
167
|
-
}
|
|
168
|
-
|
|
169
|
-
// src/improvement/raw-trace-distiller.ts
|
|
170
|
-
import { existsSync, readdirSync } from "fs";
|
|
171
|
-
import { basename, join, resolve } from "path";
|
|
172
|
-
import { makeFinding } from "@tangle-network/agent-eval";
|
|
173
|
-
var ANALYST_ID = "raw-trace-distiller";
|
|
174
|
-
var PASS_THRESHOLD = 0.999;
|
|
175
|
-
function rawTraceDistiller(options = {}) {
|
|
176
|
-
const maxCandidates = options.maxCandidates ?? 12;
|
|
177
|
-
const maxCellsPerCandidate = options.maxCellsPerCandidate ?? 8;
|
|
178
|
-
const maxFilesPerCell = options.maxFilesPerCell ?? 24;
|
|
179
|
-
return async (input) => {
|
|
180
|
-
const genRoot = absoluteRunDir(options.runDir ?? input.runDir);
|
|
181
|
-
const durable = isDurable(genRoot);
|
|
182
|
-
const ranked = [...input.candidates].map((c) => ({
|
|
183
|
-
surfaceHash: c.surfaceHash,
|
|
184
|
-
composite: c.composite,
|
|
185
|
-
campaignDir: absoluteRunDir(c.campaign.runDir),
|
|
186
|
-
cells: failingCells(c.campaign, maxCellsPerCandidate, maxFilesPerCell)
|
|
187
|
-
})).sort((a, b) => a.composite - b.composite).slice(0, maxCandidates);
|
|
188
|
-
const totalFailingCells = ranked.reduce((n, c) => n + c.cells.length, 0);
|
|
189
|
-
if (totalFailingCells === 0) {
|
|
190
|
-
if (options.fallbackFindings && options.fallbackFindings.length > 0) {
|
|
191
|
-
return options.fallbackFindings;
|
|
192
|
-
}
|
|
193
|
-
return [
|
|
194
|
-
makeFinding({
|
|
195
|
-
analyst_id: ANALYST_ID,
|
|
196
|
-
severity: "info",
|
|
197
|
-
area: "raw-trace-context",
|
|
198
|
-
confidence: 1,
|
|
199
|
-
claim: `Generation ${input.generation} had no failing cells. The full raw run traces are on disk under ${genRoot}.`,
|
|
200
|
-
recommended_action: `To keep improving, grep/cat the raw traces under ${genRoot} (per-cell spans.jsonl + cached-result.json) to find the weakest passing runs, then make a targeted harness-code edit.`,
|
|
201
|
-
evidence_refs: [{ kind: "artifact", uri: genRoot }],
|
|
202
|
-
metadata: { generation: input.generation, runDir: genRoot, failingCells: 0 }
|
|
203
|
-
})
|
|
204
|
-
];
|
|
205
|
-
}
|
|
206
|
-
const findings = [];
|
|
207
|
-
findings.push(
|
|
208
|
-
makeFinding({
|
|
209
|
-
analyst_id: ANALYST_ID,
|
|
210
|
-
severity: "high",
|
|
211
|
-
area: "raw-trace-context",
|
|
212
|
-
confidence: 1,
|
|
213
|
-
claim: `Generation ${input.generation} produced ${totalFailingCells} failing/low-scoring cell(s) across ${ranked.length} candidate(s). Their FULL RAW run traces are on disk under ${genRoot} \u2014 the actual event logs (spans.jsonl), scores (cached-result.json), and artifacts, not a summary.${durable ? "" : " (WARNING: this run root does not exist on disk \u2014 it looks like an in-memory run; pass a real runDir to improve() to get raw-trace context.)"}`,
|
|
214
|
-
recommended_action: `Do NOT rely on a pre-summarized finding. Before editing, DIAGNOSE from the raw traces: run \`grep\`/\`cat\`/\`ls\` over the trace files and directories named in the following findings to see exactly what each failing run did and why it scored low, then make the smallest harness-code edit that fixes the dominant failure. Start with \`grep -rIn "error" ${genRoot}\` then \`cat\` the spans.jsonl of the worst cell.`,
|
|
215
|
-
evidence_refs: [{ kind: "artifact", uri: genRoot }],
|
|
216
|
-
metadata: {
|
|
217
|
-
generation: input.generation,
|
|
218
|
-
runDir: genRoot,
|
|
219
|
-
failingCells: totalFailingCells,
|
|
220
|
-
candidates: ranked.length
|
|
221
|
-
}
|
|
222
|
-
})
|
|
223
|
-
);
|
|
224
|
-
for (const cand of ranked) {
|
|
225
|
-
if (cand.cells.length === 0) continue;
|
|
226
|
-
const scenarioList = cand.cells.map((c) => c.scenarioId).join(", ");
|
|
227
|
-
const fileLines = cand.cells.map((c) => {
|
|
228
|
-
const header = ` cell ${c.scenarioId} (composite ${c.composite.toFixed(3)}${c.error ? `, error: ${truncate(c.error, 160)}` : ""}) \u2014 dir ${c.cellDir}`;
|
|
229
|
-
const files = c.files.map((f) => ` - ${f}`).join("\n");
|
|
230
|
-
const more = c.truncatedFiles ? `
|
|
231
|
-
- \u2026(ls ${c.cellDir} for the rest)` : "";
|
|
232
|
-
return c.files.length > 0 ? `${header}
|
|
233
|
-
${files}${more}` : header;
|
|
234
|
-
}).join("\n");
|
|
235
|
-
findings.push(
|
|
236
|
-
makeFinding({
|
|
237
|
-
analyst_id: ANALYST_ID,
|
|
238
|
-
severity: cand.composite < 0.5 ? "critical" : "high",
|
|
239
|
-
area: "raw-trace-context",
|
|
240
|
-
confidence: 1,
|
|
241
|
-
subject: cand.surfaceHash,
|
|
242
|
-
claim: `Candidate ${cand.surfaceHash} scored composite ${cand.composite.toFixed(3)} with ${cand.cells.length} failing cell(s) [${scenarioList}]. Its raw traces are under ${cand.campaignDir}.`,
|
|
243
|
-
recommended_action: `grep/cat these raw trace files to diagnose WHY this candidate failed before editing:
|
|
244
|
-
${fileLines}
|
|
245
|
-
Or scan the whole candidate at once: \`grep -rIn . ${cand.campaignDir}\` and \`ls -R ${cand.campaignDir}\`.`,
|
|
246
|
-
evidence_refs: [
|
|
247
|
-
{ kind: "artifact", uri: cand.campaignDir },
|
|
248
|
-
...cand.cells.flatMap(
|
|
249
|
-
(c) => c.files.map((f) => ({ kind: "artifact", uri: f }))
|
|
250
|
-
)
|
|
251
|
-
],
|
|
252
|
-
metadata: {
|
|
253
|
-
surfaceHash: cand.surfaceHash,
|
|
254
|
-
composite: cand.composite,
|
|
255
|
-
campaignDir: cand.campaignDir,
|
|
256
|
-
cells: cand.cells.map((c) => ({
|
|
257
|
-
scenarioId: c.scenarioId,
|
|
258
|
-
composite: c.composite,
|
|
259
|
-
cellDir: c.cellDir,
|
|
260
|
-
files: c.files,
|
|
261
|
-
...c.error ? { error: c.error } : {}
|
|
262
|
-
}))
|
|
263
|
-
}
|
|
264
|
-
})
|
|
265
|
-
);
|
|
266
|
-
}
|
|
267
|
-
return findings;
|
|
268
|
-
};
|
|
269
|
-
}
|
|
270
|
-
function failingCells(campaign, maxCells, maxFiles) {
|
|
271
|
-
const campaignDir = absoluteRunDir(campaign.runDir);
|
|
272
|
-
const durable = isDurable(campaignDir);
|
|
273
|
-
const out = [];
|
|
274
|
-
for (const cell of campaign.cells) {
|
|
275
|
-
const scores = Object.values(cell.judgeScores ?? {});
|
|
276
|
-
const composite = scores.length === 0 ? 0 : scores.reduce((sum, s) => sum + (s.composite ?? 0), 0) / scores.length;
|
|
277
|
-
if (!cell.error && composite >= PASS_THRESHOLD) continue;
|
|
278
|
-
const cellDir = join(campaignDir, sanitizeCellId(cell.cellId));
|
|
279
|
-
const artifactPaths = artifactPathsForCell(campaign.artifactsByPath, cell.cellId);
|
|
280
|
-
const discovered = durable ? listTraceFiles(cellDir) : [];
|
|
281
|
-
const canonical = [join(cellDir, "spans.jsonl"), join(cellDir, "cached-result.json")];
|
|
282
|
-
const files = dedupeSorted([...discovered, ...artifactPaths, ...canonical]);
|
|
283
|
-
out.push({
|
|
284
|
-
scenarioId: cell.scenarioId,
|
|
285
|
-
composite: Number(composite.toFixed(3)),
|
|
286
|
-
...cell.error ? { error: cell.error } : {},
|
|
287
|
-
cellDir,
|
|
288
|
-
files: files.slice(0, maxFiles),
|
|
289
|
-
truncatedFiles: files.length > maxFiles
|
|
290
|
-
});
|
|
291
|
-
if (out.length >= maxCells) break;
|
|
292
|
-
}
|
|
293
|
-
return out;
|
|
294
|
-
}
|
|
295
|
-
function artifactPathsForCell(artifactsByPath, cellId) {
|
|
296
|
-
if (!artifactsByPath) return [];
|
|
297
|
-
const prefix = `${cellId}/`;
|
|
298
|
-
return Object.entries(artifactsByPath).filter(([key]) => key.startsWith(prefix)).map(([, absPath]) => resolve(absPath));
|
|
299
|
-
}
|
|
300
|
-
function listTraceFiles(dir) {
|
|
301
|
-
const out = [];
|
|
302
|
-
for (const entry of safeReadDir(dir)) {
|
|
303
|
-
const full = join(dir, entry.name);
|
|
304
|
-
if (entry.isFile()) {
|
|
305
|
-
out.push(full);
|
|
306
|
-
} else if (!entry.isSymbolicLink() && entry.isDirectory()) {
|
|
307
|
-
for (const sub of safeReadDir(full)) {
|
|
308
|
-
if (sub.isFile()) out.push(join(full, sub.name));
|
|
309
|
-
}
|
|
310
|
-
}
|
|
311
|
-
}
|
|
312
|
-
return out;
|
|
313
|
-
}
|
|
314
|
-
function safeReadDir(dir) {
|
|
315
|
-
try {
|
|
316
|
-
return readdirSync(dir, { withFileTypes: true });
|
|
317
|
-
} catch {
|
|
318
|
-
return [];
|
|
319
|
-
}
|
|
320
|
-
}
|
|
321
|
-
function sanitizeCellId(cellId) {
|
|
322
|
-
return cellId.replace(/[^a-zA-Z0-9_-]/g, "_");
|
|
323
|
-
}
|
|
324
|
-
function isDurable(runDir) {
|
|
325
|
-
return !runDir.startsWith("mem://") && existsSync(runDir);
|
|
326
|
-
}
|
|
327
|
-
function absoluteRunDir(runDir) {
|
|
328
|
-
return runDir.startsWith("mem://") ? runDir : resolve(runDir);
|
|
329
|
-
}
|
|
330
|
-
function dedupeSorted(paths) {
|
|
331
|
-
return [...new Set(paths)].sort((a, b) => {
|
|
332
|
-
const da = a.slice(0, a.length - basename(a).length);
|
|
333
|
-
const db = b.slice(0, b.length - basename(b).length);
|
|
334
|
-
return da === db ? basename(a).localeCompare(basename(b)) : da.localeCompare(db);
|
|
335
|
-
});
|
|
336
|
-
}
|
|
337
|
-
function truncate(s, n) {
|
|
338
|
-
return s.length <= n ? s : `${s.slice(0, n - 1)}\u2026`;
|
|
339
|
-
}
|
|
340
|
-
|
|
341
|
-
// src/improvement/improve.ts
|
|
342
|
-
import { canonicalJson } from "@tangle-network/agent-eval";
|
|
343
|
-
import {
|
|
344
|
-
gepaProposer,
|
|
345
|
-
gitWorktreeAdapter,
|
|
346
|
-
memoryCurationProposer,
|
|
347
|
-
skillOptProposer
|
|
348
|
-
} from "@tangle-network/agent-eval/campaign";
|
|
349
|
-
import {
|
|
350
|
-
selfImprove
|
|
351
|
-
} from "@tangle-network/agent-eval/contract";
|
|
352
|
-
import { agentProfileSchema } from "@tangle-network/agent-interface";
|
|
353
|
-
var defaultReflectionModel = "deepseek-v4-flash";
|
|
354
|
-
function llmClientOptions(llm) {
|
|
355
|
-
return { baseUrl: llm?.baseUrl, apiKey: llm?.apiKey };
|
|
356
|
-
}
|
|
357
|
-
function defaultGeneratorFor(surface, llm) {
|
|
358
|
-
const model = llm?.model ?? defaultReflectionModel;
|
|
359
|
-
switch (surface) {
|
|
360
|
-
case "prompt":
|
|
361
|
-
return gepaProposer({ llm: llmClientOptions(llm), model, target: "agent system prompt" });
|
|
362
|
-
case "skills":
|
|
363
|
-
return skillOptProposer({ llm: llmClientOptions(llm), model, target: "agent skill document" });
|
|
364
|
-
case "memory":
|
|
365
|
-
return memoryCurationProposer();
|
|
366
|
-
default:
|
|
367
|
-
return void 0;
|
|
368
|
-
}
|
|
369
|
-
}
|
|
370
|
-
function baselineSurfaceFor(profile, surface, skills, memory) {
|
|
371
|
-
switch (surface) {
|
|
372
|
-
case "prompt":
|
|
373
|
-
return profile.prompt?.systemPrompt ?? "";
|
|
374
|
-
case "skills":
|
|
375
|
-
return skills?.document ?? JSON.stringify(profile.resources?.skills ?? []);
|
|
376
|
-
case "tools":
|
|
377
|
-
return JSON.stringify(profile.tools ?? {});
|
|
378
|
-
case "mcp":
|
|
379
|
-
return JSON.stringify(profile.mcp ?? {});
|
|
380
|
-
case "hooks":
|
|
381
|
-
return JSON.stringify(profile.hooks ?? {});
|
|
382
|
-
case "subagents":
|
|
383
|
-
return JSON.stringify(profile.subagents ?? {});
|
|
384
|
-
case "agent-profile":
|
|
385
|
-
return canonicalJson(profile);
|
|
386
|
-
case "memory":
|
|
387
|
-
if (!memory) {
|
|
388
|
-
throw new ConfigError("improve(): surface 'memory' requires opts.memory.document");
|
|
389
|
-
}
|
|
390
|
-
return memory.document;
|
|
391
|
-
case "code":
|
|
392
|
-
throw new ConfigError(
|
|
393
|
-
"improve(): code requires the isolated baseline created from opts.code.repoRoot"
|
|
394
|
-
);
|
|
395
|
-
}
|
|
396
|
-
}
|
|
397
|
-
var DISTILLED_NOTES_MAX_CHARS = 1500;
|
|
398
|
-
var DISTILLED_ERROR_MAX_CHARS = 500;
|
|
399
|
-
function generationFailureDistiller(staticFindings) {
|
|
400
|
-
const CAP = 12;
|
|
401
|
-
return async (input) => {
|
|
402
|
-
const failures = [];
|
|
403
|
-
for (const candidate of input.candidates) {
|
|
404
|
-
for (const rawCell of candidate.campaign.cells) {
|
|
405
|
-
const cell = rawCell;
|
|
406
|
-
const scenario = String(cell.scenarioId ?? "unknown");
|
|
407
|
-
const error = typeof cell.error === "string" ? cell.error : void 0;
|
|
408
|
-
const judgeScores = cell.judgeScores && typeof cell.judgeScores === "object" ? Object.values(
|
|
409
|
-
cell.judgeScores
|
|
410
|
-
) : [];
|
|
411
|
-
const composite = judgeScores.length === 0 ? 0 : judgeScores.reduce((sum, j) => sum + (j.composite ?? 0), 0) / judgeScores.length;
|
|
412
|
-
if (!error && composite >= 0.999) continue;
|
|
413
|
-
const notes = judgeScores.map((j) => j.notes).filter((n) => typeof n === "string" && n.length > 0).join("; ").slice(0, DISTILLED_NOTES_MAX_CHARS);
|
|
414
|
-
const claim = notes || (error ? `Scenario ${scenario} failed: ${error.slice(0, DISTILLED_ERROR_MAX_CHARS)}` : "");
|
|
415
|
-
failures.push({
|
|
416
|
-
scenario,
|
|
417
|
-
composite: Number(composite.toFixed(3)),
|
|
418
|
-
notes,
|
|
419
|
-
...claim ? { claim } : {},
|
|
420
|
-
...error ? { error: error.slice(0, DISTILLED_ERROR_MAX_CHARS) } : {}
|
|
421
|
-
});
|
|
422
|
-
}
|
|
423
|
-
}
|
|
424
|
-
if (failures.length === 0) return staticFindings;
|
|
425
|
-
failures.sort((a, b) => a.composite - b.composite);
|
|
426
|
-
return failures.slice(0, CAP);
|
|
427
|
-
};
|
|
428
|
-
}
|
|
429
|
-
function memoryGenerationDistiller(staticFindings) {
|
|
430
|
-
const distillFailures = generationFailureDistiller(staticFindings);
|
|
431
|
-
return async (input) => {
|
|
432
|
-
const fresh = await distillFailures(input);
|
|
433
|
-
return fresh === staticFindings ? staticFindings : [...staticFindings, ...fresh];
|
|
434
|
-
};
|
|
435
|
-
}
|
|
436
|
-
async function discardPreparedBaseline(worktree, baselineWorktree, cause) {
|
|
437
|
-
return rethrowAfterCleanup(
|
|
438
|
-
cause,
|
|
439
|
-
() => worktree.discard(baselineWorktree),
|
|
440
|
-
"improve(): code preparation failed"
|
|
441
|
-
);
|
|
442
|
-
}
|
|
443
|
-
function isCodeSurface(surface) {
|
|
444
|
-
return typeof surface === "object" && surface !== null && surface.kind === "code";
|
|
445
|
-
}
|
|
446
|
-
async function prepareCodeRun(code) {
|
|
447
|
-
const baseRef = code.baseRef ?? "main";
|
|
448
|
-
const worktree = code.worktree ?? gitWorktreeAdapter({
|
|
449
|
-
repoRoot: code.repoRoot,
|
|
450
|
-
...code.worktreeDir ? { worktreeDir: code.worktreeDir } : {}
|
|
451
|
-
});
|
|
452
|
-
const baselineWorktree = await worktree.create({ baseRef, label: "incumbent-baseline" });
|
|
453
|
-
try {
|
|
454
|
-
const baseline = await worktree.finalize(baselineWorktree, "Incumbent code checkout");
|
|
455
|
-
let baselineDiscarded = false;
|
|
456
|
-
const generator = code.generator ?? agenticGenerator({
|
|
457
|
-
...code.harness ? { harness: code.harness } : {},
|
|
458
|
-
...code.verify ? { verify: code.verify } : {},
|
|
459
|
-
...code.timeoutMs ? { timeoutMs: code.timeoutMs } : {}
|
|
460
|
-
});
|
|
461
|
-
const managed = improvementDriver({ worktree, generator, baseRef });
|
|
462
|
-
return {
|
|
463
|
-
baseline,
|
|
464
|
-
proposer: managed,
|
|
465
|
-
async cleanup(retainedWinner) {
|
|
466
|
-
const errors = [];
|
|
467
|
-
const retainedWorktreeRef = isCodeSurface(retainedWinner) ? retainedWinner.worktreeRef : void 0;
|
|
468
|
-
try {
|
|
469
|
-
await managed?.cleanup(retainedWorktreeRef ? [retainedWorktreeRef] : []);
|
|
470
|
-
} catch (cause) {
|
|
471
|
-
errors.push(cause);
|
|
472
|
-
}
|
|
473
|
-
if (!baselineDiscarded && retainedWorktreeRef !== baseline.worktreeRef) {
|
|
474
|
-
try {
|
|
475
|
-
await worktree.discard(baselineWorktree);
|
|
476
|
-
baselineDiscarded = true;
|
|
477
|
-
} catch (cause) {
|
|
478
|
-
errors.push(cause);
|
|
479
|
-
}
|
|
480
|
-
}
|
|
481
|
-
if (errors.length > 0) {
|
|
482
|
-
throw new AggregateError(errors, "improve(): failed to clean code improvement worktrees");
|
|
483
|
-
}
|
|
484
|
-
}
|
|
485
|
-
};
|
|
486
|
-
} catch (cause) {
|
|
487
|
-
return discardPreparedBaseline(worktree, baselineWorktree, cause);
|
|
488
|
-
}
|
|
489
|
-
}
|
|
490
|
-
function idempotentDispose(dispose) {
|
|
491
|
-
let disposed = false;
|
|
492
|
-
let inFlight;
|
|
493
|
-
return async () => {
|
|
494
|
-
if (disposed) return;
|
|
495
|
-
if (inFlight) return inFlight;
|
|
496
|
-
inFlight = (async () => {
|
|
497
|
-
await dispose();
|
|
498
|
-
disposed = true;
|
|
499
|
-
})();
|
|
500
|
-
try {
|
|
501
|
-
await inFlight;
|
|
502
|
-
} finally {
|
|
503
|
-
inFlight = void 0;
|
|
504
|
-
}
|
|
505
|
-
};
|
|
506
|
-
}
|
|
507
|
-
function parseWinnerJson(winner, surface) {
|
|
508
|
-
try {
|
|
509
|
-
return JSON.parse(winner);
|
|
510
|
-
} catch (cause) {
|
|
511
|
-
throw new ConfigError(
|
|
512
|
-
`improve(): the shipped '${surface}' winner is not valid JSON, so it cannot be applied back to the profile: ${cause.message}`
|
|
513
|
-
);
|
|
514
|
-
}
|
|
515
|
-
}
|
|
516
|
-
function applyImprovementWinnerToProfile(profile, surface, winner) {
|
|
517
|
-
if (typeof winner !== "string") return profile;
|
|
518
|
-
let candidate;
|
|
519
|
-
switch (surface) {
|
|
520
|
-
case "prompt":
|
|
521
|
-
candidate = { ...profile, prompt: { ...profile.prompt, systemPrompt: winner } };
|
|
522
|
-
break;
|
|
523
|
-
case "skills":
|
|
524
|
-
candidate = {
|
|
525
|
-
...profile,
|
|
526
|
-
resources: { ...profile.resources, skills: parseWinnerJson(winner, surface) }
|
|
527
|
-
};
|
|
528
|
-
break;
|
|
529
|
-
case "tools":
|
|
530
|
-
candidate = { ...profile, tools: parseWinnerJson(winner, surface) };
|
|
531
|
-
break;
|
|
532
|
-
case "mcp":
|
|
533
|
-
candidate = { ...profile, mcp: parseWinnerJson(winner, surface) };
|
|
534
|
-
break;
|
|
535
|
-
case "hooks":
|
|
536
|
-
candidate = { ...profile, hooks: parseWinnerJson(winner, surface) };
|
|
537
|
-
break;
|
|
538
|
-
case "subagents":
|
|
539
|
-
candidate = { ...profile, subagents: parseWinnerJson(winner, surface) };
|
|
540
|
-
break;
|
|
541
|
-
case "agent-profile":
|
|
542
|
-
candidate = parseWinnerJson(winner, surface);
|
|
543
|
-
break;
|
|
544
|
-
case "memory":
|
|
545
|
-
return profile;
|
|
546
|
-
case "code":
|
|
547
|
-
return profile;
|
|
548
|
-
}
|
|
549
|
-
const parsed = agentProfileSchema.safeParse(candidate);
|
|
550
|
-
if (!parsed.success) {
|
|
551
|
-
throw new ConfigError(
|
|
552
|
-
`improve(): the shipped '${surface}' winner does not produce a valid AgentProfile: ${parsed.error.message}`
|
|
553
|
-
);
|
|
554
|
-
}
|
|
555
|
-
return parsed.data;
|
|
556
|
-
}
|
|
557
|
-
async function improve(profile, findings, opts) {
|
|
558
|
-
const {
|
|
559
|
-
surface = "prompt",
|
|
560
|
-
gate = "holdout",
|
|
561
|
-
generator,
|
|
562
|
-
allowedModels,
|
|
563
|
-
rawTraceContext,
|
|
564
|
-
code,
|
|
565
|
-
skills,
|
|
566
|
-
memory,
|
|
567
|
-
promotionGate,
|
|
568
|
-
analyzeGeneration,
|
|
569
|
-
...sharedOptions
|
|
570
|
-
} = opts;
|
|
571
|
-
const parsedProfile = agentProfileSchema.safeParse(profile);
|
|
572
|
-
if (!parsedProfile.success) {
|
|
573
|
-
throw new ConfigError(
|
|
574
|
-
`improve(): input is not a valid AgentProfile: ${parsedProfile.error.message}`
|
|
575
|
-
);
|
|
576
|
-
}
|
|
577
|
-
if (surface === "skills" && !generator && !skills) {
|
|
578
|
-
throw new ConfigError(
|
|
579
|
-
"improve(): the default skills optimizer requires opts.skills.document; pass the skill text or an explicit generator that understands resource refs"
|
|
580
|
-
);
|
|
581
|
-
}
|
|
582
|
-
if (surface === "memory" && !memory) {
|
|
583
|
-
throw new ConfigError("improve(): surface 'memory' requires opts.memory.document");
|
|
584
|
-
}
|
|
585
|
-
if (surface === "code" && generator) {
|
|
586
|
-
throw new ConfigError(
|
|
587
|
-
"improve(): surface 'code' forbids opts.generator because an external SurfaceProposer cannot transfer checkout ownership; pass opts.code.generator instead"
|
|
588
|
-
);
|
|
589
|
-
}
|
|
590
|
-
const usesReflectionModel = !generator && (surface === "prompt" || surface === "skills");
|
|
591
|
-
if (usesReflectionModel) {
|
|
592
|
-
assertModelAllowed(sharedOptions.llm?.model ?? defaultReflectionModel, allowedModels);
|
|
593
|
-
}
|
|
594
|
-
let preparedCode;
|
|
595
|
-
if (surface === "code") {
|
|
596
|
-
if (!code) {
|
|
597
|
-
throw new ConfigError(
|
|
598
|
-
"improve(): surface 'code' requires opts.code.repoRoot so the incumbent can run from an isolated checkout"
|
|
599
|
-
);
|
|
600
|
-
}
|
|
601
|
-
preparedCode = await prepareCodeRun(code);
|
|
602
|
-
}
|
|
603
|
-
const proposer = preparedCode?.proposer ?? generator ?? defaultGeneratorFor(surface, sharedOptions.llm);
|
|
604
|
-
if (!proposer) {
|
|
605
|
-
throw new ConfigError(
|
|
606
|
-
`improve(): surface '${surface}' has no default generator \u2014 pass opts.generator (a SurfaceProposer) explicitly`
|
|
607
|
-
);
|
|
608
|
-
}
|
|
609
|
-
const budget = gate === "none" ? { ...sharedOptions.budget, generations: 0 } : { ...sharedOptions.budget };
|
|
610
|
-
let raw;
|
|
611
|
-
try {
|
|
612
|
-
raw = await selfImprove({
|
|
613
|
-
...sharedOptions,
|
|
614
|
-
baselineSurface: preparedCode?.baseline ?? baselineSurfaceFor(profile, surface, skills, memory),
|
|
615
|
-
proposer,
|
|
616
|
-
budget,
|
|
617
|
-
findings,
|
|
618
|
-
...promotionGate !== void 0 ? { gate: promotionGate } : {},
|
|
619
|
-
...analyzeGeneration === null ? {} : {
|
|
620
|
-
analyzeGeneration: analyzeGeneration ?? (rawTraceContext ? rawTraceDistiller({ fallbackFindings: findings }) : surface === "memory" ? memoryGenerationDistiller(findings) : generationFailureDistiller(findings))
|
|
621
|
-
}
|
|
622
|
-
});
|
|
623
|
-
} catch (cause) {
|
|
624
|
-
if (!preparedCode) throw cause;
|
|
625
|
-
return rethrowAfterCleanup(
|
|
626
|
-
cause,
|
|
627
|
-
() => preparedCode.cleanup(),
|
|
628
|
-
"improve(): code improvement failed"
|
|
629
|
-
);
|
|
630
|
-
}
|
|
631
|
-
const shipped = raw.gateDecision === "ship";
|
|
632
|
-
const winnerSurface = raw.winner.surface;
|
|
633
|
-
if (preparedCode) {
|
|
634
|
-
try {
|
|
635
|
-
await preparedCode.cleanup(winnerSurface);
|
|
636
|
-
} catch (cleanupCause) {
|
|
637
|
-
try {
|
|
638
|
-
await preparedCode.cleanup();
|
|
639
|
-
} catch (finalCleanupCause) {
|
|
640
|
-
throw new AggregateError(
|
|
641
|
-
[cleanupCause, finalCleanupCause],
|
|
642
|
-
"improve(): code result cleanup failed, including the final all-worktree retry"
|
|
643
|
-
);
|
|
644
|
-
}
|
|
645
|
-
throw new AggregateError(
|
|
646
|
-
[cleanupCause],
|
|
647
|
-
"improve(): code result cleanup failed; the final all-worktree retry succeeded"
|
|
648
|
-
);
|
|
649
|
-
}
|
|
650
|
-
}
|
|
651
|
-
const dispose = idempotentDispose(async () => preparedCode?.cleanup());
|
|
652
|
-
const externalDocument = surface === "skills" && skills ? skills : surface === "memory" && memory ? memory : void 0;
|
|
653
|
-
if (shipped && externalDocument) {
|
|
654
|
-
if (typeof winnerSurface !== "string") {
|
|
655
|
-
throw new ConfigError(
|
|
656
|
-
`improve(): the shipped '${surface}' winner must be text before it can be persisted`
|
|
657
|
-
);
|
|
658
|
-
}
|
|
659
|
-
await externalDocument.writeBack?.(winnerSurface);
|
|
660
|
-
}
|
|
661
|
-
const nextProfile = shipped && !externalDocument ? applyImprovementWinnerToProfile(profile, surface, winnerSurface) : profile;
|
|
662
|
-
return {
|
|
663
|
-
profile: nextProfile,
|
|
664
|
-
shipped,
|
|
665
|
-
lift: raw.lift,
|
|
666
|
-
gateDecision: raw.gateDecision,
|
|
667
|
-
raw,
|
|
668
|
-
dispose
|
|
669
|
-
};
|
|
670
|
-
}
|
|
671
|
-
|
|
672
|
-
// src/intelligence/improvement-cycle.ts
|
|
673
|
-
import { assertNoJudgeVerdict } from "@tangle-network/agent-eval/analyst";
|
|
674
|
-
import {
|
|
675
|
-
measuredComparisonFromCandidateExperiment,
|
|
676
|
-
runCandidateExperiment,
|
|
677
|
-
verifyCandidateExperiment,
|
|
678
|
-
verifyCandidateExperimentComparison
|
|
679
|
-
} from "@tangle-network/agent-eval/contract";
|
|
680
|
-
import {
|
|
681
|
-
agentCandidateMaterializationReceiptSchema,
|
|
682
|
-
agentCandidateRunReceiptSchema,
|
|
683
|
-
agentImprovementActivationSchema,
|
|
684
|
-
agentImprovementProposalSchema,
|
|
685
|
-
agentImprovementReviewSchema,
|
|
686
|
-
candidateExecutionEvidenceSchema
|
|
687
|
-
} from "@tangle-network/agent-interface";
|
|
688
|
-
import { materializeCandidateProfile } from "@tangle-network/agent-profile-materialize";
|
|
689
|
-
var AgentCandidateExperimentCellExecutionError = class extends Error {
|
|
690
|
-
finalization;
|
|
691
|
-
constructor(finalization) {
|
|
692
|
-
super(`candidate experiment cell failed: ${finalization.reason}`);
|
|
693
|
-
this.name = "AgentCandidateExperimentCellExecutionError";
|
|
694
|
-
this.finalization = finalization;
|
|
695
|
-
}
|
|
696
|
-
};
|
|
697
|
-
async function runAgentCandidateExperiment(options) {
|
|
698
|
-
const experiment = verifyCandidateExperiment(options.experiment);
|
|
699
|
-
const measurements = await runCandidateExperiment({
|
|
700
|
-
experiment,
|
|
701
|
-
...options.maxConcurrency === void 0 ? {} : { maxConcurrency: options.maxConcurrency },
|
|
702
|
-
...options.signal ? { signal: options.signal } : {},
|
|
703
|
-
execute: async (input) => {
|
|
704
|
-
const placement = await options.placeCell(input);
|
|
705
|
-
return await executeAgentCandidateExperimentCell({ ...input, ...placement });
|
|
706
|
-
}
|
|
707
|
-
});
|
|
708
|
-
const evaluation = createAgentImprovementMeasuredComparison({
|
|
709
|
-
experiment,
|
|
710
|
-
measurements,
|
|
711
|
-
runId: options.runId,
|
|
712
|
-
...options.candidate ? { candidate: options.candidate } : {},
|
|
713
|
-
...options.generationsExplored === void 0 ? {} : { generationsExplored: options.generationsExplored },
|
|
714
|
-
...options.searchDurationMs === void 0 ? {} : { searchDurationMs: options.searchDurationMs },
|
|
715
|
-
...options.searchCostUsd === void 0 ? {} : { searchCostUsd: options.searchCostUsd },
|
|
716
|
-
...options.metadata ? { metadata: options.metadata } : {}
|
|
717
|
-
});
|
|
718
|
-
return { experiment, measurements, evaluation };
|
|
719
|
-
}
|
|
720
|
-
async function executeAgentCandidateExperimentCell(options) {
|
|
721
|
-
const experiment = verifyCandidateExperiment(options.experiment);
|
|
722
|
-
const bundle = experiment[options.arm];
|
|
723
|
-
assertExactExperimentInput(options, experiment, bundle);
|
|
724
|
-
const attempt = options.attempt ?? 1;
|
|
725
|
-
if (attempt > options.task.attempt.maxAttempts) {
|
|
726
|
-
throw new Error("candidate experiment attempt exceeds the signed task policy");
|
|
727
|
-
}
|
|
728
|
-
const runCell = canonicalCandidateDocument({
|
|
729
|
-
kind: "agent-candidate-run-cell",
|
|
730
|
-
experimentDigest: experiment.digest,
|
|
731
|
-
arm: options.arm,
|
|
732
|
-
bundleDigest: bundle.digest,
|
|
733
|
-
suiteDigest: options.benchmarkCell.suiteDigest,
|
|
734
|
-
taskDigest: options.task.digest,
|
|
735
|
-
taskIndex: options.benchmarkCell.taskIndex,
|
|
736
|
-
repetition: options.benchmarkCell.repetition,
|
|
737
|
-
seed: options.seed,
|
|
738
|
-
attempt
|
|
739
|
-
}).value;
|
|
740
|
-
const verified = await verifyAgentCandidateBundle(bundle, options.ports);
|
|
741
|
-
const prepared = await prepareAgentCandidateExecution(
|
|
742
|
-
verified,
|
|
743
|
-
{
|
|
744
|
-
executionId: options.executionId,
|
|
745
|
-
runCell,
|
|
746
|
-
benchmarkSuite: experiment.benchmark.suite,
|
|
747
|
-
task: options.task,
|
|
748
|
-
executionRoots: options.executionRoots,
|
|
749
|
-
stagingRoots: options.stagingRoots
|
|
750
|
-
},
|
|
751
|
-
options.ports,
|
|
752
|
-
options.preparation
|
|
753
|
-
);
|
|
754
|
-
const finalization = await executePreparedAgentCandidate(prepared, options.execution);
|
|
755
|
-
if (!finalization.succeeded) {
|
|
756
|
-
throw new AgentCandidateExperimentCellExecutionError(finalization);
|
|
757
|
-
}
|
|
758
|
-
const evidence = canonicalCandidateDocument({
|
|
759
|
-
kind: "agent-candidate-execution-evidence",
|
|
760
|
-
materializationReceipt: prepared.materializationReceipt.value,
|
|
761
|
-
receipt: finalization.receipt.value
|
|
762
|
-
}).value;
|
|
763
|
-
return verifyCandidateExecutionEvidence(evidence, {
|
|
764
|
-
experiment,
|
|
765
|
-
arm: options.arm,
|
|
766
|
-
benchmarkCell: options.benchmarkCell,
|
|
767
|
-
seed: options.seed,
|
|
768
|
-
attempt,
|
|
769
|
-
resolvedResources: verifiedResourceTextByDigest(verified)
|
|
770
|
-
});
|
|
771
|
-
}
|
|
772
|
-
function createAgentImprovementMeasuredComparison(options) {
|
|
773
|
-
return verifyCandidateExperimentComparison(measuredComparisonFromCandidateExperiment(options));
|
|
774
|
-
}
|
|
775
|
-
async function proposeAgentImprovement(options) {
|
|
776
|
-
const writeBackSurface = options.improvement.skills?.writeBack ? "skill" : options.improvement.memory?.writeBack ? "memory" : null;
|
|
777
|
-
if (writeBackSurface) {
|
|
778
|
-
throw new Error(`proposeAgentImprovement cannot write ${writeBackSurface} before approval`);
|
|
779
|
-
}
|
|
780
|
-
const analysis = await runAnalystLoop({ ...options.analysis, runId: options.runId });
|
|
781
|
-
const findings = assertNoJudgeVerdict(
|
|
782
|
-
analysis.analystResult.findings,
|
|
783
|
-
"proposeAgentImprovement findings"
|
|
784
|
-
);
|
|
785
|
-
const improvement = await improve(options.profile, [...findings], options.improvement);
|
|
786
|
-
try {
|
|
787
|
-
if (!improvement.shipped) {
|
|
788
|
-
throw new Error("agent improvement search did not produce a promotable candidate");
|
|
789
|
-
}
|
|
790
|
-
const experiment = verifyCandidateExperiment(
|
|
791
|
-
await options.buildExperiment({ analysis, improvement })
|
|
792
|
-
);
|
|
793
|
-
if (canonicalCandidateDigest(experiment.baseline.profile) !== canonicalCandidateDigest(options.profile)) {
|
|
794
|
-
throw new Error("candidate experiment baseline does not match the analyzed agent profile");
|
|
795
|
-
}
|
|
796
|
-
const measured = await runAgentCandidateExperiment({
|
|
797
|
-
experiment,
|
|
798
|
-
runId: options.runId,
|
|
799
|
-
placeCell: options.placeCell,
|
|
800
|
-
...options.maxConcurrency === void 0 ? {} : { maxConcurrency: options.maxConcurrency },
|
|
801
|
-
...options.signal ? { signal: options.signal } : {},
|
|
802
|
-
...options.candidate ? { candidate: options.candidate } : {},
|
|
803
|
-
...options.metadata ? { metadata: options.metadata } : {},
|
|
804
|
-
generationsExplored: improvement.raw.generationsExplored,
|
|
805
|
-
searchDurationMs: improvement.raw.durationMs,
|
|
806
|
-
searchCostUsd: improvement.raw.totalCostUsd
|
|
807
|
-
});
|
|
808
|
-
const proposal = createAgentImprovementProposal({
|
|
809
|
-
runId: options.runId,
|
|
810
|
-
findings,
|
|
811
|
-
evaluation: measured.evaluation,
|
|
812
|
-
...options.now ? { now: options.now } : {}
|
|
813
|
-
});
|
|
814
|
-
return {
|
|
815
|
-
analysis,
|
|
816
|
-
improvement,
|
|
817
|
-
experiment,
|
|
818
|
-
measurements: measured.measurements,
|
|
819
|
-
proposal
|
|
820
|
-
};
|
|
821
|
-
} catch (cause) {
|
|
822
|
-
return rethrowAfterCleanup(cause, () => improvement.dispose(), "proposeAgentImprovement failed");
|
|
823
|
-
}
|
|
824
|
-
}
|
|
825
|
-
function createAgentImprovementProposal(options) {
|
|
826
|
-
const findings = assertNoJudgeVerdict(
|
|
827
|
-
[...options.findings],
|
|
828
|
-
"createAgentImprovementProposal findings"
|
|
829
|
-
);
|
|
830
|
-
const evaluation = verifyCandidateExperimentComparison(options.evaluation);
|
|
831
|
-
if (evaluation.decision.outcome !== "ship") {
|
|
832
|
-
throw new Error("agent improvement proposal requires a passing experiment");
|
|
833
|
-
}
|
|
834
|
-
if (options.runId !== evaluation.provenance.runId) {
|
|
835
|
-
throw new Error("proposal runId does not match its measured experiment");
|
|
836
|
-
}
|
|
837
|
-
const changedSurfaces = deriveChangedSurfaces(
|
|
838
|
-
evaluation.experiment.baseline,
|
|
839
|
-
evaluation.experiment.candidate
|
|
840
|
-
);
|
|
841
|
-
return agentImprovementProposalSchema.parse(
|
|
842
|
-
canonicalCandidateDocument({
|
|
843
|
-
kind: "agent-improvement-proposal",
|
|
844
|
-
runId: options.runId,
|
|
845
|
-
changedSurfaces,
|
|
846
|
-
proposedAt: (options.now ?? (() => /* @__PURE__ */ new Date()))().toISOString(),
|
|
847
|
-
findings: [...findings],
|
|
848
|
-
evaluation
|
|
849
|
-
}).value
|
|
850
|
-
);
|
|
851
|
-
}
|
|
852
|
-
function reviewAgentImprovementProposal(inputProposal, input) {
|
|
853
|
-
const proposal = verifyAgentImprovementProposal(inputProposal);
|
|
854
|
-
if (!input.reviewedBy.trim()) throw new Error("candidate review requires reviewedBy");
|
|
855
|
-
if (!input.reason.trim()) throw new Error("candidate review requires a reason");
|
|
856
|
-
if (input.decision === "approve" && proposal.evaluation.decision.outcome !== "ship") {
|
|
857
|
-
throw new Error("candidate cannot be approved without a passing experiment");
|
|
858
|
-
}
|
|
859
|
-
return agentImprovementReviewSchema.parse(
|
|
860
|
-
canonicalCandidateDocument({
|
|
861
|
-
kind: "agent-improvement-review",
|
|
862
|
-
proposalDigest: proposal.digest,
|
|
863
|
-
decision: input.decision,
|
|
864
|
-
reviewedBy: input.reviewedBy,
|
|
865
|
-
reviewedAt: (input.now ?? (() => /* @__PURE__ */ new Date()))().toISOString(),
|
|
866
|
-
reason: input.reason,
|
|
867
|
-
...input.feedback === void 0 ? {} : { feedback: input.feedback }
|
|
868
|
-
}).value
|
|
869
|
-
);
|
|
870
|
-
}
|
|
871
|
-
function createAgentImprovementActivation(inputProposal, inputReview, options) {
|
|
872
|
-
const proposal = verifyAgentImprovementProposal(inputProposal);
|
|
873
|
-
const review = verifyAgentImprovementReview(inputReview);
|
|
874
|
-
if (review.decision !== "approve" || review.proposalDigest !== proposal.digest) {
|
|
875
|
-
throw new Error("candidate activation requires an approval for the exact proposal");
|
|
876
|
-
}
|
|
877
|
-
if (!options.fundingOwner.trim() || !options.authorizedBy.trim()) {
|
|
878
|
-
throw new Error("candidate activation authority must be non-empty");
|
|
879
|
-
}
|
|
880
|
-
const experiment = proposal.evaluation.experiment;
|
|
881
|
-
assertActivationTargets(proposal.changedSurfaces, experiment, options.targets);
|
|
882
|
-
return agentImprovementActivationSchema.parse(
|
|
883
|
-
canonicalCandidateDocument({
|
|
884
|
-
kind: "agent-improvement-activation",
|
|
885
|
-
proposalDigest: proposal.digest,
|
|
886
|
-
reviewDigest: review.digest,
|
|
887
|
-
experimentDigest: experiment.digest,
|
|
888
|
-
candidateBundleDigest: experiment.candidate.digest,
|
|
889
|
-
targets: options.targets,
|
|
890
|
-
fundingOwner: options.fundingOwner,
|
|
891
|
-
authorizedBy: options.authorizedBy,
|
|
892
|
-
authorizedAt: (options.now ?? (() => /* @__PURE__ */ new Date()))().toISOString()
|
|
893
|
-
}).value
|
|
894
|
-
);
|
|
895
|
-
}
|
|
896
|
-
function verifyAgentImprovementProposal(input) {
|
|
897
|
-
const proposal = verifyCanonicalCandidateDocument(
|
|
898
|
-
agentImprovementProposalSchema.parse(input),
|
|
899
|
-
"agent improvement proposal"
|
|
900
|
-
);
|
|
901
|
-
const evaluation = verifyCandidateExperimentComparison(proposal.evaluation);
|
|
902
|
-
if (evaluation.decision.outcome !== "ship") {
|
|
903
|
-
throw new Error("agent improvement proposal does not contain a passing experiment");
|
|
904
|
-
}
|
|
905
|
-
if (proposal.runId !== evaluation.provenance.runId) {
|
|
906
|
-
throw new Error("proposal runId does not match its measured experiment");
|
|
907
|
-
}
|
|
908
|
-
const changedSurfaces = deriveChangedSurfaces(
|
|
909
|
-
evaluation.experiment.baseline,
|
|
910
|
-
evaluation.experiment.candidate
|
|
911
|
-
);
|
|
912
|
-
if (!sameOrderedValues(proposal.changedSurfaces, changedSurfaces)) {
|
|
913
|
-
throw new Error("proposal changed surfaces do not match its exact experiment");
|
|
914
|
-
}
|
|
915
|
-
assertNoJudgeDerivedProposalFindings(proposal.findings);
|
|
916
|
-
return proposal;
|
|
917
|
-
}
|
|
918
|
-
function verifyAgentImprovementReview(input) {
|
|
919
|
-
return verifyCanonicalCandidateDocument(
|
|
920
|
-
agentImprovementReviewSchema.parse(input),
|
|
921
|
-
"agent improvement review"
|
|
922
|
-
);
|
|
923
|
-
}
|
|
924
|
-
function verifyAgentImprovementActivation(input) {
|
|
925
|
-
const proposal = verifyAgentImprovementProposal(input.proposal);
|
|
926
|
-
const review = verifyAgentImprovementReview(input.review);
|
|
927
|
-
const activation = verifyCanonicalCandidateDocument(
|
|
928
|
-
agentImprovementActivationSchema.parse(input.activation),
|
|
929
|
-
"agent improvement activation"
|
|
930
|
-
);
|
|
931
|
-
const experiment = proposal.evaluation.experiment;
|
|
932
|
-
if (review.decision !== "approve" || review.proposalDigest !== proposal.digest || activation.proposalDigest !== proposal.digest || activation.reviewDigest !== review.digest || activation.experimentDigest !== experiment.digest || activation.candidateBundleDigest !== experiment.candidate.digest) {
|
|
933
|
-
throw new Error("candidate activation does not bind the measured and approved candidate");
|
|
934
|
-
}
|
|
935
|
-
assertActivationTargets(proposal.changedSurfaces, experiment, activation.targets);
|
|
936
|
-
return activation;
|
|
937
|
-
}
|
|
938
|
-
function verifyCandidateExecutionEvidence(input, options) {
|
|
939
|
-
const experiment = verifyCandidateExperiment(options.experiment);
|
|
940
|
-
const bundle = experiment[options.arm];
|
|
941
|
-
const task = experiment.benchmark.tasks[options.benchmarkCell.taskIndex];
|
|
942
|
-
const index = options.benchmarkCell.taskIndex * experiment.benchmark.suite.reps + options.benchmarkCell.repetition;
|
|
943
|
-
if (!task || options.benchmarkCell.suiteDigest !== experiment.benchmark.suite.digest || options.seed !== experiment.benchmark.suite.seeds[index]) {
|
|
944
|
-
throw new Error("candidate execution evidence points outside its signed experiment");
|
|
945
|
-
}
|
|
946
|
-
const evidence = verifyCanonicalCandidateDocument(
|
|
947
|
-
candidateExecutionEvidenceSchema.parse(input),
|
|
948
|
-
"candidate execution evidence"
|
|
949
|
-
);
|
|
950
|
-
const materialization = verifyCanonicalCandidateDocument(
|
|
951
|
-
agentCandidateMaterializationReceiptSchema.parse(evidence.materializationReceipt),
|
|
952
|
-
"candidate materialization receipt"
|
|
953
|
-
);
|
|
954
|
-
const receipt = verifyCanonicalCandidateDocument(
|
|
955
|
-
agentCandidateRunReceiptSchema.parse(evidence.receipt),
|
|
956
|
-
"candidate run receipt"
|
|
957
|
-
);
|
|
958
|
-
const plan = materialization.executionPlan;
|
|
959
|
-
const cell = plan.material.runCell;
|
|
960
|
-
const attempt = options.attempt ?? 1;
|
|
961
|
-
if (cell.experimentDigest !== experiment.digest || cell.arm !== options.arm || cell.bundleDigest !== bundle.digest || cell.suiteDigest !== experiment.benchmark.suite.digest || cell.taskDigest !== task.digest || cell.taskIndex !== options.benchmarkCell.taskIndex || cell.repetition !== options.benchmarkCell.repetition || cell.seed !== options.seed || cell.attempt !== attempt || canonicalCandidateDigest(omitTopLevelDigest(cell)) !== cell.digest) {
|
|
962
|
-
throw new Error("candidate execution receipt substituted its signed experiment cell");
|
|
963
|
-
}
|
|
964
|
-
assertCapturedInput(
|
|
965
|
-
materialization.benchmark.suite,
|
|
966
|
-
experiment.benchmark.suite,
|
|
967
|
-
"benchmark suite"
|
|
968
|
-
);
|
|
969
|
-
assertCapturedInput(materialization.benchmark.task, task, "benchmark task");
|
|
970
|
-
assertEvidenceMaterialDigest(plan, "candidate execution plan");
|
|
971
|
-
assertEvidenceMaterialDigest(
|
|
972
|
-
materialization.profileActivation.profilePlan,
|
|
973
|
-
"candidate profile plan"
|
|
974
|
-
);
|
|
975
|
-
const expectedProfilePlan = materializeCandidateProfile(
|
|
976
|
-
bundle.profile,
|
|
977
|
-
candidateMaterializerHarness(materialization.harness),
|
|
978
|
-
{ resolvedResources: options.resolvedResources }
|
|
979
|
-
);
|
|
980
|
-
const activation = parseAgentCandidateProfileActivation(
|
|
981
|
-
materialization.profileActivation,
|
|
982
|
-
materialization.profileActivation.profilePlan.digest
|
|
983
|
-
);
|
|
984
|
-
const regeneratedActivation = createAgentCandidateProfileActivation(
|
|
985
|
-
expectedProfilePlan,
|
|
986
|
-
materialization.profileActivation.profilePlan
|
|
987
|
-
);
|
|
988
|
-
if (activation.digest !== regeneratedActivation.digest) {
|
|
989
|
-
throw new Error("candidate profile activation does not match the experiment bundle");
|
|
990
|
-
}
|
|
991
|
-
if (materialization.bundleDigest !== bundle.digest || receipt.bundleDigest !== bundle.digest || receipt.runCellDigest !== cell.digest || receipt.materializationReceiptDigest !== materialization.digest || receipt.executionPlanDigest !== plan.digest) {
|
|
992
|
-
throw new Error("candidate execution evidence does not bind one exact Runtime run");
|
|
993
|
-
}
|
|
994
|
-
assertEvidenceMaterialDigest(receipt.modelSettlement, "candidate model settlement");
|
|
995
|
-
assertEvidenceMaterialDigest(receipt.taskOutcome, "candidate task outcome");
|
|
996
|
-
assertEvidenceMaterialDigest(receipt.benchmarkResult, "candidate benchmark result");
|
|
997
|
-
return immutableCandidateValue(evidence);
|
|
998
|
-
}
|
|
999
|
-
function assertExactExperimentInput(input, experiment, bundle) {
|
|
1000
|
-
const task = experiment.benchmark.tasks[input.benchmarkCell.taskIndex];
|
|
1001
|
-
const index = input.benchmarkCell.taskIndex * experiment.benchmark.suite.reps + input.benchmarkCell.repetition;
|
|
1002
|
-
if (input.experiment.digest !== experiment.digest || input.bundle.digest !== bundle.digest || !task || input.task.digest !== task.digest || input.benchmarkCell.suiteDigest !== experiment.benchmark.suite.digest || input.seed !== experiment.benchmark.suite.seeds[index]) {
|
|
1003
|
-
throw new Error("Runtime received a substituted candidate experiment cell");
|
|
1004
|
-
}
|
|
1005
|
-
}
|
|
1006
|
-
function assertCapturedInput(captured, expected, label) {
|
|
1007
|
-
const bytes = canonicalCandidateBytes(omitTopLevelDigest(expected));
|
|
1008
|
-
if (captured.digest !== expected.digest || captured.material.sha256 !== expected.digest || captured.material.byteLength !== bytes.byteLength) {
|
|
1009
|
-
throw new Error(`candidate materialization substituted its ${label}`);
|
|
1010
|
-
}
|
|
1011
|
-
}
|
|
1012
|
-
function assertEvidenceMaterialDigest(evidence, label) {
|
|
1013
|
-
const bytes = canonicalCandidateBytes(evidence.material);
|
|
1014
|
-
if (canonicalCandidateDigest(evidence.material) !== evidence.digest || evidence.artifact.sha256 !== evidence.digest || evidence.artifact.byteLength !== bytes.byteLength) {
|
|
1015
|
-
throw new Error(`${label} digest does not match its canonical material`);
|
|
1016
|
-
}
|
|
1017
|
-
}
|
|
1018
|
-
var CHANGED_SURFACE_ORDER = [
|
|
1019
|
-
"prompt",
|
|
1020
|
-
"skills",
|
|
1021
|
-
"tools",
|
|
1022
|
-
"mcp",
|
|
1023
|
-
"hooks",
|
|
1024
|
-
"subagents",
|
|
1025
|
-
"agent-profile",
|
|
1026
|
-
"memory",
|
|
1027
|
-
"code",
|
|
1028
|
-
"knowledge"
|
|
1029
|
-
];
|
|
1030
|
-
function deriveChangedSurfaces(baselineBundle, candidateBundle) {
|
|
1031
|
-
const baseline = improvementSurfaceValues(baselineBundle);
|
|
1032
|
-
const candidate = improvementSurfaceValues(candidateBundle);
|
|
1033
|
-
const changed = /* @__PURE__ */ new Set();
|
|
1034
|
-
for (const surface of CHANGED_SURFACE_ORDER) {
|
|
1035
|
-
if (canonicalCandidateDigest(baseline[surface]) !== canonicalCandidateDigest(candidate[surface])) {
|
|
1036
|
-
changed.add(surface);
|
|
1037
|
-
}
|
|
1038
|
-
}
|
|
1039
|
-
const ordered = CHANGED_SURFACE_ORDER.filter((surface) => changed.has(surface));
|
|
1040
|
-
if (ordered.length === 0) throw new Error("candidate experiment does not change an agent surface");
|
|
1041
|
-
return ordered;
|
|
1042
|
-
}
|
|
1043
|
-
function improvementSurfaceValues(bundle) {
|
|
1044
|
-
const profile = agentCandidateProfileAsAgentProfile(bundle.profile);
|
|
1045
|
-
return {
|
|
1046
|
-
prompt: {
|
|
1047
|
-
prompt: profile.prompt ?? null,
|
|
1048
|
-
instructions: profile.resources?.instructions ?? null
|
|
1049
|
-
},
|
|
1050
|
-
skills: profile.resources?.skills ?? null,
|
|
1051
|
-
tools: {
|
|
1052
|
-
tools: profile.tools ?? null,
|
|
1053
|
-
resources: profile.resources?.tools ?? null
|
|
1054
|
-
},
|
|
1055
|
-
mcp: profile.mcp ?? null,
|
|
1056
|
-
hooks: profile.hooks ?? null,
|
|
1057
|
-
subagents: {
|
|
1058
|
-
subagents: profile.subagents ?? null,
|
|
1059
|
-
resources: profile.resources?.agents ?? null
|
|
1060
|
-
},
|
|
1061
|
-
"agent-profile": { profile: opaqueProfileSlice(profile), execution: bundle.execution },
|
|
1062
|
-
memory: bundle.memory,
|
|
1063
|
-
code: bundle.code,
|
|
1064
|
-
knowledge: bundle.knowledge ?? null
|
|
1065
|
-
};
|
|
1066
|
-
}
|
|
1067
|
-
function opaqueProfileSlice(profile) {
|
|
1068
|
-
const {
|
|
1069
|
-
prompt: _prompt,
|
|
1070
|
-
tools: _tools,
|
|
1071
|
-
mcp: _mcp,
|
|
1072
|
-
hooks: _hooks,
|
|
1073
|
-
subagents: _subagents,
|
|
1074
|
-
resources,
|
|
1075
|
-
...opaqueProfile
|
|
1076
|
-
} = profile;
|
|
1077
|
-
const {
|
|
1078
|
-
instructions: _instructions,
|
|
1079
|
-
skills: _skills,
|
|
1080
|
-
tools: _resourceTools,
|
|
1081
|
-
agents: _agents,
|
|
1082
|
-
...opaqueResources
|
|
1083
|
-
} = resources ?? {};
|
|
1084
|
-
return {
|
|
1085
|
-
...opaqueProfile,
|
|
1086
|
-
...Object.keys(opaqueResources).length > 0 ? { resources: opaqueResources } : {}
|
|
1087
|
-
};
|
|
1088
|
-
}
|
|
1089
|
-
function assertActivationTargets(surfaces, experiment, targets) {
|
|
1090
|
-
const expected = new Set(surfaces);
|
|
1091
|
-
const actual = new Set(targets.map((target) => target.surface));
|
|
1092
|
-
const baselineValues = improvementSurfaceValues(experiment.baseline);
|
|
1093
|
-
if (targets.some((target) => !target.identity.trim()) || targets.some(
|
|
1094
|
-
(target) => target.expectedBaseDigest !== expectedActivationBaseDigest(experiment, target.surface, baselineValues)
|
|
1095
|
-
) || expected.size !== actual.size || [...expected].some((surface) => !actual.has(surface))) {
|
|
1096
|
-
throw new Error("candidate activation targets must cover exactly the changed surfaces");
|
|
1097
|
-
}
|
|
1098
|
-
}
|
|
1099
|
-
function expectedActivationBaseDigest(experiment, surface, baselineValues) {
|
|
1100
|
-
if (surface === "knowledge" && experiment.candidate.knowledge) {
|
|
1101
|
-
return experiment.candidate.knowledge.candidate.baseHash;
|
|
1102
|
-
}
|
|
1103
|
-
return canonicalCandidateDigest(baselineValues[surface]);
|
|
1104
|
-
}
|
|
1105
|
-
function sameOrderedValues(left, right) {
|
|
1106
|
-
return left.length === right.length && left.every((value, index) => value === right[index]);
|
|
1107
|
-
}
|
|
1108
|
-
function assertNoJudgeDerivedProposalFindings(findings) {
|
|
1109
|
-
const leaked = findings.filter((finding) => finding.derived_from_judge === true);
|
|
1110
|
-
if (leaked.length === 0) return;
|
|
1111
|
-
const identifiers = leaked.map(
|
|
1112
|
-
(finding) => typeof finding.finding_id === "string" ? finding.finding_id : "<unknown>"
|
|
1113
|
-
);
|
|
1114
|
-
throw new Error(
|
|
1115
|
-
`agent improvement proposal findings: judge-derived findings cannot steer an improvement: [${identifiers.join(", ")}]`
|
|
1116
|
-
);
|
|
1117
|
-
}
|
|
1118
|
-
|
|
1119
|
-
export {
|
|
1120
|
-
improvementDriver,
|
|
1121
|
-
rawTraceDistiller,
|
|
1122
|
-
applyImprovementWinnerToProfile,
|
|
1123
|
-
improve,
|
|
1124
|
-
AgentCandidateExperimentCellExecutionError,
|
|
1125
|
-
runAgentCandidateExperiment,
|
|
1126
|
-
executeAgentCandidateExperimentCell,
|
|
1127
|
-
createAgentImprovementMeasuredComparison,
|
|
1128
|
-
proposeAgentImprovement,
|
|
1129
|
-
createAgentImprovementProposal,
|
|
1130
|
-
reviewAgentImprovementProposal,
|
|
1131
|
-
createAgentImprovementActivation,
|
|
1132
|
-
verifyAgentImprovementProposal,
|
|
1133
|
-
verifyAgentImprovementReview,
|
|
1134
|
-
verifyAgentImprovementActivation,
|
|
1135
|
-
verifyCandidateExecutionEvidence
|
|
1136
|
-
};
|
|
1137
|
-
//# sourceMappingURL=chunk-G55QE4IQ.js.map
|