@tangle-network/agent-runtime 0.90.1 → 0.92.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -2
- package/dist/agent.d.ts +3 -3
- package/dist/agent.js +88 -9
- package/dist/agent.js.map +1 -1
- package/dist/{mcp-serve-verifier-XsX8rkB9.d.ts → agentic-generator-B8oeE2Yv.d.ts} +6 -33
- package/dist/analyst-loop.d.ts +1 -1
- package/dist/candidate-execution/index.d.ts +104 -0
- package/dist/candidate-execution/index.js +34 -0
- package/dist/candidate-execution/index.js.map +1 -0
- package/dist/chunk-3BE7KTMU.js +1229 -0
- package/dist/chunk-3BE7KTMU.js.map +1 -0
- package/dist/chunk-3D2RHC4K.js +73 -0
- package/dist/chunk-3D2RHC4K.js.map +1 -0
- package/dist/chunk-3MDZX7YU.js +125 -0
- package/dist/chunk-3MDZX7YU.js.map +1 -0
- package/dist/{chunk-RYBVU4M3.js → chunk-6O5USWVH.js} +32 -1413
- package/dist/chunk-6O5USWVH.js.map +1 -0
- package/dist/chunk-6O73TRHW.js +142 -0
- package/dist/chunk-6O73TRHW.js.map +1 -0
- package/dist/{chunk-R2VAJGR3.js → chunk-7VJJJ2T2.js} +2 -2
- package/dist/chunk-A62TP7SK.js +4784 -0
- package/dist/chunk-A62TP7SK.js.map +1 -0
- package/dist/chunk-APVPRF4Y.js +2166 -0
- package/dist/chunk-APVPRF4Y.js.map +1 -0
- package/dist/{chunk-QK4DV5PR.js → chunk-AUEIDTR3.js} +2 -2
- package/dist/{chunk-OOL3675H.js → chunk-FRBHUNQ7.js} +2 -139
- package/dist/chunk-FRBHUNQ7.js.map +1 -0
- package/dist/{chunk-7ON74BQO.js → chunk-GDAQUFG6.js} +2 -2
- package/dist/{chunk-ZV4LXYCJ.js → chunk-I7WVPJBZ.js} +23 -1231
- package/dist/chunk-I7WVPJBZ.js.map +1 -0
- package/dist/{chunk-WRUSWK4F.js → chunk-IGGZGKJD.js} +3 -3
- package/dist/chunk-PH65PR4F.js +860 -0
- package/dist/chunk-PH65PR4F.js.map +1 -0
- package/dist/chunk-RSWM2ZKM.js +659 -0
- package/dist/chunk-RSWM2ZKM.js.map +1 -0
- package/dist/{chunk-BZF3KQ6G.js → chunk-VSWBYWFK.js} +4 -122
- package/dist/chunk-VSWBYWFK.js.map +1 -0
- package/dist/{completion-gate-DkAnUmpb.d.ts → completion-gate-BLaiN0-X.d.ts} +1 -1
- package/dist/{coordination-rRj5hjJK.d.ts → coordination-DxJ83oZA.d.ts} +12 -5
- package/dist/environment-provider.d.ts +2 -2
- package/dist/environment-provider.js +2 -1
- package/dist/improve-CUVCq7xg.d.ts +152 -0
- package/dist/{improvement-adapter-CDR8QNVM.d.ts → improvement-adapter-BieWeK5J.d.ts} +16 -0
- package/dist/index.d.ts +27 -291
- package/dist/index.js +70 -889
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +160 -13
- package/dist/intelligence.js +535 -59
- package/dist/intelligence.js.map +1 -1
- package/dist/knowledge.d.ts +6 -6
- package/dist/knowledge.js +6 -4
- package/dist/lifecycle.d.ts +2 -1
- package/dist/lifecycle.js +5 -3
- package/dist/lifecycle.js.map +1 -1
- package/dist/{loop-runner-bin-DTbZVGfM.d.ts → loop-runner-bin-kKUNGLyV.d.ts} +2 -2
- package/dist/loop-runner-bin.d.ts +5 -5
- package/dist/loop-runner-bin.js +8 -5
- package/dist/loops.d.ts +16 -16
- package/dist/loops.js +47 -41
- package/dist/mcp/bin.js +6 -4
- package/dist/mcp/bin.js.map +1 -1
- package/dist/mcp/index.d.ts +8 -8
- package/dist/mcp/index.js +9 -6
- package/dist/mcp/index.js.map +1 -1
- package/dist/mcp-serve-verifier-Bg4C3p5S.d.ts +34 -0
- package/dist/{openai-tools-C4ZfUD4L.d.ts → openai-tools-E3woykz9.d.ts} +1 -1
- package/dist/prepare-Z08a4heC.d.ts +713 -0
- package/dist/profiles.d.ts +1 -1
- package/dist/{router-client-DJImUDlm.d.ts → sanitize-C9go6tXj.d.ts} +113 -1
- package/dist/{structural-rollout-MwlpgQ-6.d.ts → structural-rollout-DHGDbhvR.d.ts} +3 -3
- package/dist/{supervise-DPmYPk0j.d.ts → supervise-T2pazU3G.d.ts} +4 -4
- package/dist/{types-SyuwunY_.d.ts → types-B00NtbCs.d.ts} +1 -1
- package/dist/{types-eMNgWgFi.d.ts → types-DAdIm4AC.d.ts} +1 -1
- package/dist/{worktree-fanout-BDFQIO-Y.d.ts → worktree-fanout-BUb2Ag02.d.ts} +3 -3
- package/package.json +26 -36
- package/skills/build-with-agent-runtime/SKILL.md +1 -1
- package/dist/chunk-BZF3KQ6G.js.map +0 -1
- package/dist/chunk-IVGYLCFH.js +0 -381
- package/dist/chunk-IVGYLCFH.js.map +0 -1
- package/dist/chunk-OOL3675H.js.map +0 -1
- package/dist/chunk-RYBVU4M3.js.map +0 -1
- package/dist/chunk-ZV4LXYCJ.js.map +0 -1
- /package/dist/{chunk-R2VAJGR3.js.map → chunk-7VJJJ2T2.js.map} +0 -0
- /package/dist/{chunk-QK4DV5PR.js.map → chunk-AUEIDTR3.js.map} +0 -0
- /package/dist/{chunk-7ON74BQO.js.map → chunk-GDAQUFG6.js.map} +0 -0
- /package/dist/{chunk-WRUSWK4F.js.map → chunk-IGGZGKJD.js.map} +0 -0
package/dist/index.js
CHANGED
|
@@ -1,7 +1,13 @@
|
|
|
1
|
+
import {
|
|
2
|
+
FileAgentCandidateExecutionClaimStore,
|
|
3
|
+
createProtectedAgentCandidateModelPort,
|
|
4
|
+
disposePreparedAgentCandidateExecution,
|
|
5
|
+
recoverExpiredAgentCandidateExecution
|
|
6
|
+
} from "./chunk-PH65PR4F.js";
|
|
1
7
|
import {
|
|
2
8
|
mcpToolsForRuntimeMcp,
|
|
3
9
|
mcpToolsForRuntimeMcpSubset
|
|
4
|
-
} from "./chunk-
|
|
10
|
+
} from "./chunk-AUEIDTR3.js";
|
|
5
11
|
import {
|
|
6
12
|
DEFAULT_ROUTER_BASE_URL,
|
|
7
13
|
cleanModelId,
|
|
@@ -20,27 +26,53 @@ import {
|
|
|
20
26
|
runLoopRunnerCli,
|
|
21
27
|
selfImproveLoopRunner,
|
|
22
28
|
worktreeLoopRunner
|
|
23
|
-
} from "./chunk-
|
|
29
|
+
} from "./chunk-IGGZGKJD.js";
|
|
24
30
|
import "./chunk-SGKPNBXE.js";
|
|
31
|
+
import {
|
|
32
|
+
ROLLOUT_POLICY_BOUNDS,
|
|
33
|
+
ROLLOUT_POLICY_EXTENSION,
|
|
34
|
+
applyRolloutPolicyToProfile,
|
|
35
|
+
enumerateNeighborPolicies,
|
|
36
|
+
improve,
|
|
37
|
+
improvementDriver,
|
|
38
|
+
normalizeRolloutPolicy,
|
|
39
|
+
parseRolloutPolicy,
|
|
40
|
+
rawTraceDistiller,
|
|
41
|
+
rolloutPolicyProposer,
|
|
42
|
+
serializeRolloutPolicy,
|
|
43
|
+
structuralRolloutPolicyFromProfile
|
|
44
|
+
} from "./chunk-RSWM2ZKM.js";
|
|
45
|
+
import {
|
|
46
|
+
CANDIDATE_TRACE_ENV,
|
|
47
|
+
CANDIDATE_TRACE_TAGS,
|
|
48
|
+
InMemoryAgentCandidateExecutionClaimStore,
|
|
49
|
+
candidateExecutionClaim,
|
|
50
|
+
executePreparedAgentCandidate,
|
|
51
|
+
persistCandidateOutputArtifact,
|
|
52
|
+
prepareAgentCandidateExecution,
|
|
53
|
+
verifyAgentCandidateBundle
|
|
54
|
+
} from "./chunk-A62TP7SK.js";
|
|
25
55
|
import {
|
|
26
56
|
InMemoryRuntimeSessionStore,
|
|
27
57
|
createIterableBackend,
|
|
28
58
|
createOpenAICompatibleBackend,
|
|
29
59
|
createSandboxPromptBackend,
|
|
30
|
-
defaultStructuralRolloutPolicy,
|
|
31
60
|
newRuntimeSession,
|
|
32
61
|
normalizeBackendStreamEvent,
|
|
33
62
|
nowIso,
|
|
34
63
|
touchSession
|
|
35
|
-
} from "./chunk-
|
|
64
|
+
} from "./chunk-I7WVPJBZ.js";
|
|
65
|
+
import "./chunk-3BE7KTMU.js";
|
|
36
66
|
import "./chunk-ZQZX77MM.js";
|
|
37
67
|
import {
|
|
38
|
-
agenticGenerator,
|
|
39
|
-
commandVerifier,
|
|
40
68
|
mcpBuildPrompt,
|
|
41
69
|
mcpServeVerifier,
|
|
42
70
|
toolBuildPrompt
|
|
43
|
-
} from "./chunk-
|
|
71
|
+
} from "./chunk-6O73TRHW.js";
|
|
72
|
+
import {
|
|
73
|
+
agenticGenerator,
|
|
74
|
+
commandVerifier
|
|
75
|
+
} from "./chunk-FRBHUNQ7.js";
|
|
44
76
|
import {
|
|
45
77
|
RESEARCH_SUPERVISOR_SYSTEM_PROMPT,
|
|
46
78
|
createAgentKnowledgeReadinessCheck,
|
|
@@ -49,25 +81,31 @@ import {
|
|
|
49
81
|
knowledgeReadinessDeliverable,
|
|
50
82
|
runKnowledgeImprovementJob,
|
|
51
83
|
runSupervisedKnowledgeUpdate
|
|
52
|
-
} from "./chunk-
|
|
84
|
+
} from "./chunk-7VJJJ2T2.js";
|
|
53
85
|
import "./chunk-5QOB7H74.js";
|
|
54
|
-
import
|
|
55
|
-
assertModelAllowed,
|
|
56
|
-
composeRuntimeHooks,
|
|
57
|
-
defineRuntimeHooks,
|
|
58
|
-
notifyRuntimeDecisionPoint,
|
|
59
|
-
notifyRuntimeHookEvent
|
|
60
|
-
} from "./chunk-RYBVU4M3.js";
|
|
86
|
+
import "./chunk-6O5USWVH.js";
|
|
61
87
|
import "./chunk-DPEUKJRO.js";
|
|
62
88
|
import {
|
|
63
89
|
INTELLIGENCE_WIRE_VERSION,
|
|
64
90
|
buildLoopOtelSpans,
|
|
65
91
|
buildLoopSpanNodes,
|
|
92
|
+
buildRuntimeEventOtelSpans,
|
|
93
|
+
composeRuntimeHooks,
|
|
66
94
|
createOtelExporter,
|
|
95
|
+
createRuntimeEventCollector,
|
|
96
|
+
createRuntimeStreamEventCollector,
|
|
97
|
+
defineRuntimeHooks,
|
|
67
98
|
exportEvalRuns,
|
|
68
|
-
loopEventToOtelSpan
|
|
69
|
-
|
|
70
|
-
|
|
99
|
+
loopEventToOtelSpan,
|
|
100
|
+
notifyRuntimeDecisionPoint,
|
|
101
|
+
notifyRuntimeHookEvent,
|
|
102
|
+
sanitizeAgentRuntimeEvent,
|
|
103
|
+
sanitizeKnowledgeReadinessReport,
|
|
104
|
+
sanitizeRuntimeStreamEvent
|
|
105
|
+
} from "./chunk-APVPRF4Y.js";
|
|
106
|
+
import "./chunk-3D2RHC4K.js";
|
|
107
|
+
import "./chunk-VSWBYWFK.js";
|
|
108
|
+
import "./chunk-3MDZX7YU.js";
|
|
71
109
|
import "./chunk-FVJ7M3DA.js";
|
|
72
110
|
import "./chunk-CMYMTRGA.js";
|
|
73
111
|
import {
|
|
@@ -182,7 +220,7 @@ function computeBackoff(spec, attempt) {
|
|
|
182
220
|
return Math.max(0, spec);
|
|
183
221
|
}
|
|
184
222
|
function sleep(ms) {
|
|
185
|
-
return new Promise((
|
|
223
|
+
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
186
224
|
}
|
|
187
225
|
|
|
188
226
|
// src/conversation/headers.ts
|
|
@@ -1282,508 +1320,6 @@ function deriveExecutionId(input) {
|
|
|
1282
1320
|
return `${input.projectId}:${input.sessionId}:${input.turnIndex}`;
|
|
1283
1321
|
}
|
|
1284
1322
|
|
|
1285
|
-
// src/improvement/improve.ts
|
|
1286
|
-
import {
|
|
1287
|
-
gepaProposer,
|
|
1288
|
-
gitWorktreeAdapter,
|
|
1289
|
-
skillOptProposer
|
|
1290
|
-
} from "@tangle-network/agent-eval/campaign";
|
|
1291
|
-
import {
|
|
1292
|
-
selfImprove
|
|
1293
|
-
} from "@tangle-network/agent-eval/contract";
|
|
1294
|
-
|
|
1295
|
-
// src/improvement/improvement-driver.ts
|
|
1296
|
-
function improvementDriver(opts) {
|
|
1297
|
-
const baseRef = opts.baseRef ?? "main";
|
|
1298
|
-
return {
|
|
1299
|
-
kind: `improvement:${opts.generator.kind}`,
|
|
1300
|
-
async propose(ctx) {
|
|
1301
|
-
const findings = resolveFindings(ctx);
|
|
1302
|
-
if (findings.length === 0 && ctx.report === void 0 && !opts.generator.proposesWithoutFindings) {
|
|
1303
|
-
return [];
|
|
1304
|
-
}
|
|
1305
|
-
const surfaces = [];
|
|
1306
|
-
for (let i = 0; i < ctx.populationSize; i++) {
|
|
1307
|
-
if (ctx.signal.aborted) break;
|
|
1308
|
-
const wt = await opts.worktree.create({
|
|
1309
|
-
baseRef,
|
|
1310
|
-
label: `${opts.generator.kind}-gen${ctx.generation}-cand${i}`
|
|
1311
|
-
});
|
|
1312
|
-
try {
|
|
1313
|
-
const { applied, summary } = await opts.generator.generate({
|
|
1314
|
-
worktreePath: wt.path,
|
|
1315
|
-
report: ctx.report,
|
|
1316
|
-
findings,
|
|
1317
|
-
dataset: ctx.dataset,
|
|
1318
|
-
maxShots: ctx.maxImprovementShots ?? 1,
|
|
1319
|
-
signal: ctx.signal
|
|
1320
|
-
});
|
|
1321
|
-
if (!applied) {
|
|
1322
|
-
await opts.worktree.discard(wt);
|
|
1323
|
-
continue;
|
|
1324
|
-
}
|
|
1325
|
-
surfaces.push(await opts.worktree.finalize(wt, summary));
|
|
1326
|
-
} catch (err) {
|
|
1327
|
-
await opts.worktree.discard(wt).catch(() => {
|
|
1328
|
-
});
|
|
1329
|
-
throw err;
|
|
1330
|
-
}
|
|
1331
|
-
}
|
|
1332
|
-
return surfaces;
|
|
1333
|
-
}
|
|
1334
|
-
};
|
|
1335
|
-
}
|
|
1336
|
-
function resolveFindings(ctx) {
|
|
1337
|
-
const report = ctx.report;
|
|
1338
|
-
if (report && typeof report === "object" && "findings" in report) {
|
|
1339
|
-
const f = report.findings;
|
|
1340
|
-
if (Array.isArray(f) && f.length > 0) return f;
|
|
1341
|
-
}
|
|
1342
|
-
return ctx.findings;
|
|
1343
|
-
}
|
|
1344
|
-
|
|
1345
|
-
// src/improvement/raw-trace-distiller.ts
|
|
1346
|
-
import { existsSync, readdirSync } from "fs";
|
|
1347
|
-
import { basename, join, resolve } from "path";
|
|
1348
|
-
import { makeFinding } from "@tangle-network/agent-eval";
|
|
1349
|
-
var ANALYST_ID = "raw-trace-distiller";
|
|
1350
|
-
var PASS_THRESHOLD = 0.999;
|
|
1351
|
-
function rawTraceDistiller(options = {}) {
|
|
1352
|
-
const maxCandidates = options.maxCandidates ?? 12;
|
|
1353
|
-
const maxCellsPerCandidate = options.maxCellsPerCandidate ?? 8;
|
|
1354
|
-
const maxFilesPerCell = options.maxFilesPerCell ?? 24;
|
|
1355
|
-
return async (input) => {
|
|
1356
|
-
const genRoot = absoluteRunDir(options.runDir ?? input.runDir);
|
|
1357
|
-
const durable = isDurable(genRoot);
|
|
1358
|
-
const ranked = [...input.candidates].map((c) => ({
|
|
1359
|
-
surfaceHash: c.surfaceHash,
|
|
1360
|
-
composite: c.composite,
|
|
1361
|
-
campaignDir: absoluteRunDir(c.campaign.runDir),
|
|
1362
|
-
cells: failingCells(c.campaign, maxCellsPerCandidate, maxFilesPerCell)
|
|
1363
|
-
})).sort((a, b) => a.composite - b.composite).slice(0, maxCandidates);
|
|
1364
|
-
const totalFailingCells = ranked.reduce((n, c) => n + c.cells.length, 0);
|
|
1365
|
-
if (totalFailingCells === 0) {
|
|
1366
|
-
if (options.fallbackFindings && options.fallbackFindings.length > 0) {
|
|
1367
|
-
return options.fallbackFindings;
|
|
1368
|
-
}
|
|
1369
|
-
return [
|
|
1370
|
-
makeFinding({
|
|
1371
|
-
analyst_id: ANALYST_ID,
|
|
1372
|
-
severity: "info",
|
|
1373
|
-
area: "raw-trace-context",
|
|
1374
|
-
confidence: 1,
|
|
1375
|
-
claim: `Generation ${input.generation} had no failing cells. The full raw run traces are on disk under ${genRoot}.`,
|
|
1376
|
-
recommended_action: `To keep improving, grep/cat the raw traces under ${genRoot} (per-cell spans.jsonl + cached-result.json) to find the weakest passing runs, then make a targeted harness-code edit.`,
|
|
1377
|
-
evidence_refs: [{ kind: "artifact", uri: genRoot }],
|
|
1378
|
-
metadata: { generation: input.generation, runDir: genRoot, failingCells: 0 }
|
|
1379
|
-
})
|
|
1380
|
-
];
|
|
1381
|
-
}
|
|
1382
|
-
const findings = [];
|
|
1383
|
-
findings.push(
|
|
1384
|
-
makeFinding({
|
|
1385
|
-
analyst_id: ANALYST_ID,
|
|
1386
|
-
severity: "high",
|
|
1387
|
-
area: "raw-trace-context",
|
|
1388
|
-
confidence: 1,
|
|
1389
|
-
claim: `Generation ${input.generation} produced ${totalFailingCells} failing/low-scoring cell(s) across ${ranked.length} candidate(s). Their FULL RAW run traces are on disk under ${genRoot} \u2014 the actual event logs (spans.jsonl), scores (cached-result.json), and artifacts, not a summary.${durable ? "" : " (WARNING: this run root does not exist on disk \u2014 it looks like an in-memory run; pass a real runDir to improve() to get raw-trace context.)"}`,
|
|
1390
|
-
recommended_action: `Do NOT rely on a pre-summarized finding. Before editing, DIAGNOSE from the raw traces: run \`grep\`/\`cat\`/\`ls\` over the trace files and directories named in the following findings to see exactly what each failing run did and why it scored low, then make the smallest harness-code edit that fixes the dominant failure. Start with \`grep -rIn "error" ${genRoot}\` then \`cat\` the spans.jsonl of the worst cell.`,
|
|
1391
|
-
evidence_refs: [{ kind: "artifact", uri: genRoot }],
|
|
1392
|
-
metadata: {
|
|
1393
|
-
generation: input.generation,
|
|
1394
|
-
runDir: genRoot,
|
|
1395
|
-
failingCells: totalFailingCells,
|
|
1396
|
-
candidates: ranked.length
|
|
1397
|
-
}
|
|
1398
|
-
})
|
|
1399
|
-
);
|
|
1400
|
-
for (const cand of ranked) {
|
|
1401
|
-
if (cand.cells.length === 0) continue;
|
|
1402
|
-
const scenarioList = cand.cells.map((c) => c.scenarioId).join(", ");
|
|
1403
|
-
const fileLines = cand.cells.map((c) => {
|
|
1404
|
-
const header = ` cell ${c.scenarioId} (composite ${c.composite.toFixed(3)}${c.error ? `, error: ${truncate(c.error, 160)}` : ""}) \u2014 dir ${c.cellDir}`;
|
|
1405
|
-
const files = c.files.map((f) => ` - ${f}`).join("\n");
|
|
1406
|
-
const more = c.truncatedFiles ? `
|
|
1407
|
-
- \u2026(ls ${c.cellDir} for the rest)` : "";
|
|
1408
|
-
return c.files.length > 0 ? `${header}
|
|
1409
|
-
${files}${more}` : header;
|
|
1410
|
-
}).join("\n");
|
|
1411
|
-
findings.push(
|
|
1412
|
-
makeFinding({
|
|
1413
|
-
analyst_id: ANALYST_ID,
|
|
1414
|
-
severity: cand.composite < 0.5 ? "critical" : "high",
|
|
1415
|
-
area: "raw-trace-context",
|
|
1416
|
-
confidence: 1,
|
|
1417
|
-
subject: cand.surfaceHash,
|
|
1418
|
-
claim: `Candidate ${cand.surfaceHash} scored composite ${cand.composite.toFixed(3)} with ${cand.cells.length} failing cell(s) [${scenarioList}]. Its raw traces are under ${cand.campaignDir}.`,
|
|
1419
|
-
recommended_action: `grep/cat these raw trace files to diagnose WHY this candidate failed before editing:
|
|
1420
|
-
${fileLines}
|
|
1421
|
-
Or scan the whole candidate at once: \`grep -rIn . ${cand.campaignDir}\` and \`ls -R ${cand.campaignDir}\`.`,
|
|
1422
|
-
evidence_refs: [
|
|
1423
|
-
{ kind: "artifact", uri: cand.campaignDir },
|
|
1424
|
-
...cand.cells.flatMap(
|
|
1425
|
-
(c) => c.files.map((f) => ({ kind: "artifact", uri: f }))
|
|
1426
|
-
)
|
|
1427
|
-
],
|
|
1428
|
-
metadata: {
|
|
1429
|
-
surfaceHash: cand.surfaceHash,
|
|
1430
|
-
composite: cand.composite,
|
|
1431
|
-
campaignDir: cand.campaignDir,
|
|
1432
|
-
cells: cand.cells.map((c) => ({
|
|
1433
|
-
scenarioId: c.scenarioId,
|
|
1434
|
-
composite: c.composite,
|
|
1435
|
-
cellDir: c.cellDir,
|
|
1436
|
-
files: c.files,
|
|
1437
|
-
...c.error ? { error: c.error } : {}
|
|
1438
|
-
}))
|
|
1439
|
-
}
|
|
1440
|
-
})
|
|
1441
|
-
);
|
|
1442
|
-
}
|
|
1443
|
-
return findings;
|
|
1444
|
-
};
|
|
1445
|
-
}
|
|
1446
|
-
function failingCells(campaign, maxCells, maxFiles) {
|
|
1447
|
-
const campaignDir = absoluteRunDir(campaign.runDir);
|
|
1448
|
-
const durable = isDurable(campaignDir);
|
|
1449
|
-
const out = [];
|
|
1450
|
-
for (const cell of campaign.cells) {
|
|
1451
|
-
const scores = Object.values(cell.judgeScores ?? {});
|
|
1452
|
-
const composite = scores.length === 0 ? 0 : scores.reduce((sum, s) => sum + (s.composite ?? 0), 0) / scores.length;
|
|
1453
|
-
if (!cell.error && composite >= PASS_THRESHOLD) continue;
|
|
1454
|
-
const cellDir = join(campaignDir, sanitizeCellId(cell.cellId));
|
|
1455
|
-
const artifactPaths = artifactPathsForCell(campaign.artifactsByPath, cell.cellId);
|
|
1456
|
-
const discovered = durable ? listTraceFiles(cellDir) : [];
|
|
1457
|
-
const canonical = [join(cellDir, "spans.jsonl"), join(cellDir, "cached-result.json")];
|
|
1458
|
-
const files = dedupeSorted([...discovered, ...artifactPaths, ...canonical]);
|
|
1459
|
-
out.push({
|
|
1460
|
-
scenarioId: cell.scenarioId,
|
|
1461
|
-
composite: Number(composite.toFixed(3)),
|
|
1462
|
-
...cell.error ? { error: cell.error } : {},
|
|
1463
|
-
cellDir,
|
|
1464
|
-
files: files.slice(0, maxFiles),
|
|
1465
|
-
truncatedFiles: files.length > maxFiles
|
|
1466
|
-
});
|
|
1467
|
-
if (out.length >= maxCells) break;
|
|
1468
|
-
}
|
|
1469
|
-
return out;
|
|
1470
|
-
}
|
|
1471
|
-
function artifactPathsForCell(artifactsByPath, cellId) {
|
|
1472
|
-
if (!artifactsByPath) return [];
|
|
1473
|
-
const prefix = `${cellId}/`;
|
|
1474
|
-
return Object.entries(artifactsByPath).filter(([key]) => key.startsWith(prefix)).map(([, absPath]) => resolve(absPath));
|
|
1475
|
-
}
|
|
1476
|
-
function listTraceFiles(dir) {
|
|
1477
|
-
const out = [];
|
|
1478
|
-
for (const entry of safeReadDir(dir)) {
|
|
1479
|
-
const full = join(dir, entry.name);
|
|
1480
|
-
if (entry.isFile()) {
|
|
1481
|
-
out.push(full);
|
|
1482
|
-
} else if (!entry.isSymbolicLink() && entry.isDirectory()) {
|
|
1483
|
-
for (const sub of safeReadDir(full)) {
|
|
1484
|
-
if (sub.isFile()) out.push(join(full, sub.name));
|
|
1485
|
-
}
|
|
1486
|
-
}
|
|
1487
|
-
}
|
|
1488
|
-
return out;
|
|
1489
|
-
}
|
|
1490
|
-
function safeReadDir(dir) {
|
|
1491
|
-
try {
|
|
1492
|
-
return readdirSync(dir, { withFileTypes: true });
|
|
1493
|
-
} catch {
|
|
1494
|
-
return [];
|
|
1495
|
-
}
|
|
1496
|
-
}
|
|
1497
|
-
function sanitizeCellId(cellId) {
|
|
1498
|
-
return cellId.replace(/[^a-zA-Z0-9_-]/g, "_");
|
|
1499
|
-
}
|
|
1500
|
-
function isDurable(runDir) {
|
|
1501
|
-
return !runDir.startsWith("mem://") && existsSync(runDir);
|
|
1502
|
-
}
|
|
1503
|
-
function absoluteRunDir(runDir) {
|
|
1504
|
-
return runDir.startsWith("mem://") ? runDir : resolve(runDir);
|
|
1505
|
-
}
|
|
1506
|
-
function dedupeSorted(paths) {
|
|
1507
|
-
return [...new Set(paths)].sort((a, b) => {
|
|
1508
|
-
const da = a.slice(0, a.length - basename(a).length);
|
|
1509
|
-
const db = b.slice(0, b.length - basename(b).length);
|
|
1510
|
-
return da === db ? basename(a).localeCompare(basename(b)) : da.localeCompare(db);
|
|
1511
|
-
});
|
|
1512
|
-
}
|
|
1513
|
-
function truncate(s, n) {
|
|
1514
|
-
return s.length <= n ? s : `${s.slice(0, n - 1)}\u2026`;
|
|
1515
|
-
}
|
|
1516
|
-
|
|
1517
|
-
// src/improvement/rollout-policy.ts
|
|
1518
|
-
var ROLLOUT_POLICY_EXTENSION = "structural-rollout";
|
|
1519
|
-
var ROLLOUT_POLICY_BOUNDS = {
|
|
1520
|
-
k: { min: 1, max: 10, step: 2 },
|
|
1521
|
-
repairRounds: { min: 0, max: 3, step: 1 },
|
|
1522
|
-
testgen: { min: 0, max: 10, step: 3 }
|
|
1523
|
-
};
|
|
1524
|
-
var MAX_CANDIDATES_PER_GENERATION = 4;
|
|
1525
|
-
var clamp = (v, min, max) => Math.min(max, Math.max(min, v));
|
|
1526
|
-
var isBoundedInt = (v, min) => typeof v === "number" && Number.isInteger(v) && v >= min;
|
|
1527
|
-
function parseRolloutPolicy(surface) {
|
|
1528
|
-
if (typeof surface !== "string" || surface.trim().length === 0) return void 0;
|
|
1529
|
-
let raw;
|
|
1530
|
-
try {
|
|
1531
|
-
raw = JSON.parse(surface);
|
|
1532
|
-
} catch {
|
|
1533
|
-
return void 0;
|
|
1534
|
-
}
|
|
1535
|
-
return normalizeRolloutPolicy(raw);
|
|
1536
|
-
}
|
|
1537
|
-
function normalizeRolloutPolicy(raw) {
|
|
1538
|
-
if (typeof raw !== "object" || raw === null || Array.isArray(raw)) return void 0;
|
|
1539
|
-
const bag = raw;
|
|
1540
|
-
const k = bag.k ?? defaultStructuralRolloutPolicy.k;
|
|
1541
|
-
const repairRounds = bag.repairRounds ?? defaultStructuralRolloutPolicy.repairRounds;
|
|
1542
|
-
const testgen = bag.testgen ?? defaultStructuralRolloutPolicy.testgen;
|
|
1543
|
-
if (!isBoundedInt(k, 1) || !isBoundedInt(repairRounds, 0) || !isBoundedInt(testgen, 0)) {
|
|
1544
|
-
return void 0;
|
|
1545
|
-
}
|
|
1546
|
-
return {
|
|
1547
|
-
k,
|
|
1548
|
-
repairRounds,
|
|
1549
|
-
testgen,
|
|
1550
|
-
...typeof bag.diverse === "boolean" ? { diverse: bag.diverse } : {},
|
|
1551
|
-
...typeof bag.temperature === "number" ? { temperature: bag.temperature } : {}
|
|
1552
|
-
};
|
|
1553
|
-
}
|
|
1554
|
-
function serializeRolloutPolicy(policy) {
|
|
1555
|
-
return JSON.stringify({
|
|
1556
|
-
k: policy.k,
|
|
1557
|
-
repairRounds: policy.repairRounds,
|
|
1558
|
-
testgen: policy.testgen,
|
|
1559
|
-
...policy.diverse !== void 0 ? { diverse: policy.diverse } : {},
|
|
1560
|
-
...policy.temperature !== void 0 ? { temperature: policy.temperature } : {}
|
|
1561
|
-
});
|
|
1562
|
-
}
|
|
1563
|
-
function structuralRolloutPolicyFromProfile(profile) {
|
|
1564
|
-
const bag = profile.extensions?.[ROLLOUT_POLICY_EXTENSION];
|
|
1565
|
-
if (bag === void 0) return void 0;
|
|
1566
|
-
return normalizeRolloutPolicy(bag);
|
|
1567
|
-
}
|
|
1568
|
-
function applyRolloutPolicyToProfile(profile, policy) {
|
|
1569
|
-
const bag = {
|
|
1570
|
-
k: policy.k,
|
|
1571
|
-
repairRounds: policy.repairRounds,
|
|
1572
|
-
testgen: policy.testgen,
|
|
1573
|
-
...policy.diverse !== void 0 ? { diverse: policy.diverse } : {},
|
|
1574
|
-
...policy.temperature !== void 0 ? { temperature: policy.temperature } : {}
|
|
1575
|
-
};
|
|
1576
|
-
return {
|
|
1577
|
-
...profile,
|
|
1578
|
-
extensions: { ...profile.extensions, [ROLLOUT_POLICY_EXTENSION]: bag }
|
|
1579
|
-
};
|
|
1580
|
-
}
|
|
1581
|
-
function enumerateNeighborPolicies(policy) {
|
|
1582
|
-
const moves = [
|
|
1583
|
-
{ dial: "k", delta: 1 },
|
|
1584
|
-
{ dial: "k", delta: -1 },
|
|
1585
|
-
{ dial: "repairRounds", delta: 1 },
|
|
1586
|
-
{ dial: "repairRounds", delta: -1 },
|
|
1587
|
-
{ dial: "testgen", delta: 1 },
|
|
1588
|
-
{ dial: "testgen", delta: -1 }
|
|
1589
|
-
];
|
|
1590
|
-
const seen = /* @__PURE__ */ new Set([serializeRolloutPolicy(policy)]);
|
|
1591
|
-
const neighbors = [];
|
|
1592
|
-
for (const move of moves) {
|
|
1593
|
-
const bounds = ROLLOUT_POLICY_BOUNDS[move.dial];
|
|
1594
|
-
const next = clamp(policy[move.dial] + move.delta * bounds.step, bounds.min, bounds.max);
|
|
1595
|
-
const candidate = { ...policy, [move.dial]: next };
|
|
1596
|
-
const key = serializeRolloutPolicy(candidate);
|
|
1597
|
-
if (seen.has(key)) continue;
|
|
1598
|
-
seen.add(key);
|
|
1599
|
-
neighbors.push(candidate);
|
|
1600
|
-
}
|
|
1601
|
-
return neighbors;
|
|
1602
|
-
}
|
|
1603
|
-
function candidateLabel(base, next) {
|
|
1604
|
-
for (const dial of ["k", "repairRounds", "testgen"]) {
|
|
1605
|
-
if (next[dial] !== base[dial]) return `${dial} ${base[dial]}\u2192${next[dial]}`;
|
|
1606
|
-
}
|
|
1607
|
-
return "unchanged";
|
|
1608
|
-
}
|
|
1609
|
-
function rolloutPolicyProposer() {
|
|
1610
|
-
return {
|
|
1611
|
-
kind: "rollout-policy",
|
|
1612
|
-
async propose(ctx) {
|
|
1613
|
-
const policy = parseRolloutPolicy(ctx.currentSurface);
|
|
1614
|
-
if (!policy) return [];
|
|
1615
|
-
const neighbors = enumerateNeighborPolicies(policy);
|
|
1616
|
-
if (neighbors.length === 0) return [];
|
|
1617
|
-
const cap = Math.max(1, Math.min(ctx.populationSize, MAX_CANDIDATES_PER_GENERATION));
|
|
1618
|
-
const start = ctx.generation * cap % neighbors.length;
|
|
1619
|
-
const window = [];
|
|
1620
|
-
for (let i = 0; i < Math.min(cap, neighbors.length); i += 1) {
|
|
1621
|
-
window.push(neighbors[(start + i) % neighbors.length]);
|
|
1622
|
-
}
|
|
1623
|
-
return window.map((candidate) => ({
|
|
1624
|
-
surface: serializeRolloutPolicy(candidate),
|
|
1625
|
-
label: candidateLabel(policy, candidate),
|
|
1626
|
-
rationale: "bounded single-dial neighbor of the current structuralRollout policy; the held-out gate decides (deterministic enumeration \u2014 the dial space is tiny and prompt-style reflective proposals are a measured zero here)"
|
|
1627
|
-
}));
|
|
1628
|
-
}
|
|
1629
|
-
};
|
|
1630
|
-
}
|
|
1631
|
-
|
|
1632
|
-
// src/improvement/improve.ts
|
|
1633
|
-
var defaultReflectionModel = "deepseek-v4-flash";
|
|
1634
|
-
function llmClientOptions(llm) {
|
|
1635
|
-
return { baseUrl: llm?.baseUrl, apiKey: llm?.apiKey };
|
|
1636
|
-
}
|
|
1637
|
-
function defaultGeneratorFor(surface, llm) {
|
|
1638
|
-
const model = llm?.model ?? defaultReflectionModel;
|
|
1639
|
-
switch (surface) {
|
|
1640
|
-
case "prompt":
|
|
1641
|
-
return gepaProposer({ llm: llmClientOptions(llm), model, target: "agent system prompt" });
|
|
1642
|
-
case "skills":
|
|
1643
|
-
return skillOptProposer({ llm: llmClientOptions(llm), model, target: "agent skill document" });
|
|
1644
|
-
case "rollout-policy":
|
|
1645
|
-
return rolloutPolicyProposer();
|
|
1646
|
-
default:
|
|
1647
|
-
return void 0;
|
|
1648
|
-
}
|
|
1649
|
-
}
|
|
1650
|
-
function baselineSurfaceFor(profile, surface, skills) {
|
|
1651
|
-
switch (surface) {
|
|
1652
|
-
case "prompt":
|
|
1653
|
-
return profile.prompt?.systemPrompt ?? "";
|
|
1654
|
-
case "skills":
|
|
1655
|
-
return skills?.document ?? JSON.stringify(profile.resources?.skills ?? []);
|
|
1656
|
-
case "tools":
|
|
1657
|
-
return JSON.stringify(profile.tools ?? {});
|
|
1658
|
-
case "mcp":
|
|
1659
|
-
return JSON.stringify(profile.mcp ?? {});
|
|
1660
|
-
case "hooks":
|
|
1661
|
-
return JSON.stringify(profile.hooks ?? {});
|
|
1662
|
-
case "rollout-policy": {
|
|
1663
|
-
const policy = structuralRolloutPolicyFromProfile(profile);
|
|
1664
|
-
return policy ? serializeRolloutPolicy(policy) : "";
|
|
1665
|
-
}
|
|
1666
|
-
case "code":
|
|
1667
|
-
return "";
|
|
1668
|
-
}
|
|
1669
|
-
}
|
|
1670
|
-
function generationFailureDistiller(staticFindings) {
|
|
1671
|
-
const CAP = 12;
|
|
1672
|
-
return async (input) => {
|
|
1673
|
-
const failures = [];
|
|
1674
|
-
for (const candidate of input.candidates) {
|
|
1675
|
-
for (const rawCell of candidate.campaign.cells) {
|
|
1676
|
-
const cell = rawCell;
|
|
1677
|
-
const scenario = String(cell.scenarioId ?? "unknown");
|
|
1678
|
-
const error = typeof cell.error === "string" ? cell.error : void 0;
|
|
1679
|
-
const judgeScores = cell.judgeScores && typeof cell.judgeScores === "object" ? Object.values(
|
|
1680
|
-
cell.judgeScores
|
|
1681
|
-
) : [];
|
|
1682
|
-
const composite = judgeScores.length === 0 ? 0 : judgeScores.reduce((sum, j) => sum + (j.composite ?? 0), 0) / judgeScores.length;
|
|
1683
|
-
if (!error && composite >= 0.999) continue;
|
|
1684
|
-
const notes = judgeScores.map((j) => j.notes).filter((n) => typeof n === "string" && n.length > 0).join("; ").slice(0, 400);
|
|
1685
|
-
failures.push({
|
|
1686
|
-
scenario,
|
|
1687
|
-
composite: Number(composite.toFixed(3)),
|
|
1688
|
-
notes,
|
|
1689
|
-
...error ? { error: error.slice(0, 200) } : {}
|
|
1690
|
-
});
|
|
1691
|
-
}
|
|
1692
|
-
}
|
|
1693
|
-
if (failures.length === 0) return staticFindings;
|
|
1694
|
-
failures.sort((a, b) => a.composite - b.composite);
|
|
1695
|
-
return failures.slice(0, CAP);
|
|
1696
|
-
};
|
|
1697
|
-
}
|
|
1698
|
-
function codeProposerFor(surface, code) {
|
|
1699
|
-
if (surface !== "code" || !code) return void 0;
|
|
1700
|
-
const generator = code.generator ?? agenticGenerator({
|
|
1701
|
-
...code.harness ? { harness: code.harness } : {},
|
|
1702
|
-
...code.verify ? { verify: code.verify } : {},
|
|
1703
|
-
...code.timeoutMs ? { timeoutMs: code.timeoutMs } : {}
|
|
1704
|
-
});
|
|
1705
|
-
return improvementDriver({
|
|
1706
|
-
worktree: gitWorktreeAdapter({
|
|
1707
|
-
repoRoot: code.repoRoot,
|
|
1708
|
-
...code.worktreeDir ? { worktreeDir: code.worktreeDir } : {}
|
|
1709
|
-
}),
|
|
1710
|
-
generator,
|
|
1711
|
-
...code.baseRef ? { baseRef: code.baseRef } : {}
|
|
1712
|
-
});
|
|
1713
|
-
}
|
|
1714
|
-
function parseWinnerJson(winner, surface) {
|
|
1715
|
-
try {
|
|
1716
|
-
return JSON.parse(winner);
|
|
1717
|
-
} catch (cause) {
|
|
1718
|
-
throw new ConfigError(
|
|
1719
|
-
`improve(): the shipped '${surface}' winner is not valid JSON, so it cannot be applied back to the profile: ${cause.message}`
|
|
1720
|
-
);
|
|
1721
|
-
}
|
|
1722
|
-
}
|
|
1723
|
-
function applyWinnerToProfile(profile, surface, winner) {
|
|
1724
|
-
if (typeof winner !== "string") return profile;
|
|
1725
|
-
switch (surface) {
|
|
1726
|
-
case "prompt":
|
|
1727
|
-
return { ...profile, prompt: { ...profile.prompt, systemPrompt: winner } };
|
|
1728
|
-
case "skills":
|
|
1729
|
-
return {
|
|
1730
|
-
...profile,
|
|
1731
|
-
resources: { ...profile.resources, skills: parseWinnerJson(winner, surface) }
|
|
1732
|
-
};
|
|
1733
|
-
case "tools":
|
|
1734
|
-
return { ...profile, tools: parseWinnerJson(winner, surface) };
|
|
1735
|
-
case "mcp":
|
|
1736
|
-
return { ...profile, mcp: parseWinnerJson(winner, surface) };
|
|
1737
|
-
case "hooks":
|
|
1738
|
-
return { ...profile, hooks: parseWinnerJson(winner, surface) };
|
|
1739
|
-
case "rollout-policy": {
|
|
1740
|
-
const policy = normalizeRolloutPolicy(parseWinnerJson(winner, surface));
|
|
1741
|
-
if (!policy) {
|
|
1742
|
-
throw new ConfigError(
|
|
1743
|
-
`improve(): the shipped 'rollout-policy' winner is not a valid StructuralRolloutPolicy (integer k >= 1, repairRounds >= 0, testgen >= 0), so it cannot be applied: ${winner}`
|
|
1744
|
-
);
|
|
1745
|
-
}
|
|
1746
|
-
return applyRolloutPolicyToProfile(profile, policy);
|
|
1747
|
-
}
|
|
1748
|
-
case "code":
|
|
1749
|
-
return profile;
|
|
1750
|
-
}
|
|
1751
|
-
}
|
|
1752
|
-
async function improve(profile, findings, opts) {
|
|
1753
|
-
const surface = opts.surface ?? "prompt";
|
|
1754
|
-
const gate = opts.gate ?? "holdout";
|
|
1755
|
-
assertModelAllowed(opts.llm?.model ?? defaultReflectionModel, opts.allowedModels);
|
|
1756
|
-
const proposer = opts.generator ?? defaultGeneratorFor(surface, opts.llm) ?? codeProposerFor(surface, opts.code);
|
|
1757
|
-
if (!proposer) {
|
|
1758
|
-
throw new ConfigError(
|
|
1759
|
-
surface === "code" ? `improve(): surface 'code' needs either opts.generator or opts.code ({ repoRoot, ... }) \u2014 there is no safe zero-config repo to invent` : `improve(): surface '${surface}' has no default generator \u2014 pass opts.generator (a SurfaceProposer) explicitly`
|
|
1760
|
-
);
|
|
1761
|
-
}
|
|
1762
|
-
const budget = gate === "none" ? { ...opts.budget, generations: 0 } : { ...opts.budget };
|
|
1763
|
-
const raw = await selfImprove({
|
|
1764
|
-
agent: opts.agent,
|
|
1765
|
-
scenarios: opts.scenarios,
|
|
1766
|
-
judge: opts.judge,
|
|
1767
|
-
baselineSurface: baselineSurfaceFor(profile, surface, opts.skills),
|
|
1768
|
-
proposer,
|
|
1769
|
-
budget,
|
|
1770
|
-
llm: opts.llm,
|
|
1771
|
-
findings,
|
|
1772
|
-
...opts.runDir !== void 0 ? { runDir: opts.runDir } : {},
|
|
1773
|
-
...opts.storage !== void 0 ? { storage: opts.storage } : {},
|
|
1774
|
-
...opts.analyzeGeneration === null ? {} : {
|
|
1775
|
-
analyzeGeneration: opts.analyzeGeneration ?? (opts.rawTraceContext ? rawTraceDistiller({ fallbackFindings: findings }) : generationFailureDistiller(findings))
|
|
1776
|
-
}
|
|
1777
|
-
});
|
|
1778
|
-
const shipped = raw.gateDecision === "ship";
|
|
1779
|
-
const usedSkillDocument = surface === "skills" && opts.skills !== void 0;
|
|
1780
|
-
if (shipped && usedSkillDocument && typeof raw.winner.surface === "string") {
|
|
1781
|
-
opts.skills?.writeBack?.(raw.winner.surface);
|
|
1782
|
-
}
|
|
1783
|
-
const nextProfile = shipped && !usedSkillDocument ? applyWinnerToProfile(profile, surface, raw.winner.surface) : profile;
|
|
1784
|
-
return { profile: nextProfile, shipped, lift: raw.lift, gateDecision: raw.gateDecision, raw };
|
|
1785
|
-
}
|
|
1786
|
-
|
|
1787
1323
|
// src/improvement/reflective-generator.ts
|
|
1788
1324
|
import { spawnSync } from "child_process";
|
|
1789
1325
|
function reflectiveGenerator(opts) {
|
|
@@ -2359,374 +1895,6 @@ function randomSuffix() {
|
|
|
2359
1895
|
return Math.random().toString(36).slice(2, 10);
|
|
2360
1896
|
}
|
|
2361
1897
|
|
|
2362
|
-
// src/sanitize.ts
|
|
2363
|
-
function sanitizeKnowledgeReadinessReport(report, options = {}) {
|
|
2364
|
-
return {
|
|
2365
|
-
taskId: report.taskId,
|
|
2366
|
-
readinessScore: report.readinessScore,
|
|
2367
|
-
recommendedAction: report.recommendedAction,
|
|
2368
|
-
severity: report.severity,
|
|
2369
|
-
reason: report.reason,
|
|
2370
|
-
blockingMissingRequirements: report.blockingMissingRequirements.map(
|
|
2371
|
-
(requirement) => sanitizeKnowledgeRequirement(requirement, options)
|
|
2372
|
-
),
|
|
2373
|
-
nonBlockingGaps: report.nonBlockingGaps.map(
|
|
2374
|
-
(requirement) => sanitizeKnowledgeRequirement(requirement, options)
|
|
2375
|
-
),
|
|
2376
|
-
evidenceCount: report.bundle.evidenceIds.length,
|
|
2377
|
-
evidenceIds: options.includeEvidenceIds ? report.bundle.evidenceIds : void 0,
|
|
2378
|
-
missingRequirementIds: report.bundle.missing.map((requirement) => requirement.id)
|
|
2379
|
-
};
|
|
2380
|
-
}
|
|
2381
|
-
function sanitizeAgentRuntimeEvent(event, options = {}) {
|
|
2382
|
-
const base = { type: event.type, task: sanitizeTask(event.task, options) };
|
|
2383
|
-
if (event.type === "readiness_start" || event.type === "task_start" || event.type === "control_start") {
|
|
2384
|
-
return event.type === "control_start" ? { ...base, knowledge: sanitizeKnowledgeReadinessReport(event.knowledge, options) } : base;
|
|
2385
|
-
}
|
|
2386
|
-
if (event.type === "readiness_end") {
|
|
2387
|
-
return { ...base, knowledge: sanitizeKnowledgeReadinessReport(event.knowledge, options) };
|
|
2388
|
-
}
|
|
2389
|
-
if (event.type === "questions_start") {
|
|
2390
|
-
return {
|
|
2391
|
-
...base,
|
|
2392
|
-
questions: event.questions.map((question) => sanitizeQuestion(question, options))
|
|
2393
|
-
};
|
|
2394
|
-
}
|
|
2395
|
-
if (event.type === "questions_end") {
|
|
2396
|
-
return {
|
|
2397
|
-
...base,
|
|
2398
|
-
questions: event.questions.map((question) => sanitizeQuestion(question, options)),
|
|
2399
|
-
userAnswers: options.includeUserAnswers ? event.userAnswers : redactRecord(event.userAnswers)
|
|
2400
|
-
};
|
|
2401
|
-
}
|
|
2402
|
-
if (event.type === "acquisition_start") {
|
|
2403
|
-
return { ...base, acquisitionPlans: event.acquisitionPlans.map(sanitizeAcquisitionPlan) };
|
|
2404
|
-
}
|
|
2405
|
-
if (event.type === "acquisition_end") {
|
|
2406
|
-
return {
|
|
2407
|
-
...base,
|
|
2408
|
-
acquisitionPlans: event.acquisitionPlans.map(sanitizeAcquisitionPlan),
|
|
2409
|
-
acquiredEvidenceCount: event.acquiredEvidenceIds.length,
|
|
2410
|
-
acquiredEvidenceIds: options.includeEvidenceIds ? event.acquiredEvidenceIds : void 0
|
|
2411
|
-
};
|
|
2412
|
-
}
|
|
2413
|
-
if (event.type === "control_step") {
|
|
2414
|
-
return { ...base, step: sanitizeControlStep(event.step, options) };
|
|
2415
|
-
}
|
|
2416
|
-
if (event.type === "control_end") {
|
|
2417
|
-
return { ...base, control: sanitizeControlRun(event.control, options) };
|
|
2418
|
-
}
|
|
2419
|
-
return { ...base, status: event.status, reason: event.reason };
|
|
2420
|
-
}
|
|
2421
|
-
function sanitizeRuntimeStreamEvent(event, options = {}) {
|
|
2422
|
-
const withTask = "task" in event && event.task ? { task: sanitizeTask(event.task, options) } : {};
|
|
2423
|
-
const withSession = "session" in event && event.session ? { session: sanitizeRuntimeSession(event.session, options) } : {};
|
|
2424
|
-
if (event.type === "readiness_end") {
|
|
2425
|
-
return {
|
|
2426
|
-
type: event.type,
|
|
2427
|
-
...withTask,
|
|
2428
|
-
timestamp: event.timestamp,
|
|
2429
|
-
decision: event.decision,
|
|
2430
|
-
knowledge: sanitizeKnowledgeReadinessReport(event.knowledge, options)
|
|
2431
|
-
};
|
|
2432
|
-
}
|
|
2433
|
-
if (event.type === "questions_start") {
|
|
2434
|
-
return {
|
|
2435
|
-
type: event.type,
|
|
2436
|
-
...withTask,
|
|
2437
|
-
timestamp: event.timestamp,
|
|
2438
|
-
questions: event.questions.map((question) => sanitizeQuestion(question, options))
|
|
2439
|
-
};
|
|
2440
|
-
}
|
|
2441
|
-
if (event.type === "questions_end") {
|
|
2442
|
-
return {
|
|
2443
|
-
type: event.type,
|
|
2444
|
-
...withTask,
|
|
2445
|
-
timestamp: event.timestamp,
|
|
2446
|
-
questions: event.questions.map((question) => sanitizeQuestion(question, options)),
|
|
2447
|
-
userAnswers: options.includeUserAnswers ? event.userAnswers : redactRecord(event.userAnswers)
|
|
2448
|
-
};
|
|
2449
|
-
}
|
|
2450
|
-
if (event.type === "acquisition_start") {
|
|
2451
|
-
return {
|
|
2452
|
-
type: event.type,
|
|
2453
|
-
...withTask,
|
|
2454
|
-
timestamp: event.timestamp,
|
|
2455
|
-
acquisitionPlans: event.acquisitionPlans.map(sanitizeAcquisitionPlan)
|
|
2456
|
-
};
|
|
2457
|
-
}
|
|
2458
|
-
if (event.type === "acquisition_end") {
|
|
2459
|
-
return {
|
|
2460
|
-
type: event.type,
|
|
2461
|
-
...withTask,
|
|
2462
|
-
timestamp: event.timestamp,
|
|
2463
|
-
acquisitionPlans: event.acquisitionPlans.map(sanitizeAcquisitionPlan),
|
|
2464
|
-
acquiredEvidenceCount: event.acquiredEvidenceIds.length,
|
|
2465
|
-
acquiredEvidenceIds: options.includeEvidenceIds ? event.acquiredEvidenceIds : void 0
|
|
2466
|
-
};
|
|
2467
|
-
}
|
|
2468
|
-
if (event.type === "tool_call") {
|
|
2469
|
-
return {
|
|
2470
|
-
type: event.type,
|
|
2471
|
-
...withTask,
|
|
2472
|
-
...withSession,
|
|
2473
|
-
timestamp: event.timestamp,
|
|
2474
|
-
toolName: event.toolName,
|
|
2475
|
-
toolCallId: event.toolCallId,
|
|
2476
|
-
args: options.includeControlPayloads ? event.args : void 0
|
|
2477
|
-
};
|
|
2478
|
-
}
|
|
2479
|
-
if (event.type === "tool_result") {
|
|
2480
|
-
return {
|
|
2481
|
-
type: event.type,
|
|
2482
|
-
...withTask,
|
|
2483
|
-
...withSession,
|
|
2484
|
-
timestamp: event.timestamp,
|
|
2485
|
-
toolName: event.toolName,
|
|
2486
|
-
toolCallId: event.toolCallId,
|
|
2487
|
-
result: options.includeControlPayloads ? event.result : void 0
|
|
2488
|
-
};
|
|
2489
|
-
}
|
|
2490
|
-
if (event.type === "llm_call") {
|
|
2491
|
-
return {
|
|
2492
|
-
type: event.type,
|
|
2493
|
-
...withTask,
|
|
2494
|
-
...withSession,
|
|
2495
|
-
timestamp: event.timestamp,
|
|
2496
|
-
model: event.model,
|
|
2497
|
-
tokensIn: event.tokensIn,
|
|
2498
|
-
tokensOut: event.tokensOut,
|
|
2499
|
-
costUsd: event.costUsd,
|
|
2500
|
-
latencyMs: event.latencyMs,
|
|
2501
|
-
finishReason: event.finishReason
|
|
2502
|
-
};
|
|
2503
|
-
}
|
|
2504
|
-
if (event.type === "artifact") {
|
|
2505
|
-
return {
|
|
2506
|
-
type: event.type,
|
|
2507
|
-
...withTask,
|
|
2508
|
-
...withSession,
|
|
2509
|
-
timestamp: event.timestamp,
|
|
2510
|
-
artifactId: event.artifactId,
|
|
2511
|
-
name: event.name,
|
|
2512
|
-
mimeType: event.mimeType,
|
|
2513
|
-
uri: options.includeEvidenceIds ? event.uri : void 0,
|
|
2514
|
-
content: options.includeControlPayloads ? event.content : void 0,
|
|
2515
|
-
metadata: options.includeMetadata ? event.metadata : void 0
|
|
2516
|
-
};
|
|
2517
|
-
}
|
|
2518
|
-
if (event.type === "proposal_created") {
|
|
2519
|
-
return {
|
|
2520
|
-
type: event.type,
|
|
2521
|
-
...withTask,
|
|
2522
|
-
...withSession,
|
|
2523
|
-
timestamp: event.timestamp,
|
|
2524
|
-
proposalId: event.proposalId,
|
|
2525
|
-
title: options.includeControlPayloads ? event.title : void 0,
|
|
2526
|
-
content: options.includeControlPayloads ? event.content : void 0,
|
|
2527
|
-
status: event.status
|
|
2528
|
-
};
|
|
2529
|
-
}
|
|
2530
|
-
if (event.type === "final") {
|
|
2531
|
-
const sanitizedError = event.error !== void 0 ? {
|
|
2532
|
-
kind: event.error.kind,
|
|
2533
|
-
message: event.error.message,
|
|
2534
|
-
status: event.error.status,
|
|
2535
|
-
body: options.includeControlPayloads ? event.error.body : void 0
|
|
2536
|
-
} : void 0;
|
|
2537
|
-
return {
|
|
2538
|
-
type: event.type,
|
|
2539
|
-
...withTask,
|
|
2540
|
-
...withSession,
|
|
2541
|
-
timestamp: event.timestamp,
|
|
2542
|
-
status: event.status,
|
|
2543
|
-
reason: event.reason,
|
|
2544
|
-
text: options.includeControlPayloads ? event.text : void 0,
|
|
2545
|
-
metadata: options.includeMetadata ? event.metadata : void 0,
|
|
2546
|
-
...sanitizedError !== void 0 ? { error: sanitizedError } : {}
|
|
2547
|
-
};
|
|
2548
|
-
}
|
|
2549
|
-
return {
|
|
2550
|
-
type: event.type,
|
|
2551
|
-
...withTask,
|
|
2552
|
-
...withSession,
|
|
2553
|
-
timestamp: "timestamp" in event ? event.timestamp : void 0,
|
|
2554
|
-
...pickPublicStreamFields(event)
|
|
2555
|
-
};
|
|
2556
|
-
}
|
|
2557
|
-
function sanitizeTask(task, options) {
|
|
2558
|
-
return {
|
|
2559
|
-
id: task.id,
|
|
2560
|
-
intent: task.intent,
|
|
2561
|
-
domain: task.domain,
|
|
2562
|
-
inputs: options.includeInputs ? task.inputs : task.inputs ? "[redacted]" : void 0,
|
|
2563
|
-
requiredKnowledge: task.requiredKnowledge?.map(
|
|
2564
|
-
(requirement) => sanitizeKnowledgeRequirement(requirement, options)
|
|
2565
|
-
),
|
|
2566
|
-
metadata: options.includeMetadata ? task.metadata : task.metadata ? "[redacted]" : void 0
|
|
2567
|
-
};
|
|
2568
|
-
}
|
|
2569
|
-
function sanitizeRuntimeSession(session, options) {
|
|
2570
|
-
return {
|
|
2571
|
-
id: session.id,
|
|
2572
|
-
backend: session.backend,
|
|
2573
|
-
status: session.status,
|
|
2574
|
-
hasResumeToken: Boolean(session.resumeToken),
|
|
2575
|
-
createdAt: session.createdAt,
|
|
2576
|
-
updatedAt: session.updatedAt,
|
|
2577
|
-
metadata: options.includeMetadata ? session.metadata : session.metadata ? "[redacted]" : void 0
|
|
2578
|
-
};
|
|
2579
|
-
}
|
|
2580
|
-
function sanitizeKnowledgeRequirement(requirement, options) {
|
|
2581
|
-
const includeDescription = options.includeRequirementDescriptions && requirement.sensitivity !== "secret";
|
|
2582
|
-
return {
|
|
2583
|
-
id: requirement.id,
|
|
2584
|
-
description: includeDescription ? requirement.description : void 0,
|
|
2585
|
-
requiredFor: requirement.requiredFor,
|
|
2586
|
-
category: requirement.category,
|
|
2587
|
-
acquisitionMode: requirement.acquisitionMode,
|
|
2588
|
-
importance: requirement.importance,
|
|
2589
|
-
freshness: requirement.freshness,
|
|
2590
|
-
sensitivity: requirement.sensitivity,
|
|
2591
|
-
confidenceNeeded: requirement.confidenceNeeded,
|
|
2592
|
-
currentConfidence: requirement.currentConfidence,
|
|
2593
|
-
evidenceCount: requirement.evidenceIds.length,
|
|
2594
|
-
evidenceIds: options.includeEvidenceIds ? requirement.evidenceIds : void 0,
|
|
2595
|
-
fallbackPolicy: requirement.fallbackPolicy
|
|
2596
|
-
};
|
|
2597
|
-
}
|
|
2598
|
-
function sanitizeQuestion(question, options) {
|
|
2599
|
-
return {
|
|
2600
|
-
id: question.id,
|
|
2601
|
-
question: options.includeRequirementDescriptions && question.answerType !== "credential" ? question.question : void 0,
|
|
2602
|
-
reason: options.includeRequirementDescriptions ? question.reason : void 0,
|
|
2603
|
-
requirementId: question.requirementId,
|
|
2604
|
-
importance: question.importance,
|
|
2605
|
-
answerType: question.answerType,
|
|
2606
|
-
impactIfUnknown: options.includeRequirementDescriptions ? question.impactIfUnknown : void 0,
|
|
2607
|
-
optionCount: question.options?.length ?? 0
|
|
2608
|
-
};
|
|
2609
|
-
}
|
|
2610
|
-
function sanitizeAcquisitionPlan(plan) {
|
|
2611
|
-
return {
|
|
2612
|
-
id: plan.id,
|
|
2613
|
-
requirementIds: plan.requirementIds,
|
|
2614
|
-
mode: plan.mode,
|
|
2615
|
-
priority: plan.priority,
|
|
2616
|
-
expectedEvidenceCount: plan.expectedEvidenceIds?.length ?? 0,
|
|
2617
|
-
questionCount: plan.questions?.length ?? 0
|
|
2618
|
-
};
|
|
2619
|
-
}
|
|
2620
|
-
function sanitizeControlStep(step, options) {
|
|
2621
|
-
const actionOutcome = step.actionOutcome;
|
|
2622
|
-
return {
|
|
2623
|
-
index: step.index,
|
|
2624
|
-
decisionType: step.decision.type,
|
|
2625
|
-
reason: step.decision.reason,
|
|
2626
|
-
action: options.includeControlPayloads && step.decision.type === "continue" ? step.decision.action : void 0,
|
|
2627
|
-
result: options.includeControlPayloads && actionOutcome?.ok ? actionOutcome.result : void 0,
|
|
2628
|
-
actionOk: actionOutcome?.ok,
|
|
2629
|
-
actionError: actionOutcome?.ok === false ? actionOutcome.error : void 0,
|
|
2630
|
-
durationMs: actionOutcome?.durationMs,
|
|
2631
|
-
evalsBefore: summarizeEvals(step.evalsBefore, options),
|
|
2632
|
-
evalsAfter: summarizeEvals(step.evalsAfter, options),
|
|
2633
|
-
startedAt: step.startedAt,
|
|
2634
|
-
endedAt: step.endedAt
|
|
2635
|
-
};
|
|
2636
|
-
}
|
|
2637
|
-
function sanitizeControlRun(control, options) {
|
|
2638
|
-
return {
|
|
2639
|
-
pass: control.pass,
|
|
2640
|
-
completed: control.completed,
|
|
2641
|
-
reason: control.reason,
|
|
2642
|
-
score: control.score,
|
|
2643
|
-
stepCount: control.steps.length,
|
|
2644
|
-
wallMs: control.wallMs,
|
|
2645
|
-
spentCostUsd: control.spentCostUsd,
|
|
2646
|
-
failureClass: control.failureClass,
|
|
2647
|
-
stoppedBy: control.stoppedBy,
|
|
2648
|
-
runId: control.runId,
|
|
2649
|
-
runtimeErrorCount: control.runtimeErrors.length,
|
|
2650
|
-
finalEvals: summarizeEvals(control.finalEvals, options)
|
|
2651
|
-
};
|
|
2652
|
-
}
|
|
2653
|
-
function summarizeEvals(evals, options) {
|
|
2654
|
-
return evals.map((evalResult) => ({
|
|
2655
|
-
id: evalResult.id,
|
|
2656
|
-
passed: evalResult.passed,
|
|
2657
|
-
score: evalResult.score,
|
|
2658
|
-
severity: evalResult.severity,
|
|
2659
|
-
objective: evalResult.objective,
|
|
2660
|
-
detail: options.includeEvalDetails ? evalResult.detail : void 0,
|
|
2661
|
-
evidence: options.includeEvalDetails ? evalResult.evidence : void 0
|
|
2662
|
-
}));
|
|
2663
|
-
}
|
|
2664
|
-
function redactRecord(record) {
|
|
2665
|
-
return Object.fromEntries(Object.keys(record).map((key) => [key, "[redacted]"]));
|
|
2666
|
-
}
|
|
2667
|
-
function pickPublicStreamFields(event) {
|
|
2668
|
-
if (event.type === "session_created" || event.type === "session_resumed") return {};
|
|
2669
|
-
if (event.type === "backend_start" || event.type === "backend_end")
|
|
2670
|
-
return { backend: event.backend };
|
|
2671
|
-
if (event.type === "backend_error") {
|
|
2672
|
-
const sanitizedError = event.error !== void 0 ? {
|
|
2673
|
-
kind: event.error.kind,
|
|
2674
|
-
status: event.error.status
|
|
2675
|
-
} : void 0;
|
|
2676
|
-
return {
|
|
2677
|
-
backend: event.backend,
|
|
2678
|
-
message: event.message,
|
|
2679
|
-
recoverable: event.recoverable,
|
|
2680
|
-
...sanitizedError !== void 0 ? { error: sanitizedError } : {}
|
|
2681
|
-
};
|
|
2682
|
-
}
|
|
2683
|
-
if (event.type === "task_end") return { status: event.status, reason: event.reason };
|
|
2684
|
-
if (event.type === "text_delta" || event.type === "reasoning_delta") return { text: event.text };
|
|
2685
|
-
return {};
|
|
2686
|
-
}
|
|
2687
|
-
function createRuntimeEventCollector(options = {}) {
|
|
2688
|
-
const events = [];
|
|
2689
|
-
return {
|
|
2690
|
-
events,
|
|
2691
|
-
onEvent: (event) => {
|
|
2692
|
-
events.push(sanitizeAgentRuntimeEvent(event, options));
|
|
2693
|
-
}
|
|
2694
|
-
};
|
|
2695
|
-
}
|
|
2696
|
-
function createRuntimeStreamEventCollector(options = {}) {
|
|
2697
|
-
const events = [];
|
|
2698
|
-
const eventCountsByType = {};
|
|
2699
|
-
let firstSessionId;
|
|
2700
|
-
let finalStatus;
|
|
2701
|
-
let finalReason;
|
|
2702
|
-
let finalText = "";
|
|
2703
|
-
return {
|
|
2704
|
-
events,
|
|
2705
|
-
onEvent: (event) => {
|
|
2706
|
-
events.push(sanitizeRuntimeStreamEvent(event, options));
|
|
2707
|
-
eventCountsByType[event.type] = (eventCountsByType[event.type] ?? 0) + 1;
|
|
2708
|
-
if (event.type === "text_delta") finalText += event.text;
|
|
2709
|
-
if (!firstSessionId && (event.type === "session_created" || event.type === "session_resumed")) {
|
|
2710
|
-
firstSessionId = event.session.id;
|
|
2711
|
-
}
|
|
2712
|
-
if (event.type === "final") {
|
|
2713
|
-
finalStatus = event.status;
|
|
2714
|
-
finalReason = event.reason;
|
|
2715
|
-
}
|
|
2716
|
-
},
|
|
2717
|
-
summary() {
|
|
2718
|
-
return {
|
|
2719
|
-
eventCount: events.length,
|
|
2720
|
-
eventCountsByType: { ...eventCountsByType },
|
|
2721
|
-
firstSessionId,
|
|
2722
|
-
finalStatus,
|
|
2723
|
-
finalReason,
|
|
2724
|
-
finalText
|
|
2725
|
-
};
|
|
2726
|
-
}
|
|
2727
|
-
};
|
|
2728
|
-
}
|
|
2729
|
-
|
|
2730
1898
|
// src/sse.ts
|
|
2731
1899
|
function encodeServerSentEvent(data, options = {}) {
|
|
2732
1900
|
const lines = [];
|
|
@@ -3247,6 +2415,8 @@ function randomSuffix2(len = 8) {
|
|
|
3247
2415
|
export {
|
|
3248
2416
|
AgentEvalError,
|
|
3249
2417
|
BackendTransportError,
|
|
2418
|
+
CANDIDATE_TRACE_ENV,
|
|
2419
|
+
CANDIDATE_TRACE_TAGS,
|
|
3250
2420
|
CircuitBreakerState,
|
|
3251
2421
|
CircuitOpenError,
|
|
3252
2422
|
ConfigError,
|
|
@@ -3255,8 +2425,10 @@ export {
|
|
|
3255
2425
|
DELEGATED_LOOP_MODES,
|
|
3256
2426
|
DeadlineExceededError,
|
|
3257
2427
|
FORWARD_HEADERS,
|
|
2428
|
+
FileAgentCandidateExecutionClaimStore,
|
|
3258
2429
|
FileConversationJournal,
|
|
3259
2430
|
INTELLIGENCE_WIRE_VERSION,
|
|
2431
|
+
InMemoryAgentCandidateExecutionClaimStore,
|
|
3260
2432
|
InMemoryConversationJournal,
|
|
3261
2433
|
InMemoryRuntimeSessionStore,
|
|
3262
2434
|
JudgeError,
|
|
@@ -3275,6 +2447,8 @@ export {
|
|
|
3275
2447
|
buildForwardHeaders,
|
|
3276
2448
|
buildLoopOtelSpans,
|
|
3277
2449
|
buildLoopSpanNodes,
|
|
2450
|
+
buildRuntimeEventOtelSpans,
|
|
2451
|
+
candidateExecutionClaim,
|
|
3278
2452
|
cleanModelId,
|
|
3279
2453
|
commandVerifier,
|
|
3280
2454
|
composeRuntimeHooks,
|
|
@@ -3284,6 +2458,7 @@ export {
|
|
|
3284
2458
|
createIterableBackend,
|
|
3285
2459
|
createOpenAICompatibleBackend,
|
|
3286
2460
|
createOtelExporter,
|
|
2461
|
+
createProtectedAgentCandidateModelPort,
|
|
3287
2462
|
createRuntimeEventCollector,
|
|
3288
2463
|
createRuntimeStreamEventCollector,
|
|
3289
2464
|
createSandboxPromptBackend,
|
|
@@ -3294,7 +2469,9 @@ export {
|
|
|
3294
2469
|
defineConversation,
|
|
3295
2470
|
defineRuntimeHooks,
|
|
3296
2471
|
deriveExecutionId,
|
|
2472
|
+
disposePreparedAgentCandidateExecution,
|
|
3297
2473
|
enumerateNeighborPolicies,
|
|
2474
|
+
executePreparedAgentCandidate,
|
|
3298
2475
|
exportEvalRuns,
|
|
3299
2476
|
formatSupervisedKnowledgeTask,
|
|
3300
2477
|
getModels,
|
|
@@ -3315,9 +2492,12 @@ export {
|
|
|
3315
2492
|
notifyRuntimeHookEvent,
|
|
3316
2493
|
parseLoopRunnerArgv,
|
|
3317
2494
|
parseRolloutPolicy,
|
|
2495
|
+
persistCandidateOutputArtifact,
|
|
2496
|
+
prepareAgentCandidateExecution,
|
|
3318
2497
|
rawTraceDistiller,
|
|
3319
2498
|
readDepth,
|
|
3320
2499
|
readinessServerSentEvent,
|
|
2500
|
+
recoverExpiredAgentCandidateExecution,
|
|
3321
2501
|
reflectiveGenerator,
|
|
3322
2502
|
researchLoopRunner,
|
|
3323
2503
|
resolveAgentBackend,
|
|
@@ -3349,6 +2529,7 @@ export {
|
|
|
3349
2529
|
toolBuildPrompt,
|
|
3350
2530
|
turnId,
|
|
3351
2531
|
validateChatModelId,
|
|
2532
|
+
verifyAgentCandidateBundle,
|
|
3352
2533
|
worktreeLoopRunner
|
|
3353
2534
|
};
|
|
3354
2535
|
//# sourceMappingURL=index.js.map
|