@tangle-network/agent-eval 0.79.0 → 0.81.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +101 -169
- package/dist/adapters/http.d.ts +2 -2
- package/dist/adapters/langchain.d.ts +2 -2
- package/dist/adapters/otel.d.ts +4 -4
- package/dist/{agent-profile-aSEaJ9Pl.d.ts → agent-profile-D0PBIWlV.d.ts} +14 -2
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +3 -3
- package/dist/{analyst-t7zZS3TV.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/belief-state/index.d.ts +524 -0
- package/dist/belief-state/index.js +1862 -0
- package/dist/belief-state/index.js.map +1 -0
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/calibration-Cpr3WaX3.d.ts +101 -0
- package/dist/campaign/index.d.ts +40 -120
- package/dist/campaign/index.js +129 -238
- package/dist/campaign/index.js.map +1 -1
- package/dist/chunk-4DIJWVUT.js +131 -0
- package/dist/chunk-4DIJWVUT.js.map +1 -0
- package/dist/{chunk-RPLZ4OIB.js → chunk-BABOZOSN.js} +7 -4
- package/dist/{chunk-RPLZ4OIB.js.map → chunk-BABOZOSN.js.map} +1 -1
- package/dist/{chunk-IHDHUN2X.js → chunk-CVVHBFGN.js} +3 -3
- package/dist/chunk-CVVHBFGN.js.map +1 -0
- package/dist/{chunk-B26KI423.js → chunk-FZWAFVAA.js} +2 -2
- package/dist/{chunk-ITBRCT73.js → chunk-IDVBLYCY.js} +2 -10
- package/dist/chunk-IDVBLYCY.js.map +1 -0
- package/dist/{chunk-5LVWPNS5.js → chunk-L5G7OUKD.js} +4 -4
- package/dist/{chunk-5LVWPNS5.js.map → chunk-L5G7OUKD.js.map} +1 -1
- package/dist/chunk-NPCTHQIO.js +91 -0
- package/dist/chunk-NPCTHQIO.js.map +1 -0
- package/dist/{chunk-GXHLRXDI.js → chunk-OTYQPHPL.js} +4 -4
- package/dist/{chunk-6REHLN5J.js → chunk-QS3RBQPI.js} +2 -2
- package/dist/{chunk-GWGO2K6Y.js → chunk-RBNA5AZT.js} +2 -2
- package/dist/chunk-S42AWHMP.js +697 -0
- package/dist/chunk-S42AWHMP.js.map +1 -0
- package/dist/chunk-VI2UW6B6.js +162 -0
- package/dist/chunk-VI2UW6B6.js.map +1 -0
- package/dist/{chunk-CF67I6QY.js → chunk-VIDQF3F5.js} +2 -2
- package/dist/{chunk-XXNIODOM.js → chunk-WJL2NJXN.js} +3 -3
- package/dist/{chunk-LB2UOI5F.js → chunk-YGYXHNAQ.js} +47 -160
- package/dist/chunk-YGYXHNAQ.js.map +1 -0
- package/dist/{chunk-KX6F6NCG.js → chunk-Z7VFTS2J.js} +2 -2
- package/dist/{chunk-ZPSKPT3V.js → chunk-ZZ2HOPME.js} +5 -2
- package/dist/chunk-ZZ2HOPME.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/code-agent-session-BRXmavYv.d.ts +80 -0
- package/dist/contract/index.d.ts +132 -18
- package/dist/contract/index.js +139 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CehLtoET.d.ts → control-GeE8OhpN.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/governance/index.d.ts +1 -1
- package/dist/hosted/index.d.ts +4 -4
- package/dist/{index-B1RKber3.d.ts → index-DE3RXAXD.d.ts} +1 -1
- package/dist/index.d.ts +79 -288
- package/dist/index.js +87 -410
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-dlpEzQDi.d.ts → insight-report-3ADTfClO.d.ts} +1 -1
- package/dist/{kind-factory-DqV2t1Xk.d.ts → kind-factory-CVecZZG_.d.ts} +2 -2
- package/dist/{llm-client-DbjLfz-K.d.ts → llm-client-CuUg2Mn3.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +6 -99
- package/dist/meta-eval/index.js +7 -76
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/off-policy-DiwuKKg7.d.ts +132 -0
- package/dist/openapi.json +1 -1
- package/dist/{outcome-store-D6KWmYvj.d.ts → outcome-store-rnXLEqSn.d.ts} +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{provenance-CEAJI9rm.d.ts → provenance-B9Q4886D.d.ts} +4 -4
- package/dist/{registry-BmEuU94S.d.ts → registry-DrEQ3Luj.d.ts} +2 -2
- package/dist/{release-report-CXXZlR8g.d.ts → release-report-hlNtD12q.d.ts} +2 -2
- package/dist/reporting.d.ts +6 -6
- package/dist/reporting.js +3 -3
- package/dist/{researcher-rInLj9De.d.ts → researcher-BLPHBbNV.d.ts} +3 -3
- package/dist/rl.d.ts +11 -141
- package/dist/rl.js +10 -124
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-CWyWWLBg.d.ts → rubric-predictive-validity-CnEl9Jc8.d.ts} +2 -2
- package/dist/{run-campaign-OVEZF24D.js → run-campaign-4Y5V5CN3.js} +3 -3
- package/dist/{run-improvement-loop-Bgu4C59E.d.ts → run-improvement-loop-D6PZOoQL.d.ts} +2 -2
- package/dist/{run-record-sItO5ftF.d.ts → run-record-De9VarXR.d.ts} +1 -1
- package/dist/{semantic-concept-judge-Du4ZVyef.d.ts → semantic-concept-judge-DIEgr_6v.d.ts} +6 -6
- package/dist/{statistics-B7yCbi9i.d.ts → statistics-CnC1FMbx.d.ts} +3 -7
- package/dist/{store-GmBE2pZZ.d.ts → store-C1YxJDEK.d.ts} +1 -1
- package/dist/{summary-report-BTaXq1TS.d.ts → summary-report-Db0dDSWP.d.ts} +1 -1
- package/dist/traces.d.ts +5 -5
- package/dist/{types-DRvV0zRo.d.ts → types-Cu3u_x59.d.ts} +3 -3
- package/dist/{types-QHG0KnkF.d.ts → types-D7lLRYe9.d.ts} +2 -2
- package/dist/wire/index.js +2 -2
- package/dist/workflow/index.d.ts +7 -7
- package/dist/workflow/index.js +1 -1
- package/docs/concepts.md +1 -0
- package/docs/research/belief-state-agent-eval-roadmap.md +590 -0
- package/docs/research/research-roadmap.md +1 -0
- package/docs/self-improvement-map.md +111 -0
- package/package.json +7 -2
- package/dist/chunk-IHDHUN2X.js.map +0 -1
- package/dist/chunk-ITBRCT73.js.map +0 -1
- package/dist/chunk-LB2UOI5F.js.map +0 -1
- package/dist/chunk-ZPSKPT3V.js.map +0 -1
- /package/dist/{chunk-B26KI423.js.map → chunk-FZWAFVAA.js.map} +0 -0
- /package/dist/{chunk-GXHLRXDI.js.map → chunk-OTYQPHPL.js.map} +0 -0
- /package/dist/{chunk-6REHLN5J.js.map → chunk-QS3RBQPI.js.map} +0 -0
- /package/dist/{chunk-GWGO2K6Y.js.map → chunk-RBNA5AZT.js.map} +0 -0
- /package/dist/{chunk-CF67I6QY.js.map → chunk-VIDQF3F5.js.map} +0 -0
- /package/dist/{chunk-XXNIODOM.js.map → chunk-WJL2NJXN.js.map} +0 -0
- /package/dist/{chunk-KX6F6NCG.js.map → chunk-Z7VFTS2J.js.map} +0 -0
- /package/dist/{run-campaign-OVEZF24D.js.map → run-campaign-4Y5V5CN3.js.map} +0 -0
package/dist/campaign/index.js
CHANGED
|
@@ -10,14 +10,12 @@ import {
|
|
|
10
10
|
paretoPolicy,
|
|
11
11
|
paretoSignificanceGate,
|
|
12
12
|
runEval
|
|
13
|
-
} from "../chunk-
|
|
13
|
+
} from "../chunk-OTYQPHPL.js";
|
|
14
14
|
import {
|
|
15
15
|
agentProfileHash,
|
|
16
|
-
estimateCost,
|
|
17
16
|
extractProducedState,
|
|
18
|
-
isModelPriced,
|
|
19
17
|
verifyCompletion
|
|
20
|
-
} from "../chunk-
|
|
18
|
+
} from "../chunk-YGYXHNAQ.js";
|
|
21
19
|
import {
|
|
22
20
|
buildLoopProvenanceRecord,
|
|
23
21
|
campaignBreakdown,
|
|
@@ -39,26 +37,30 @@ import {
|
|
|
39
37
|
runOptimization,
|
|
40
38
|
surfaceContentHash,
|
|
41
39
|
surfaceHash
|
|
42
|
-
} from "../chunk-
|
|
40
|
+
} from "../chunk-BABOZOSN.js";
|
|
43
41
|
import {
|
|
44
42
|
assertRealBackend,
|
|
45
43
|
fsCampaignStorage,
|
|
46
44
|
inMemoryCampaignStorage,
|
|
47
45
|
runCampaign,
|
|
48
46
|
summarizeBackendIntegrity
|
|
49
|
-
} from "../chunk-
|
|
47
|
+
} from "../chunk-ZZ2HOPME.js";
|
|
48
|
+
import {
|
|
49
|
+
estimateCost,
|
|
50
|
+
isModelPriced
|
|
51
|
+
} from "../chunk-VI2UW6B6.js";
|
|
50
52
|
import {
|
|
51
53
|
AnalystRegistry,
|
|
52
54
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
53
55
|
createTraceAnalystKind
|
|
54
|
-
} from "../chunk-
|
|
56
|
+
} from "../chunk-VIDQF3F5.js";
|
|
55
57
|
import "../chunk-YV7J7X5N.js";
|
|
56
58
|
import {
|
|
57
59
|
callLlm
|
|
58
|
-
} from "../chunk-
|
|
60
|
+
} from "../chunk-CVVHBFGN.js";
|
|
59
61
|
import {
|
|
60
62
|
pairedBootstrap
|
|
61
|
-
} from "../chunk-
|
|
63
|
+
} from "../chunk-IDVBLYCY.js";
|
|
62
64
|
import "../chunk-GGE4NNQT.js";
|
|
63
65
|
import {
|
|
64
66
|
OtlpFileTraceStore,
|
|
@@ -155,16 +157,25 @@ function surfaceToText2(surface) {
|
|
|
155
157
|
`curator driver: surface must be a string prompt, got a ${surface.kind}-tier surface (${surface.worktreeRef}) \u2014 curation is prompt-tier`
|
|
156
158
|
);
|
|
157
159
|
}
|
|
160
|
+
function extractBlockBody(text, startMarker, endMarker) {
|
|
161
|
+
const start = text.indexOf(startMarker);
|
|
162
|
+
const end = text.indexOf(endMarker);
|
|
163
|
+
if (start === -1 || end === -1 || end < start) return "";
|
|
164
|
+
return text.slice(start + startMarker.length, end);
|
|
165
|
+
}
|
|
166
|
+
function stripBlock(text, startMarker, endMarker) {
|
|
167
|
+
const start = text.indexOf(startMarker);
|
|
168
|
+
const end = text.indexOf(endMarker);
|
|
169
|
+
if (start === -1 || end === -1 || end < start) return text.trimEnd();
|
|
170
|
+
return (text.slice(0, start) + text.slice(end + endMarker.length)).trimEnd();
|
|
171
|
+
}
|
|
158
172
|
|
|
159
173
|
// src/campaign/drivers/ace.ts
|
|
160
174
|
var BLOCK_START = "<!-- BEGIN ace-playbook (auto-managed by aceDriver) -->";
|
|
161
175
|
var BLOCK_END = "<!-- END ace-playbook -->";
|
|
162
176
|
var DEFAULT_HEADING = "## Playbook (accumulated lessons \u2014 append-only)";
|
|
163
177
|
function parsePlaybook(surface) {
|
|
164
|
-
const
|
|
165
|
-
const end = surface.indexOf(BLOCK_END);
|
|
166
|
-
if (start === -1 || end === -1 || end < start) return [];
|
|
167
|
-
const body = surface.slice(start + BLOCK_START.length, end);
|
|
178
|
+
const body = extractBlockBody(surface, BLOCK_START, BLOCK_END);
|
|
168
179
|
const out = [];
|
|
169
180
|
for (const raw of body.split("\n")) {
|
|
170
181
|
const line = raw.trim();
|
|
@@ -176,12 +187,6 @@ function parsePlaybook(surface) {
|
|
|
176
187
|
}
|
|
177
188
|
return out;
|
|
178
189
|
}
|
|
179
|
-
function stripBlock(surface) {
|
|
180
|
-
const start = surface.indexOf(BLOCK_START);
|
|
181
|
-
const end = surface.indexOf(BLOCK_END);
|
|
182
|
-
if (start === -1 || end === -1 || end < start) return surface.trimEnd();
|
|
183
|
-
return (surface.slice(0, start) + surface.slice(end + BLOCK_END.length)).trimEnd();
|
|
184
|
-
}
|
|
185
190
|
function aceDriver(opts = {}) {
|
|
186
191
|
const maxEntries = opts.maxEntries ?? 50;
|
|
187
192
|
if (maxEntries < 1) throw new Error("aceDriver: maxEntries must be >= 1");
|
|
@@ -209,7 +214,7 @@ function aceDriver(opts = {}) {
|
|
|
209
214
|
...all.map((b) => `- [g${b.gen}] ${b.text}`),
|
|
210
215
|
BLOCK_END
|
|
211
216
|
].join("\n");
|
|
212
|
-
const base = stripBlock(parent);
|
|
217
|
+
const base = stripBlock(parent, BLOCK_START, BLOCK_END);
|
|
213
218
|
const surface = base ? `${base}
|
|
214
219
|
|
|
215
220
|
${block}` : block;
|
|
@@ -224,115 +229,75 @@ ${block}` : block;
|
|
|
224
229
|
};
|
|
225
230
|
}
|
|
226
231
|
|
|
227
|
-
// src/campaign/drivers/guide.ts
|
|
228
|
-
var DRIVER_GUIDE = {
|
|
229
|
-
gepa: {
|
|
230
|
-
summary: "Reflective full-surface rewrite: reflects on the best parent\u2019s weakest dimensions + per-scenario scores, proposes targeted rewrites, maintains a Pareto frontier across generations.",
|
|
231
|
-
surface: "prompt",
|
|
232
|
-
strategy: "reflective-rewrite",
|
|
233
|
-
whenUse: "The default for a prompt/instruction surface with headroom \u2014 broad rewrites plus Pareto-optimal exploration across scenarios.",
|
|
234
|
-
cost: "medium"
|
|
235
|
-
},
|
|
236
|
-
skillOpt: {
|
|
237
|
-
summary: "Patch-mode: bounded, anchored add/delete/replace edits to ONE skill document, so a good rule introduced earlier is not clobbered by a later sweeping rewrite.",
|
|
238
|
-
surface: "skill-doc",
|
|
239
|
-
strategy: "anchored-patch",
|
|
240
|
-
whenUse: 'Refining a skill document incrementally where accumulated rules must be preserved; the edit budget is the "textual learning rate".',
|
|
241
|
-
cost: "medium"
|
|
242
|
-
},
|
|
243
|
-
ace: {
|
|
244
|
-
summary: "Append-mostly playbook curator: grows the playbook with provenance-tagged delta bullets, never merging \u2014 guards against context collapse.",
|
|
245
|
-
surface: "playbook",
|
|
246
|
-
strategy: "append-only",
|
|
247
|
-
whenUse: "Accumulating many specific, hard-won lessons over time where dedup/rewrite would summarize away detail.",
|
|
248
|
-
cost: "low"
|
|
249
|
-
},
|
|
250
|
-
memoryCuration: {
|
|
251
|
-
summary: "Dedup-and-rank curator: builds a compact searchable memory and grafts the most relevant, most-recurrent lessons onto the surface.",
|
|
252
|
-
surface: "memory",
|
|
253
|
-
strategy: "dedup-curate",
|
|
254
|
-
whenUse: "Accumulating lessons while keeping the surface compact \u2014 the complement to ace when context size matters more than verbatim provenance.",
|
|
255
|
-
cost: "low"
|
|
256
|
-
},
|
|
257
|
-
halo: {
|
|
258
|
-
summary: "Wraps the real external HALO engine (Inference.net, `halo` CLI) and applies its findings to the prompt via one LLM edit.",
|
|
259
|
-
surface: "prompt",
|
|
260
|
-
strategy: "analysis-edit",
|
|
261
|
-
whenUse: "Benchmarking: compete HALO head-to-head against our own analysis on identical traces via compareDrivers.",
|
|
262
|
-
cost: "high",
|
|
263
|
-
external: true
|
|
264
|
-
},
|
|
265
|
-
traceAnalyst: {
|
|
266
|
-
summary: "Wraps agent-eval\u2019s own trace-analyst engine and applies its findings to the prompt via one identical LLM edit \u2014 the symmetric opponent to haloDriver.",
|
|
267
|
-
surface: "prompt",
|
|
268
|
-
strategy: "analysis-edit",
|
|
269
|
-
whenUse: "Benchmarking our trace-analyst\u2019s analysis quality against HALO (analysis-quality head-to-head), or improving from a real OTLP trace corpus.",
|
|
270
|
-
cost: "high"
|
|
271
|
-
},
|
|
272
|
-
evolutionary: {
|
|
273
|
-
summary: "Adapts a stateless Mutator (population mutate \u2192 measure \u2192 select); no generation memory beyond the current surface.",
|
|
274
|
-
surface: "any",
|
|
275
|
-
strategy: "population-mutate",
|
|
276
|
-
whenUse: "Blind population search when you have a Mutator and don\u2019t need reflective reasoning over findings.",
|
|
277
|
-
cost: "medium"
|
|
278
|
-
}
|
|
279
|
-
};
|
|
280
|
-
var GOAL_RANK = {
|
|
281
|
-
explore: ["gepa", "evolutionary"],
|
|
282
|
-
refine: ["skillOpt", "gepa"],
|
|
283
|
-
accumulate: ["ace", "memoryCuration"],
|
|
284
|
-
benchmark: ["traceAnalyst", "halo"]
|
|
285
|
-
};
|
|
286
|
-
function selectDriver(criteria) {
|
|
287
|
-
const ranked = GOAL_RANK[criteria.goal];
|
|
288
|
-
const out = [];
|
|
289
|
-
for (const name of ranked) {
|
|
290
|
-
const entry = DRIVER_GUIDE[name];
|
|
291
|
-
if (criteria.surface && criteria.surface !== "any" && entry.surface !== criteria.surface)
|
|
292
|
-
continue;
|
|
293
|
-
out.push({
|
|
294
|
-
name,
|
|
295
|
-
entry,
|
|
296
|
-
reason: `${criteria.goal}: ${entry.strategy} on the ${entry.surface} surface \u2014 ${entry.whenUse}`
|
|
297
|
-
});
|
|
298
|
-
}
|
|
299
|
-
if (out.length === 0 && criteria.surface) {
|
|
300
|
-
for (const name of Object.keys(DRIVER_GUIDE)) {
|
|
301
|
-
const entry = DRIVER_GUIDE[name];
|
|
302
|
-
if (entry.surface === criteria.surface || entry.surface === "any") {
|
|
303
|
-
out.push({ name, entry, reason: `surface match (${entry.surface}): ${entry.whenUse}` });
|
|
304
|
-
}
|
|
305
|
-
}
|
|
306
|
-
}
|
|
307
|
-
return out;
|
|
308
|
-
}
|
|
309
|
-
|
|
310
232
|
// src/campaign/drivers/halo.ts
|
|
311
233
|
import { execFile } from "child_process";
|
|
234
|
+
import { promisify } from "util";
|
|
235
|
+
|
|
236
|
+
// src/campaign/drivers/analysis-edit.ts
|
|
312
237
|
import { mkdtempSync, writeFileSync } from "fs";
|
|
313
238
|
import { tmpdir } from "os";
|
|
314
239
|
import { join } from "path";
|
|
315
|
-
import { promisify } from "util";
|
|
316
|
-
var execFileAsync = promisify(execFile);
|
|
317
|
-
var DEFAULT_ANALYSIS_PROMPT = "Diagnose the failures in these agent execution traces \u2014 hallucinated tool calls, redundant tool arguments, refusal loops, and semantic-correctness errors \u2014 and suggest concrete, generalizable fixes to the agent instructions.";
|
|
318
240
|
var APPLY_SYSTEM = "You apply a trace-analysis report to an agent instruction prompt. Output ONLY the full revised prompt \u2014 no preamble, no commentary, no code fences. Make the minimal edits that address the report findings; preserve everything else verbatim.";
|
|
319
|
-
function
|
|
320
|
-
|
|
321
|
-
|
|
241
|
+
function surfaceToPromptText(surface) {
|
|
242
|
+
return typeof surface === "string" ? surface : JSON.stringify(surface);
|
|
243
|
+
}
|
|
244
|
+
function analysisEditDriver(opts) {
|
|
322
245
|
return {
|
|
323
|
-
kind:
|
|
246
|
+
kind: opts.kind,
|
|
324
247
|
async propose(ctx) {
|
|
325
|
-
const parent =
|
|
248
|
+
const parent = surfaceToPromptText(ctx.currentSurface);
|
|
326
249
|
const traces = await opts.resolveTraces(ctx) ?? "";
|
|
327
|
-
if (!traces.trim())
|
|
328
|
-
|
|
329
|
-
"haloDriver: resolveTraces returned no OTLP traces \u2014 the halo engine has nothing to analyze"
|
|
330
|
-
);
|
|
331
|
-
}
|
|
332
|
-
const dir = mkdtempSync(join(tmpdir(), "halo-driver-"));
|
|
250
|
+
if (!traces.trim()) throw new Error(opts.noTracesError);
|
|
251
|
+
const dir = mkdtempSync(join(tmpdir(), `${opts.kind}-driver-`));
|
|
333
252
|
const tracePath = join(dir, "traces.jsonl");
|
|
334
253
|
writeFileSync(tracePath, traces.endsWith("\n") ? traces : `${traces}
|
|
335
254
|
`);
|
|
255
|
+
const report = await opts.analyze(tracePath, ctx);
|
|
256
|
+
const applied = await callLlm(
|
|
257
|
+
{
|
|
258
|
+
model: opts.applyModel,
|
|
259
|
+
messages: [
|
|
260
|
+
{ role: "system", content: APPLY_SYSTEM },
|
|
261
|
+
{
|
|
262
|
+
role: "user",
|
|
263
|
+
content: `CURRENT PROMPT:
|
|
264
|
+
${parent}
|
|
265
|
+
|
|
266
|
+
TRACE-ANALYSIS REPORT:
|
|
267
|
+
${report}
|
|
268
|
+
|
|
269
|
+
Return the full revised prompt.`
|
|
270
|
+
}
|
|
271
|
+
]
|
|
272
|
+
},
|
|
273
|
+
{ baseUrl: opts.baseUrl, apiKey: opts.apiKey, fetch: opts.fetchImpl }
|
|
274
|
+
);
|
|
275
|
+
const text = applied.content.trim();
|
|
276
|
+
if (!text || text === parent) return [];
|
|
277
|
+
return [{ surface: text, label: opts.label, rationale: opts.rationale(report) }];
|
|
278
|
+
}
|
|
279
|
+
};
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
// src/campaign/drivers/halo.ts
|
|
283
|
+
var execFileAsync = promisify(execFile);
|
|
284
|
+
var DEFAULT_ANALYSIS_PROMPT = "Diagnose the failures in these agent execution traces \u2014 hallucinated tool calls, redundant tool arguments, refusal loops, and semantic-correctness errors \u2014 and suggest concrete, generalizable fixes to the agent instructions.";
|
|
285
|
+
function haloDriver(opts) {
|
|
286
|
+
const haloBin = opts.haloBin ?? "halo";
|
|
287
|
+
const model = opts.model ?? "gpt-5.4-mini";
|
|
288
|
+
return analysisEditDriver({
|
|
289
|
+
kind: "halo",
|
|
290
|
+
label: "halo",
|
|
291
|
+
baseUrl: opts.baseUrl,
|
|
292
|
+
apiKey: opts.apiKey,
|
|
293
|
+
applyModel: opts.applyModel ?? model,
|
|
294
|
+
fetchImpl: opts.fetchImpl,
|
|
295
|
+
resolveTraces: opts.resolveTraces,
|
|
296
|
+
noTracesError: "haloDriver: resolveTraces returned no OTLP traces \u2014 the halo engine has nothing to analyze",
|
|
297
|
+
// HALO's real findings are preserved verbatim in the rationale (attribution).
|
|
298
|
+
rationale: (findings) => `halo-engine findings:
|
|
299
|
+
${findings.slice(0, 800)}`,
|
|
300
|
+
analyze: async (tracePath, ctx) => {
|
|
336
301
|
const args = [
|
|
337
302
|
tracePath,
|
|
338
303
|
"-p",
|
|
@@ -360,37 +325,9 @@ function haloDriver(opts) {
|
|
|
360
325
|
);
|
|
361
326
|
}
|
|
362
327
|
if (!findings) throw new Error("haloDriver: halo-engine produced no findings");
|
|
363
|
-
|
|
364
|
-
{
|
|
365
|
-
model: opts.applyModel ?? model,
|
|
366
|
-
messages: [
|
|
367
|
-
{ role: "system", content: APPLY_SYSTEM },
|
|
368
|
-
{
|
|
369
|
-
role: "user",
|
|
370
|
-
content: `CURRENT PROMPT:
|
|
371
|
-
${parent}
|
|
372
|
-
|
|
373
|
-
HALO TRACE-ANALYSIS REPORT:
|
|
374
|
-
${findings}
|
|
375
|
-
|
|
376
|
-
Return the full revised prompt.`
|
|
377
|
-
}
|
|
378
|
-
]
|
|
379
|
-
},
|
|
380
|
-
{ baseUrl: opts.baseUrl, apiKey: opts.apiKey, fetch: opts.fetchImpl }
|
|
381
|
-
);
|
|
382
|
-
const text = applied.content.trim();
|
|
383
|
-
if (!text || text === parent) return [];
|
|
384
|
-
return [
|
|
385
|
-
{
|
|
386
|
-
surface: text,
|
|
387
|
-
label: "halo",
|
|
388
|
-
rationale: `halo-engine findings:
|
|
389
|
-
${findings.slice(0, 800)}`
|
|
390
|
-
}
|
|
391
|
-
];
|
|
328
|
+
return findings;
|
|
392
329
|
}
|
|
393
|
-
};
|
|
330
|
+
});
|
|
394
331
|
}
|
|
395
332
|
|
|
396
333
|
// src/campaign/drivers/memory.ts
|
|
@@ -399,16 +336,7 @@ var BLOCK_END2 = "<!-- END curated-memory -->";
|
|
|
399
336
|
var DEFAULT_HEADING2 = "## Learned from prior runs (curated memory)";
|
|
400
337
|
var DISTILL_SYSTEM = 'You compress raw trace-analysis findings into crisp, generalizable agent guidance. Output ONLY a JSON array of strings, each one imperative lesson the agent should follow (e.g. "Always fetch a resource before mutating it"). No prose outside the JSON. Deduplicate; keep the most actionable and general; drop case-specific noise.';
|
|
401
338
|
function extractExistingLessons(text) {
|
|
402
|
-
|
|
403
|
-
const end = text.indexOf(BLOCK_END2);
|
|
404
|
-
if (start === -1 || end === -1 || end < start) return [];
|
|
405
|
-
return text.slice(start + BLOCK_START2.length, end).split("\n").map((l) => l.replace(/^\s*-\s+/, "").trim()).filter((l) => l && !l.startsWith("#"));
|
|
406
|
-
}
|
|
407
|
-
function stripBlock2(text) {
|
|
408
|
-
const start = text.indexOf(BLOCK_START2);
|
|
409
|
-
const end = text.indexOf(BLOCK_END2);
|
|
410
|
-
if (start === -1 || end === -1 || end < start) return text.trimEnd();
|
|
411
|
-
return (text.slice(0, start) + text.slice(end + BLOCK_END2.length)).trimEnd();
|
|
339
|
+
return extractBlockBody(text, BLOCK_START2, BLOCK_END2).split("\n").map((l) => l.replace(/^\s*-\s+/, "").trim()).filter((l) => l && !l.startsWith("#"));
|
|
412
340
|
}
|
|
413
341
|
async function distillLessons(raw, distill) {
|
|
414
342
|
const res = await callLlm(
|
|
@@ -466,7 +394,7 @@ function memoryCurationDriver(opts = {}) {
|
|
|
466
394
|
const block = [BLOCK_START2, heading, ...ranked.map((e) => `- ${e.text}`), BLOCK_END2].join(
|
|
467
395
|
"\n"
|
|
468
396
|
);
|
|
469
|
-
const next = `${
|
|
397
|
+
const next = `${stripBlock(parent, BLOCK_START2, BLOCK_END2)}
|
|
470
398
|
|
|
471
399
|
${block}
|
|
472
400
|
`;
|
|
@@ -721,11 +649,7 @@ function snippet(s, max = 120) {
|
|
|
721
649
|
}
|
|
722
650
|
|
|
723
651
|
// src/campaign/drivers/trace-analyst.ts
|
|
724
|
-
import { mkdtempSync as mkdtempSync2, writeFileSync as writeFileSync2 } from "fs";
|
|
725
|
-
import { tmpdir as tmpdir2 } from "os";
|
|
726
|
-
import { join as join2 } from "path";
|
|
727
652
|
import { ai } from "@ax-llm/ax";
|
|
728
|
-
var APPLY_SYSTEM2 = "You apply a trace-analysis report to an agent instruction prompt. Output ONLY the full revised prompt \u2014 no preamble, no commentary, no code fences. Make the minimal edits that address the report findings; preserve everything else verbatim.";
|
|
729
653
|
function renderFindings(findings) {
|
|
730
654
|
return findings.map((f, i) => {
|
|
731
655
|
const action = f.recommended_action ? `
|
|
@@ -738,41 +662,39 @@ function traceAnalystDriver(opts) {
|
|
|
738
662
|
if (!opts.apiKey) throw new Error("traceAnalystDriver: apiKey is required");
|
|
739
663
|
if (!opts.model) throw new Error("traceAnalystDriver: model is required");
|
|
740
664
|
const kinds = opts.kinds ?? DEFAULT_TRACE_ANALYST_KINDS;
|
|
741
|
-
|
|
665
|
+
const produceFindings = opts.analyze ?? (async (path, c) => {
|
|
666
|
+
const aiService = ai({
|
|
667
|
+
name: opts.provider ?? "openai",
|
|
668
|
+
apiKey: opts.apiKey,
|
|
669
|
+
apiURL: opts.baseUrl,
|
|
670
|
+
config: { model: opts.model }
|
|
671
|
+
});
|
|
672
|
+
const registry = new AnalystRegistry();
|
|
673
|
+
for (const spec of kinds) {
|
|
674
|
+
registry.register(createTraceAnalystKind(spec, { ai: aiService, model: opts.model }));
|
|
675
|
+
}
|
|
676
|
+
const result = await registry.run(
|
|
677
|
+
`trace-analyst-gen-${c.generation}`,
|
|
678
|
+
{ traceStore: new OtlpFileTraceStore({ path }) },
|
|
679
|
+
{ signal: c.signal }
|
|
680
|
+
);
|
|
681
|
+
return result.findings;
|
|
682
|
+
});
|
|
683
|
+
return analysisEditDriver({
|
|
742
684
|
kind: "trace-analyst",
|
|
743
|
-
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
writeFileSync2(tracePath, traces.endsWith("\n") ? traces : `${traces}
|
|
754
|
-
`);
|
|
755
|
-
const runAnalyze = opts.analyze ?? (async (path, c) => {
|
|
756
|
-
const aiService = ai({
|
|
757
|
-
name: opts.provider ?? "openai",
|
|
758
|
-
apiKey: opts.apiKey,
|
|
759
|
-
apiURL: opts.baseUrl,
|
|
760
|
-
config: { model: opts.model }
|
|
761
|
-
});
|
|
762
|
-
const registry = new AnalystRegistry();
|
|
763
|
-
for (const spec of kinds) {
|
|
764
|
-
registry.register(createTraceAnalystKind(spec, { ai: aiService, model: opts.model }));
|
|
765
|
-
}
|
|
766
|
-
const result = await registry.run(
|
|
767
|
-
`trace-analyst-gen-${c.generation}`,
|
|
768
|
-
{ traceStore: new OtlpFileTraceStore({ path }) },
|
|
769
|
-
{ signal: c.signal }
|
|
770
|
-
);
|
|
771
|
-
return result.findings;
|
|
772
|
-
});
|
|
685
|
+
label: "trace-analyst",
|
|
686
|
+
baseUrl: opts.baseUrl,
|
|
687
|
+
apiKey: opts.apiKey,
|
|
688
|
+
applyModel: opts.applyModel ?? opts.model,
|
|
689
|
+
fetchImpl: opts.fetchImpl,
|
|
690
|
+
resolveTraces: opts.resolveTraces,
|
|
691
|
+
noTracesError: "traceAnalystDriver: resolveTraces returned no OTLP traces \u2014 the analyst has nothing to read",
|
|
692
|
+
rationale: (report) => `trace-analyst findings:
|
|
693
|
+
${report.slice(0, 800)}`,
|
|
694
|
+
analyze: async (tracePath, ctx) => {
|
|
773
695
|
let findings;
|
|
774
696
|
try {
|
|
775
|
-
findings = await
|
|
697
|
+
findings = await produceFindings(tracePath, ctx);
|
|
776
698
|
} catch (e) {
|
|
777
699
|
throw new Error(
|
|
778
700
|
`traceAnalystDriver: analyst engine failed \u2014 ${e instanceof Error ? e.message : String(e)}`
|
|
@@ -781,44 +703,15 @@ function traceAnalystDriver(opts) {
|
|
|
781
703
|
if (findings.length === 0) {
|
|
782
704
|
throw new Error("traceAnalystDriver: analyst engine produced no findings");
|
|
783
705
|
}
|
|
784
|
-
|
|
785
|
-
const applied = await callLlm(
|
|
786
|
-
{
|
|
787
|
-
model: opts.applyModel ?? opts.model,
|
|
788
|
-
messages: [
|
|
789
|
-
{ role: "system", content: APPLY_SYSTEM2 },
|
|
790
|
-
{
|
|
791
|
-
role: "user",
|
|
792
|
-
content: `CURRENT PROMPT:
|
|
793
|
-
${parent}
|
|
794
|
-
|
|
795
|
-
TRACE-ANALYSIS REPORT:
|
|
796
|
-
${report}
|
|
797
|
-
|
|
798
|
-
Return the full revised prompt.`
|
|
799
|
-
}
|
|
800
|
-
]
|
|
801
|
-
},
|
|
802
|
-
{ baseUrl: opts.baseUrl, apiKey: opts.apiKey, fetch: opts.fetchImpl }
|
|
803
|
-
);
|
|
804
|
-
const text = applied.content.trim();
|
|
805
|
-
if (!text || text === parent) return [];
|
|
806
|
-
return [
|
|
807
|
-
{
|
|
808
|
-
surface: text,
|
|
809
|
-
label: "trace-analyst",
|
|
810
|
-
rationale: `trace-analyst findings (${findings.length}):
|
|
811
|
-
${report.slice(0, 800)}`
|
|
812
|
-
}
|
|
813
|
-
];
|
|
706
|
+
return renderFindings(findings);
|
|
814
707
|
}
|
|
815
|
-
};
|
|
708
|
+
});
|
|
816
709
|
}
|
|
817
710
|
|
|
818
711
|
// src/campaign/labeled-store/fs-adapter.ts
|
|
819
712
|
import { createHash } from "crypto";
|
|
820
|
-
import { existsSync, mkdirSync, readFileSync, writeFileSync as
|
|
821
|
-
import { join as
|
|
713
|
+
import { existsSync, mkdirSync, readFileSync, writeFileSync as writeFileSync2 } from "fs";
|
|
714
|
+
import { join as join2 } from "path";
|
|
822
715
|
var LabeledScenarioStoreError = class extends Error {
|
|
823
716
|
constructor(code, message) {
|
|
824
717
|
super(message);
|
|
@@ -978,7 +871,7 @@ var FsLabeledScenarioStore = class {
|
|
|
978
871
|
};
|
|
979
872
|
}
|
|
980
873
|
pathForSource(source) {
|
|
981
|
-
return
|
|
874
|
+
return join2(this.options.root, `${source}.jsonl`);
|
|
982
875
|
}
|
|
983
876
|
};
|
|
984
877
|
var ALL_SOURCES = [
|
|
@@ -1020,9 +913,9 @@ function sha256(input) {
|
|
|
1020
913
|
function appendLine(path, line) {
|
|
1021
914
|
if (existsSync(path)) {
|
|
1022
915
|
const existing = readFileSync(path, "utf8");
|
|
1023
|
-
|
|
916
|
+
writeFileSync2(path, existing + line);
|
|
1024
917
|
} else {
|
|
1025
|
-
|
|
918
|
+
writeFileSync2(path, line);
|
|
1026
919
|
}
|
|
1027
920
|
}
|
|
1028
921
|
|
|
@@ -1472,7 +1365,7 @@ function renderScoreboardMarkdown(rows, opts = {}) {
|
|
|
1472
1365
|
|
|
1473
1366
|
// src/campaign/presets/run-profile-matrix.ts
|
|
1474
1367
|
import { createHash as createHash2 } from "crypto";
|
|
1475
|
-
import { join as
|
|
1368
|
+
import { join as join3 } from "path";
|
|
1476
1369
|
var ProfileMatrixError = class extends AgentEvalError {
|
|
1477
1370
|
constructor(message) {
|
|
1478
1371
|
super("profile_matrix", message);
|
|
@@ -1623,7 +1516,7 @@ async function runProfileMatrix(opts) {
|
|
|
1623
1516
|
captureSource: opts.captureSource,
|
|
1624
1517
|
storage: opts.storage,
|
|
1625
1518
|
now: opts.now,
|
|
1626
|
-
runDir:
|
|
1519
|
+
runDir: join3(opts.runDir, sanitize(profile.id))
|
|
1627
1520
|
});
|
|
1628
1521
|
const profileRecords = [];
|
|
1629
1522
|
for (const cell of campaign.cells) {
|
|
@@ -1695,7 +1588,7 @@ function rollupByPersona(records, scenarios, personaOf) {
|
|
|
1695
1588
|
// src/campaign/worktree/index.ts
|
|
1696
1589
|
import { execFileSync } from "child_process";
|
|
1697
1590
|
import { existsSync as existsSync2 } from "fs";
|
|
1698
|
-
import { basename, isAbsolute, join as
|
|
1591
|
+
import { basename, isAbsolute, join as join4 } from "path";
|
|
1699
1592
|
var WorktreeAdapterError = class extends Error {
|
|
1700
1593
|
constructor(message, cause) {
|
|
1701
1594
|
super(message);
|
|
@@ -1717,13 +1610,13 @@ function slug2(label) {
|
|
|
1717
1610
|
}
|
|
1718
1611
|
function gitWorktreeAdapter(opts) {
|
|
1719
1612
|
const git = opts.git ?? defaultGit;
|
|
1720
|
-
const worktreeDir = opts.worktreeDir ??
|
|
1613
|
+
const worktreeDir = opts.worktreeDir ?? join4(opts.repoRoot, ".worktrees");
|
|
1721
1614
|
const branchPrefix = opts.branchPrefix ?? "improve";
|
|
1722
1615
|
return {
|
|
1723
1616
|
async create({ baseRef, label }) {
|
|
1724
1617
|
const id = `${slug2(label)}-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`;
|
|
1725
1618
|
const branch = `${branchPrefix}/${id}`;
|
|
1726
|
-
const path =
|
|
1619
|
+
const path = join4(worktreeDir, id);
|
|
1727
1620
|
git(["worktree", "add", "-b", branch, path, baseRef], opts.repoRoot);
|
|
1728
1621
|
return { path, branch, baseRef };
|
|
1729
1622
|
},
|
|
@@ -1748,11 +1641,10 @@ function gitWorktreeAdapter(opts) {
|
|
|
1748
1641
|
}
|
|
1749
1642
|
function resolveWorktreePath(surface, worktreeDir) {
|
|
1750
1643
|
if (isAbsolute(surface.worktreeRef) && existsSync2(surface.worktreeRef)) return surface.worktreeRef;
|
|
1751
|
-
if (worktreeDir) return
|
|
1644
|
+
if (worktreeDir) return join4(worktreeDir, basename(surface.worktreeRef));
|
|
1752
1645
|
return surface.worktreeRef;
|
|
1753
1646
|
}
|
|
1754
1647
|
export {
|
|
1755
|
-
DRIVER_GUIDE,
|
|
1756
1648
|
FsLabeledScenarioStore,
|
|
1757
1649
|
LabeledScenarioStoreError,
|
|
1758
1650
|
ProfileMatrixError,
|
|
@@ -1808,7 +1700,6 @@ export {
|
|
|
1808
1700
|
runSkillOpt,
|
|
1809
1701
|
scoreUserStory,
|
|
1810
1702
|
scoreboardSummary,
|
|
1811
|
-
selectDriver,
|
|
1812
1703
|
skillOptDriver,
|
|
1813
1704
|
skillOptEntry,
|
|
1814
1705
|
surfaceContentHash,
|