@tangle-network/agent-runtime 0.91.0 → 0.92.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -2
- package/dist/agent.d.ts +3 -3
- package/dist/agent.js +88 -9
- package/dist/agent.js.map +1 -1
- package/dist/{mcp-serve-verifier-XsX8rkB9.d.ts → agentic-generator-B8oeE2Yv.d.ts} +6 -33
- package/dist/analyst-loop.d.ts +1 -1
- package/dist/candidate-execution/index.d.ts +104 -0
- package/dist/candidate-execution/index.js +34 -0
- package/dist/candidate-execution/index.js.map +1 -0
- package/dist/chunk-3BE7KTMU.js +1229 -0
- package/dist/chunk-3BE7KTMU.js.map +1 -0
- package/dist/chunk-3D2RHC4K.js +73 -0
- package/dist/chunk-3D2RHC4K.js.map +1 -0
- package/dist/chunk-3MDZX7YU.js +125 -0
- package/dist/chunk-3MDZX7YU.js.map +1 -0
- package/dist/{chunk-NC66AM3S.js → chunk-6O5USWVH.js} +32 -1413
- package/dist/chunk-6O5USWVH.js.map +1 -0
- package/dist/chunk-6O73TRHW.js +142 -0
- package/dist/chunk-6O73TRHW.js.map +1 -0
- package/dist/{chunk-JRS3YSRZ.js → chunk-7VJJJ2T2.js} +2 -2
- package/dist/chunk-A62TP7SK.js +4784 -0
- package/dist/chunk-A62TP7SK.js.map +1 -0
- package/dist/chunk-APVPRF4Y.js +2166 -0
- package/dist/chunk-APVPRF4Y.js.map +1 -0
- package/dist/{chunk-DWWII6N2.js → chunk-AUEIDTR3.js} +2 -2
- package/dist/{chunk-FF77IBQM.js → chunk-FRBHUNQ7.js} +2 -141
- package/dist/chunk-FRBHUNQ7.js.map +1 -0
- package/dist/{chunk-7ON74BQO.js → chunk-GDAQUFG6.js} +2 -2
- package/dist/{chunk-AD7JW4QG.js → chunk-I7WVPJBZ.js} +23 -1231
- package/dist/chunk-I7WVPJBZ.js.map +1 -0
- package/dist/{chunk-IOUUITQA.js → chunk-IGGZGKJD.js} +3 -3
- package/dist/chunk-PH65PR4F.js +860 -0
- package/dist/chunk-PH65PR4F.js.map +1 -0
- package/dist/chunk-RSWM2ZKM.js +659 -0
- package/dist/chunk-RSWM2ZKM.js.map +1 -0
- package/dist/{chunk-BZF3KQ6G.js → chunk-VSWBYWFK.js} +4 -122
- package/dist/chunk-VSWBYWFK.js.map +1 -0
- package/dist/{completion-gate-DkAnUmpb.d.ts → completion-gate-BLaiN0-X.d.ts} +1 -1
- package/dist/{coordination-rRj5hjJK.d.ts → coordination-DxJ83oZA.d.ts} +12 -5
- package/dist/environment-provider.d.ts +2 -2
- package/dist/environment-provider.js +2 -1
- package/dist/improve-CUVCq7xg.d.ts +152 -0
- package/dist/{improvement-adapter-CDR8QNVM.d.ts → improvement-adapter-BieWeK5J.d.ts} +16 -0
- package/dist/index.d.ts +27 -1092
- package/dist/index.js +108 -6491
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +160 -13
- package/dist/intelligence.js +535 -59
- package/dist/intelligence.js.map +1 -1
- package/dist/knowledge.d.ts +6 -6
- package/dist/knowledge.js +6 -4
- package/dist/lifecycle.d.ts +2 -1
- package/dist/lifecycle.js +5 -3
- package/dist/lifecycle.js.map +1 -1
- package/dist/{loop-runner-bin-DTbZVGfM.d.ts → loop-runner-bin-kKUNGLyV.d.ts} +2 -2
- package/dist/loop-runner-bin.d.ts +5 -5
- package/dist/loop-runner-bin.js +8 -5
- package/dist/loops.d.ts +16 -16
- package/dist/loops.js +47 -41
- package/dist/mcp/bin.js +6 -4
- package/dist/mcp/bin.js.map +1 -1
- package/dist/mcp/index.d.ts +8 -8
- package/dist/mcp/index.js +9 -6
- package/dist/mcp/index.js.map +1 -1
- package/dist/mcp-serve-verifier-Bg4C3p5S.d.ts +34 -0
- package/dist/{openai-tools-C4ZfUD4L.d.ts → openai-tools-E3woykz9.d.ts} +1 -1
- package/dist/prepare-Z08a4heC.d.ts +713 -0
- package/dist/profiles.d.ts +1 -1
- package/dist/{router-client-DJImUDlm.d.ts → sanitize-C9go6tXj.d.ts} +113 -1
- package/dist/{structural-rollout-MwlpgQ-6.d.ts → structural-rollout-DHGDbhvR.d.ts} +3 -3
- package/dist/{supervise-DPmYPk0j.d.ts → supervise-T2pazU3G.d.ts} +4 -4
- package/dist/{types-SyuwunY_.d.ts → types-B00NtbCs.d.ts} +1 -1
- package/dist/{types-eMNgWgFi.d.ts → types-DAdIm4AC.d.ts} +1 -1
- package/dist/{worktree-fanout-BDFQIO-Y.d.ts → worktree-fanout-BUb2Ag02.d.ts} +3 -3
- package/package.json +26 -35
- package/skills/build-with-agent-runtime/SKILL.md +1 -1
- package/dist/chunk-AD7JW4QG.js.map +0 -1
- package/dist/chunk-BZF3KQ6G.js.map +0 -1
- package/dist/chunk-FF77IBQM.js.map +0 -1
- package/dist/chunk-IVGYLCFH.js +0 -381
- package/dist/chunk-IVGYLCFH.js.map +0 -1
- package/dist/chunk-NC66AM3S.js.map +0 -1
- /package/dist/{chunk-JRS3YSRZ.js.map → chunk-7VJJJ2T2.js.map} +0 -0
- /package/dist/{chunk-DWWII6N2.js.map → chunk-AUEIDTR3.js.map} +0 -0
- /package/dist/{chunk-7ON74BQO.js.map → chunk-GDAQUFG6.js.map} +0 -0
- /package/dist/{chunk-IOUUITQA.js.map → chunk-IGGZGKJD.js.map} +0 -0
|
@@ -1,21 +1,31 @@
|
|
|
1
1
|
import {
|
|
2
|
-
|
|
3
|
-
|
|
2
|
+
observe,
|
|
3
|
+
refine,
|
|
4
|
+
runAgentic,
|
|
5
|
+
sample,
|
|
6
|
+
sampleThenRefine
|
|
7
|
+
} from "./chunk-3BE7KTMU.js";
|
|
8
|
+
import {
|
|
4
9
|
createExecutor,
|
|
5
10
|
createExecutorRegistry,
|
|
6
11
|
createSandboxLineage,
|
|
7
|
-
createSupervisor,
|
|
8
12
|
createWorktreeCliExecutor,
|
|
9
13
|
defaultSelectWinner,
|
|
10
14
|
gateOnDeliverable,
|
|
11
|
-
notifyRuntimeHookEvent,
|
|
12
15
|
probeSandboxCapabilities,
|
|
13
|
-
routerToolLoop,
|
|
14
16
|
runLoop,
|
|
17
|
+
supervise
|
|
18
|
+
} from "./chunk-6O5USWVH.js";
|
|
19
|
+
import {
|
|
20
|
+
createSupervisor,
|
|
21
|
+
notifyRuntimeHookEvent,
|
|
15
22
|
settledToIteration,
|
|
16
|
-
supervise,
|
|
17
23
|
withDriverExecutor
|
|
18
|
-
} from "./chunk-
|
|
24
|
+
} from "./chunk-APVPRF4Y.js";
|
|
25
|
+
import {
|
|
26
|
+
InMemoryResultBlobStore,
|
|
27
|
+
InMemorySpawnJournal
|
|
28
|
+
} from "./chunk-3D2RHC4K.js";
|
|
19
29
|
import {
|
|
20
30
|
addTokenUsage,
|
|
21
31
|
isAbortError,
|
|
@@ -23,7 +33,7 @@ import {
|
|
|
23
33
|
sleep,
|
|
24
34
|
stringifySafe,
|
|
25
35
|
zeroTokenUsage
|
|
26
|
-
} from "./chunk-
|
|
36
|
+
} from "./chunk-3MDZX7YU.js";
|
|
27
37
|
import {
|
|
28
38
|
createSandboxToolPartState,
|
|
29
39
|
mapSandboxEvent,
|
|
@@ -37,7 +47,7 @@ import {
|
|
|
37
47
|
} from "./chunk-YEJR7IXO.js";
|
|
38
48
|
|
|
39
49
|
// src/runtime/index.ts
|
|
40
|
-
import { computeFindingId, makeFinding
|
|
50
|
+
import { computeFindingId, makeFinding } from "@tangle-network/agent-eval";
|
|
41
51
|
|
|
42
52
|
// src/runtime/anytime.ts
|
|
43
53
|
var median = (xs) => {
|
|
@@ -1110,163 +1120,6 @@ function defineLeaderboard(spec) {
|
|
|
1110
1120
|
return { run, toBenchmarkAdapter };
|
|
1111
1121
|
}
|
|
1112
1122
|
|
|
1113
|
-
// src/runtime/observe.ts
|
|
1114
|
-
import { makeFinding } from "@tangle-network/agent-eval";
|
|
1115
|
-
var observerId = "observe/trace";
|
|
1116
|
-
var defaultAnalystInstruction = "You are a third-person OBSERVER watching an AI agent work. You see its TRACE (what it did), not its grader. From the trace, name SPECIFIC, behavior-grounded findings: wasted/duplicated tool calls, thrash/retries, token/cost waste, missing verification, failure patterns. For each, a concrete recommended_action, and whether the AGENT (fix its skills/prompt/tools) or the OPERATOR (fix framing/decomposition/config) should act. Only claim what the trace shows. No findings if the run was clean.";
|
|
1117
|
-
function summarizeTrace(trace, maxLines) {
|
|
1118
|
-
const lines = [];
|
|
1119
|
-
for (const ev of trace) {
|
|
1120
|
-
const e = ev;
|
|
1121
|
-
const t = (e.type ?? "").toLowerCase();
|
|
1122
|
-
const d = e.data ?? {};
|
|
1123
|
-
const part = d.part ?? {};
|
|
1124
|
-
if (part.type === "tool")
|
|
1125
|
-
lines.push(`tool:${part.tool}${part.state?.status ? `(${part.state.status})` : ""}`);
|
|
1126
|
-
else if (t.includes("error"))
|
|
1127
|
-
lines.push(`ERROR: ${String(d.message ?? d.detail ?? "").slice(0, 200)}`);
|
|
1128
|
-
else if (t === "status" && typeof d.status === "string") lines.push(`status:${d.status}`);
|
|
1129
|
-
else if (t.includes("tool")) lines.push(`tool-event:${t}`);
|
|
1130
|
-
}
|
|
1131
|
-
const out = [];
|
|
1132
|
-
for (const ln of lines) {
|
|
1133
|
-
const prev = out[out.length - 1];
|
|
1134
|
-
const m = prev?.match(/^(.*?)(?: x(\d+))?$/);
|
|
1135
|
-
if (m && m[1] === ln) out[out.length - 1] = `${ln} x${(Number(m[2]) || 1) + 1}`;
|
|
1136
|
-
else out.push(ln);
|
|
1137
|
-
}
|
|
1138
|
-
return out.slice(0, maxLines).join("\n") || "(no tool/error events in trace)";
|
|
1139
|
-
}
|
|
1140
|
-
var findingsSchema = {
|
|
1141
|
-
name: "observer_findings",
|
|
1142
|
-
schema: {
|
|
1143
|
-
type: "object",
|
|
1144
|
-
additionalProperties: false,
|
|
1145
|
-
properties: {
|
|
1146
|
-
findings: {
|
|
1147
|
-
type: "array",
|
|
1148
|
-
items: {
|
|
1149
|
-
type: "object",
|
|
1150
|
-
additionalProperties: false,
|
|
1151
|
-
properties: {
|
|
1152
|
-
area: {
|
|
1153
|
-
type: "string",
|
|
1154
|
-
description: "tool-use | cost | verification | process | failure | latency"
|
|
1155
|
-
},
|
|
1156
|
-
severity: { type: "string", enum: ["critical", "high", "medium", "low", "info"] },
|
|
1157
|
-
claim: {
|
|
1158
|
-
type: "string",
|
|
1159
|
-
description: "what you OBSERVED in the trace (a fact, with the evidence)"
|
|
1160
|
-
},
|
|
1161
|
-
recommended_action: {
|
|
1162
|
-
type: "string",
|
|
1163
|
-
description: "the concrete change for the agent or operator"
|
|
1164
|
-
},
|
|
1165
|
-
audience: {
|
|
1166
|
-
type: "string",
|
|
1167
|
-
enum: ["agent", "operator"],
|
|
1168
|
-
description: "who should act on this"
|
|
1169
|
-
},
|
|
1170
|
-
confidence: { type: "number" }
|
|
1171
|
-
},
|
|
1172
|
-
required: ["area", "severity", "claim", "recommended_action", "audience", "confidence"]
|
|
1173
|
-
}
|
|
1174
|
-
}
|
|
1175
|
-
},
|
|
1176
|
-
required: ["findings"]
|
|
1177
|
-
}
|
|
1178
|
-
};
|
|
1179
|
-
async function observe(input, opts) {
|
|
1180
|
-
const traceSummary = summarizeTrace(input.trace, opts.maxTraceLines ?? 80);
|
|
1181
|
-
const res = await opts.chat.chat(
|
|
1182
|
-
{
|
|
1183
|
-
...opts.model ? { model: opts.model } : {},
|
|
1184
|
-
jsonSchema: findingsSchema,
|
|
1185
|
-
messages: [
|
|
1186
|
-
{
|
|
1187
|
-
role: "system",
|
|
1188
|
-
content: opts.analystInstruction ?? defaultAnalystInstruction
|
|
1189
|
-
},
|
|
1190
|
-
{
|
|
1191
|
-
role: "user",
|
|
1192
|
-
content: `TASK: ${input.task}
|
|
1193
|
-
|
|
1194
|
-
OUTCOME: ${input.outcome ?? "unknown"}
|
|
1195
|
-
|
|
1196
|
-
FINAL OUTPUT (truncated):
|
|
1197
|
-
${input.output.slice(0, 1200)}
|
|
1198
|
-
|
|
1199
|
-
TRACE (in order; "xN" = repeated):
|
|
1200
|
-
${traceSummary}`
|
|
1201
|
-
}
|
|
1202
|
-
]
|
|
1203
|
-
},
|
|
1204
|
-
{ ...opts.signal ? { signal: opts.signal } : {} }
|
|
1205
|
-
);
|
|
1206
|
-
const parsed = parseFindings(res.content);
|
|
1207
|
-
const producedAt = input.runId ? `${input.runId}` : observerId;
|
|
1208
|
-
const findings = parsed.map(
|
|
1209
|
-
(f) => makeFinding({
|
|
1210
|
-
analyst_id: observerId,
|
|
1211
|
-
area: `${f.area}`,
|
|
1212
|
-
severity: f.severity,
|
|
1213
|
-
claim: f.claim,
|
|
1214
|
-
recommended_action: f.recommended_action,
|
|
1215
|
-
confidence: typeof f.confidence === "number" ? f.confidence : 0.5,
|
|
1216
|
-
evidence_refs: [],
|
|
1217
|
-
// The observer reads BEHAVIOR, never the judge verdict — firewall provenance.
|
|
1218
|
-
derived_from_judge: false,
|
|
1219
|
-
metadata: { audience: f.audience },
|
|
1220
|
-
...input.runId ? { subject: input.runId } : {}
|
|
1221
|
-
})
|
|
1222
|
-
);
|
|
1223
|
-
const learned = [];
|
|
1224
|
-
if (opts.corpus) {
|
|
1225
|
-
for (const f of findings) {
|
|
1226
|
-
const record = {
|
|
1227
|
-
schemaVersion: "1.0.0",
|
|
1228
|
-
id: f.finding_id,
|
|
1229
|
-
runId: input.runId ?? observerId,
|
|
1230
|
-
producedAt: f.produced_at ?? producedAt,
|
|
1231
|
-
area: f.area,
|
|
1232
|
-
claim: f.recommended_action ?? f.claim,
|
|
1233
|
-
...f.claim ? { rationale: f.claim } : {},
|
|
1234
|
-
tags: [...opts.tags ?? [], `audience:${f.metadata?.audience ?? "agent"}`],
|
|
1235
|
-
confidence: f.confidence,
|
|
1236
|
-
evidence: [{ kind: "finding", uri: f.finding_id }]
|
|
1237
|
-
};
|
|
1238
|
-
const r = await opts.corpus.append(record);
|
|
1239
|
-
if (r.succeeded) learned.push(record);
|
|
1240
|
-
}
|
|
1241
|
-
}
|
|
1242
|
-
return { findings, learned, report: renderReport(findings) };
|
|
1243
|
-
}
|
|
1244
|
-
function parseFindings(content) {
|
|
1245
|
-
let obj2;
|
|
1246
|
-
try {
|
|
1247
|
-
obj2 = JSON.parse(content);
|
|
1248
|
-
} catch {
|
|
1249
|
-
const m = content.match(/\{[\s\S]*\}/);
|
|
1250
|
-
obj2 = m ? JSON.parse(m[0]) : { findings: [] };
|
|
1251
|
-
}
|
|
1252
|
-
const arr = obj2.findings;
|
|
1253
|
-
return Array.isArray(arr) ? arr : [];
|
|
1254
|
-
}
|
|
1255
|
-
function renderReport(findings) {
|
|
1256
|
-
if (findings.length === 0) return "\u2713 clean run \u2014 the observer found nothing to change.";
|
|
1257
|
-
const audience = (f) => f.metadata?.audience ?? "agent";
|
|
1258
|
-
const forAgent = findings.filter((f) => audience(f) === "agent");
|
|
1259
|
-
const forOperator = findings.filter((f) => audience(f) === "operator");
|
|
1260
|
-
const block = (title, fs) => fs.length === 0 ? "" : `**${title}**
|
|
1261
|
-
${fs.map((f) => `- [${f.severity}] ${f.claim}
|
|
1262
|
-
\u2192 ${f.recommended_action ?? ""}`).join("\n")}
|
|
1263
|
-
`;
|
|
1264
|
-
return [
|
|
1265
|
-
block("For the agent (fix skills / prompt / tools)", forAgent),
|
|
1266
|
-
block("For you (the operator)", forOperator)
|
|
1267
|
-
].filter(Boolean).join("\n");
|
|
1268
|
-
}
|
|
1269
|
-
|
|
1270
1123
|
// src/runtime/harvest-corpus.ts
|
|
1271
1124
|
async function harvestCorpus(opts) {
|
|
1272
1125
|
const concurrency = Math.max(1, opts.concurrency ?? 4);
|
|
@@ -2253,13 +2106,13 @@ function shapeName(shape, _resolved) {
|
|
|
2253
2106
|
}
|
|
2254
2107
|
function resolveShapeBudget(root, over) {
|
|
2255
2108
|
const fanout2 = over?.fanout ?? defaultFanout;
|
|
2256
|
-
const
|
|
2109
|
+
const perChild = over?.perChild ?? {
|
|
2257
2110
|
maxIterations: Math.max(1, Math.floor(root.maxIterations / fanout2)),
|
|
2258
2111
|
maxTokens: Math.max(1, Math.floor(root.maxTokens / fanout2)),
|
|
2259
2112
|
...root.maxUsd !== void 0 ? { maxUsd: root.maxUsd / fanout2 } : {},
|
|
2260
2113
|
...root.deadlineMs !== void 0 ? { deadlineMs: root.deadlineMs } : {}
|
|
2261
2114
|
};
|
|
2262
|
-
return { perChild
|
|
2115
|
+
return { perChild, fanout: fanout2 };
|
|
2263
2116
|
}
|
|
2264
2117
|
var defaultFanout = 3;
|
|
2265
2118
|
function personaRegistry(persona) {
|
|
@@ -2645,668 +2498,6 @@ function promotionGate(opts) {
|
|
|
2645
2498
|
|
|
2646
2499
|
// src/runtime/run-benchmark.ts
|
|
2647
2500
|
import { pairedBootstrap as pairedBootstrap2, paretoFrontier } from "@tangle-network/agent-eval";
|
|
2648
|
-
|
|
2649
|
-
// src/runtime/strategy.ts
|
|
2650
|
-
import { createChatClient, estimateCost, isModelPriced } from "@tangle-network/agent-eval";
|
|
2651
|
-
var taskNudge = "Use the available tools to bring the artifact to the required final state. Address EVERY distinct change the request implies. After each tool result, check what remains and continue. Re-read the values you set to confirm they took. Reply DONE only once every required change is made and verified.";
|
|
2652
|
-
async function runShot(surface, _task, handle, tools, messages, opts, modelOverride) {
|
|
2653
|
-
let toolErrors = 0;
|
|
2654
|
-
const execute = async (name, args) => {
|
|
2655
|
-
try {
|
|
2656
|
-
const out = await surface.call(handle, name, args);
|
|
2657
|
-
if (out.startsWith("ERROR:")) toolErrors += 1;
|
|
2658
|
-
return out;
|
|
2659
|
-
} catch (e) {
|
|
2660
|
-
toolErrors += 1;
|
|
2661
|
-
return `ERROR: ${e instanceof Error ? e.message : String(e)}`;
|
|
2662
|
-
}
|
|
2663
|
-
};
|
|
2664
|
-
const r = await routerToolLoop(
|
|
2665
|
-
{
|
|
2666
|
-
routerBaseUrl: opts.routerBaseUrl,
|
|
2667
|
-
routerKey: opts.routerKey,
|
|
2668
|
-
model: modelOverride ?? opts.model,
|
|
2669
|
-
...opts.complete ? { complete: opts.complete } : {}
|
|
2670
|
-
},
|
|
2671
|
-
"",
|
|
2672
|
-
"",
|
|
2673
|
-
tools,
|
|
2674
|
-
execute,
|
|
2675
|
-
{
|
|
2676
|
-
maxTurns: opts.innerTurns ?? 4,
|
|
2677
|
-
temperature: opts.temperature ?? 0.7,
|
|
2678
|
-
initialMessages: messages,
|
|
2679
|
-
...opts.maxTokens ? { maxTokens: opts.maxTokens } : {}
|
|
2680
|
-
}
|
|
2681
|
-
);
|
|
2682
|
-
return {
|
|
2683
|
-
messages: r.messages,
|
|
2684
|
-
completions: r.turns,
|
|
2685
|
-
toolCalls: r.toolCalls,
|
|
2686
|
-
toolErrors,
|
|
2687
|
-
tokens: r.usage
|
|
2688
|
-
};
|
|
2689
|
-
}
|
|
2690
|
-
function compactTrajectory(messages) {
|
|
2691
|
-
return messages.filter((m) => m.role === "assistant" || m.role === "tool").map((m) => {
|
|
2692
|
-
if (m.role === "tool") return `RESULT ${String(m.content).slice(0, 280)}`;
|
|
2693
|
-
const calls = m.tool_calls?.map((c) => `${c.function.name}(${c.function.arguments})`).join(", ");
|
|
2694
|
-
return calls ? `CALL ${calls}` : `SAY ${String(m.content).slice(0, 200)}`;
|
|
2695
|
-
}).join("\n").slice(0, 7e3);
|
|
2696
|
-
}
|
|
2697
|
-
function analystChat(opts, defaultModel) {
|
|
2698
|
-
if (!opts.complete) {
|
|
2699
|
-
return createChatClient({
|
|
2700
|
-
transport: "router",
|
|
2701
|
-
apiKey: opts.routerKey,
|
|
2702
|
-
baseUrl: opts.routerBaseUrl,
|
|
2703
|
-
defaultModel
|
|
2704
|
-
});
|
|
2705
|
-
}
|
|
2706
|
-
const complete = opts.complete;
|
|
2707
|
-
return createChatClient({
|
|
2708
|
-
transport: "mock",
|
|
2709
|
-
defaultModel,
|
|
2710
|
-
handler: async (req) => {
|
|
2711
|
-
const raw = await complete({
|
|
2712
|
-
model: req.model ?? defaultModel,
|
|
2713
|
-
messages: req.messages,
|
|
2714
|
-
...req.temperature !== void 0 ? { temperature: req.temperature } : {},
|
|
2715
|
-
...req.maxTokens !== void 0 ? { max_tokens: req.maxTokens } : {}
|
|
2716
|
-
});
|
|
2717
|
-
const content = raw.choices?.[0]?.message?.content ?? "";
|
|
2718
|
-
const promptTokens = raw.usage?.prompt_tokens ?? 0;
|
|
2719
|
-
const completionTokens = raw.usage?.completion_tokens ?? 0;
|
|
2720
|
-
return {
|
|
2721
|
-
content,
|
|
2722
|
-
usage: {
|
|
2723
|
-
promptTokens,
|
|
2724
|
-
completionTokens,
|
|
2725
|
-
totalTokens: promptTokens + completionTokens
|
|
2726
|
-
},
|
|
2727
|
-
costUsd: null,
|
|
2728
|
-
model: req.model ?? defaultModel,
|
|
2729
|
-
durationMs: 0,
|
|
2730
|
-
finishReason: raw.choices?.[0]?.finish_reason ?? null,
|
|
2731
|
-
contentEmpty: content.trim().length === 0,
|
|
2732
|
-
raw
|
|
2733
|
-
};
|
|
2734
|
-
}
|
|
2735
|
-
});
|
|
2736
|
-
}
|
|
2737
|
-
async function consultAnalyst(task, messages, instruction, opts) {
|
|
2738
|
-
const trajectory = compactTrajectory(messages);
|
|
2739
|
-
const analystModel = opts.analystModel ?? opts.model;
|
|
2740
|
-
const chat = analystChat(opts, analystModel);
|
|
2741
|
-
const consultMessages = trajectory ? [
|
|
2742
|
-
{ role: "system", content: instruction },
|
|
2743
|
-
{
|
|
2744
|
-
role: "user",
|
|
2745
|
-
content: `TASK: ${task.userPrompt.slice(0, 1500)}
|
|
2746
|
-
|
|
2747
|
-
TRAJECTORY:
|
|
2748
|
-
${trajectory}`
|
|
2749
|
-
}
|
|
2750
|
-
] : [
|
|
2751
|
-
{
|
|
2752
|
-
role: "user",
|
|
2753
|
-
content: `${instruction}
|
|
2754
|
-
|
|
2755
|
-
TASK:
|
|
2756
|
-
${task.userPrompt.slice(0, 1500)}`
|
|
2757
|
-
}
|
|
2758
|
-
];
|
|
2759
|
-
const res = await chat.chat({
|
|
2760
|
-
model: analystModel,
|
|
2761
|
-
temperature: 0.2,
|
|
2762
|
-
maxTokens: 1024,
|
|
2763
|
-
messages: consultMessages
|
|
2764
|
-
});
|
|
2765
|
-
const usage = res.usage;
|
|
2766
|
-
return {
|
|
2767
|
-
steer: res.content.trim(),
|
|
2768
|
-
tokens: {
|
|
2769
|
-
input: usage?.promptTokens ?? usage?.prompt_tokens ?? 0,
|
|
2770
|
-
output: usage?.completionTokens ?? usage?.completion_tokens ?? 0
|
|
2771
|
-
}
|
|
2772
|
-
};
|
|
2773
|
-
}
|
|
2774
|
-
async function analyze(task, messages, opts) {
|
|
2775
|
-
const trajectory = compactTrajectory(messages);
|
|
2776
|
-
const analystModel = opts.analystModel ?? opts.model;
|
|
2777
|
-
const inner = analystChat(opts, analystModel);
|
|
2778
|
-
const tokens = { input: 0, output: 0 };
|
|
2779
|
-
const chat = {
|
|
2780
|
-
...inner,
|
|
2781
|
-
chat: async (req, callOpts) => {
|
|
2782
|
-
const res = await inner.chat(req, callOpts);
|
|
2783
|
-
const u = res.usage;
|
|
2784
|
-
if (u) {
|
|
2785
|
-
tokens.input += u.promptTokens ?? u.prompt_tokens ?? 0;
|
|
2786
|
-
tokens.output += u.completionTokens ?? u.completion_tokens ?? 0;
|
|
2787
|
-
}
|
|
2788
|
-
return res;
|
|
2789
|
-
}
|
|
2790
|
-
};
|
|
2791
|
-
const obs = await observe(
|
|
2792
|
-
{
|
|
2793
|
-
task: task.userPrompt,
|
|
2794
|
-
output: trajectory,
|
|
2795
|
-
trace: messages,
|
|
2796
|
-
outcome: "failed",
|
|
2797
|
-
runId: task.id
|
|
2798
|
-
},
|
|
2799
|
-
{
|
|
2800
|
-
chat,
|
|
2801
|
-
model: analystModel,
|
|
2802
|
-
...opts.analystInstruction ? { analystInstruction: opts.analystInstruction } : {},
|
|
2803
|
-
...opts.corpus ? { corpus: opts.corpus, tags: opts.corpusTags ?? [] } : {}
|
|
2804
|
-
}
|
|
2805
|
-
);
|
|
2806
|
-
const steer = obs.findings.map((f) => f.recommended_action).filter((a) => typeof a === "string" && a.trim().length > 0).join("\n").trim();
|
|
2807
|
-
return { steer: steer || "COMPLETE", tokens };
|
|
2808
|
-
}
|
|
2809
|
-
async function renderCorpusReadback(opts) {
|
|
2810
|
-
if (!opts.corpus || !opts.corpusReadback) return "";
|
|
2811
|
-
const maxFacts = opts.corpusReadback.maxFacts ?? 3;
|
|
2812
|
-
if (!Number.isInteger(maxFacts) || maxFacts < 0) {
|
|
2813
|
-
throw new Error(`corpusReadback.maxFacts must be a non-negative integer, got ${maxFacts}`);
|
|
2814
|
-
}
|
|
2815
|
-
if (maxFacts === 0) return "";
|
|
2816
|
-
const tags = [
|
|
2817
|
-
...opts.corpusTags ?? [],
|
|
2818
|
-
...opts.corpusReadback.tags ?? [],
|
|
2819
|
-
...opts.corpusReadback.includeOperatorFacts ? [] : ["audience:agent"]
|
|
2820
|
-
];
|
|
2821
|
-
const facts = await opts.corpus.query({
|
|
2822
|
-
...tags.length > 0 ? { tags } : {},
|
|
2823
|
-
minConfidence: opts.corpusReadback.minConfidence ?? 0.7,
|
|
2824
|
-
limit: maxFacts
|
|
2825
|
-
});
|
|
2826
|
-
if (facts.length === 0) return "";
|
|
2827
|
-
const rendered = facts.map(
|
|
2828
|
-
(fact) => fact.rationale ? `- ${fact.claim} (${fact.rationale})` : `- ${fact.claim}`
|
|
2829
|
-
);
|
|
2830
|
-
return `Relevant learned facts from prior attempts:
|
|
2831
|
-
${rendered.join("\n")}`;
|
|
2832
|
-
}
|
|
2833
|
-
function shotExecutor(surface, opts) {
|
|
2834
|
-
let artifact;
|
|
2835
|
-
return {
|
|
2836
|
-
runtime: "agentic-shot",
|
|
2837
|
-
async execute(task) {
|
|
2838
|
-
const t = task;
|
|
2839
|
-
const own = !t.handle;
|
|
2840
|
-
const handle = t.handle ?? await surface.open(t.task);
|
|
2841
|
-
try {
|
|
2842
|
-
const allTools = await surface.tools(t.task, handle);
|
|
2843
|
-
let tools = allTools;
|
|
2844
|
-
if (t.tools) {
|
|
2845
|
-
const known = new Set(allTools.map((tool) => tool.function.name));
|
|
2846
|
-
const unknown = t.tools.filter((name) => !known.has(name));
|
|
2847
|
-
if (unknown.length > 0) {
|
|
2848
|
-
throw new Error(
|
|
2849
|
-
`shot tools: unknown tool name(s) ${unknown.join(", ")} \u2014 domain offers: ${[...known].join(", ")}`
|
|
2850
|
-
);
|
|
2851
|
-
}
|
|
2852
|
-
const want = new Set(t.tools);
|
|
2853
|
-
tools = allTools.filter((tool) => want.has(tool.function.name));
|
|
2854
|
-
}
|
|
2855
|
-
const messages = t.messages?.length ? t.messages : [
|
|
2856
|
-
{ role: "system", content: t.persona?.systemPrompt ?? t.task.systemPrompt },
|
|
2857
|
-
{ role: "user", content: `${t.task.userPrompt}
|
|
2858
|
-
|
|
2859
|
-
${taskNudge}` }
|
|
2860
|
-
];
|
|
2861
|
-
if (t.messages?.length && t.persona?.systemPrompt) {
|
|
2862
|
-
messages.push({
|
|
2863
|
-
role: "user",
|
|
2864
|
-
content: `[hand-off] You are now acting as: ${t.persona.systemPrompt}`
|
|
2865
|
-
});
|
|
2866
|
-
}
|
|
2867
|
-
if (t.steer) messages.push({ role: "user", content: t.steer });
|
|
2868
|
-
const shot = await runShot(surface, t.task, handle, tools, messages, opts, t.persona?.model);
|
|
2869
|
-
const s = await surface.score(t.task, handle);
|
|
2870
|
-
const score = s.total > 0 ? s.passes / s.total : 0;
|
|
2871
|
-
const out = {
|
|
2872
|
-
messages: shot.messages,
|
|
2873
|
-
score,
|
|
2874
|
-
passes: s.passes,
|
|
2875
|
-
total: s.total,
|
|
2876
|
-
completions: shot.completions,
|
|
2877
|
-
toolErrors: shot.toolErrors
|
|
2878
|
-
};
|
|
2879
|
-
artifact = {
|
|
2880
|
-
outRef: `shot:${handle.id}:${shot.completions}:${s.passes}/${s.total}`,
|
|
2881
|
-
out,
|
|
2882
|
-
verdict: { valid: s.total > 0 && s.passes === s.total, score },
|
|
2883
|
-
// Real usage to the conserved pool: tokens from the router responses; usd only
|
|
2884
|
-
// when the model is in the price table (never a fabricated number).
|
|
2885
|
-
spent: {
|
|
2886
|
-
iterations: shot.completions,
|
|
2887
|
-
tokens: shot.tokens,
|
|
2888
|
-
usd: isModelPriced(opts.model) ? estimateCost(shot.tokens.input, shot.tokens.output, opts.model) : 0,
|
|
2889
|
-
ms: 0
|
|
2890
|
-
}
|
|
2891
|
-
};
|
|
2892
|
-
return artifact;
|
|
2893
|
-
} finally {
|
|
2894
|
-
if (own) await surface.close(handle);
|
|
2895
|
-
}
|
|
2896
|
-
},
|
|
2897
|
-
teardown: () => Promise.resolve({ destroyed: true }),
|
|
2898
|
-
resultArtifact() {
|
|
2899
|
-
if (!artifact) throw new Error("shotExecutor: resultArtifact before execute");
|
|
2900
|
-
return artifact;
|
|
2901
|
-
}
|
|
2902
|
-
};
|
|
2903
|
-
}
|
|
2904
|
-
function analystExecutor(opts) {
|
|
2905
|
-
let artifact;
|
|
2906
|
-
return {
|
|
2907
|
-
runtime: "agentic-analyst",
|
|
2908
|
-
async execute(task) {
|
|
2909
|
-
const t = task;
|
|
2910
|
-
const { steer, tokens } = t.rawInstruction ? await consultAnalyst(t.task, t.messages, t.rawInstruction, opts) : await analyze(t.task, t.messages, opts);
|
|
2911
|
-
const analystModel = opts.analystModel ?? opts.model;
|
|
2912
|
-
artifact = {
|
|
2913
|
-
outRef: `analyst:${steer.length}`,
|
|
2914
|
-
out: steer,
|
|
2915
|
-
spent: {
|
|
2916
|
-
iterations: 1,
|
|
2917
|
-
tokens,
|
|
2918
|
-
usd: isModelPriced(analystModel) ? estimateCost(tokens.input, tokens.output, analystModel) : 0,
|
|
2919
|
-
ms: 0
|
|
2920
|
-
}
|
|
2921
|
-
};
|
|
2922
|
-
return artifact;
|
|
2923
|
-
},
|
|
2924
|
-
teardown: () => Promise.resolve({ destroyed: true }),
|
|
2925
|
-
resultArtifact() {
|
|
2926
|
-
if (!artifact) throw new Error("analystExecutor: resultArtifact before execute");
|
|
2927
|
-
return artifact;
|
|
2928
|
-
}
|
|
2929
|
-
};
|
|
2930
|
-
}
|
|
2931
|
-
function agenticRegistry(surface, opts) {
|
|
2932
|
-
const leaves = {
|
|
2933
|
-
register() {
|
|
2934
|
-
throw new Error("agenticRegistry: register unsupported");
|
|
2935
|
-
},
|
|
2936
|
-
resolve(spec) {
|
|
2937
|
-
const role = spec.profile.metadata?.role;
|
|
2938
|
-
const factory = (_s, _ctx) => role === "analyst" ? analystExecutor(opts) : shotExecutor(surface, opts);
|
|
2939
|
-
return { succeeded: true, value: factory };
|
|
2940
|
-
}
|
|
2941
|
-
};
|
|
2942
|
-
return withDriverExecutor(leaves);
|
|
2943
|
-
}
|
|
2944
|
-
function leaf(name, role) {
|
|
2945
|
-
const agent = {
|
|
2946
|
-
name,
|
|
2947
|
-
executorSpec: { profile: { name, metadata: { role } }, harness: null },
|
|
2948
|
-
act() {
|
|
2949
|
-
throw new Error(`agentic: spawned child "${name}" was run directly (the executor drives it)`);
|
|
2950
|
-
}
|
|
2951
|
-
};
|
|
2952
|
-
return agent;
|
|
2953
|
-
}
|
|
2954
|
-
async function drainOne2(scope) {
|
|
2955
|
-
const s = await scope.next();
|
|
2956
|
-
if (!s) throw new Error("agentic: spawned child never settled");
|
|
2957
|
-
return s;
|
|
2958
|
-
}
|
|
2959
|
-
var perChild = (innerTurns) => ({
|
|
2960
|
-
maxIterations: innerTurns + 1,
|
|
2961
|
-
maxTokens: 1e6
|
|
2962
|
-
});
|
|
2963
|
-
function depthStrategy(surface, task, opts, cfg) {
|
|
2964
|
-
const innerTurns = opts.innerTurns ?? 4;
|
|
2965
|
-
let pendingSteer;
|
|
2966
|
-
return {
|
|
2967
|
-
name: "depth",
|
|
2968
|
-
async act(_t, scope) {
|
|
2969
|
-
const handle = await surface.open(task);
|
|
2970
|
-
const progression = [];
|
|
2971
|
-
let messages;
|
|
2972
|
-
let completions = 0;
|
|
2973
|
-
let shots = 0;
|
|
2974
|
-
try {
|
|
2975
|
-
for (shots = 0; shots < cfg.maxShots; shots += 1) {
|
|
2976
|
-
const child = leaf(`shot:${shots}`, "shot");
|
|
2977
|
-
const memorySteer = await renderCorpusReadback(opts);
|
|
2978
|
-
const steer = [shots === 0 ? void 0 : pendingSteer, memorySteer].filter((part) => typeof part === "string" && part.trim().length > 0).join("\n\n");
|
|
2979
|
-
const res = scope.spawn(child, { task, handle, messages, steer }, {
|
|
2980
|
-
budget: perChild(innerTurns),
|
|
2981
|
-
label: `shot:${shots}`
|
|
2982
|
-
});
|
|
2983
|
-
if (!res.ok) break;
|
|
2984
|
-
const settled = await drainOne2(scope);
|
|
2985
|
-
if (settled.kind === "down") break;
|
|
2986
|
-
const out = settled.out;
|
|
2987
|
-
messages = out.messages;
|
|
2988
|
-
completions += out.completions;
|
|
2989
|
-
progression.push(out.score);
|
|
2990
|
-
if (out.score >= 1 || shots === cfg.maxShots - 1) break;
|
|
2991
|
-
const aChild = leaf(`analyst:${shots}`, "analyst");
|
|
2992
|
-
const aRes = scope.spawn(
|
|
2993
|
-
aChild,
|
|
2994
|
-
{ task, messages },
|
|
2995
|
-
{ budget: perChild(1), label: `analyst:${shots}` }
|
|
2996
|
-
);
|
|
2997
|
-
if (!aRes.ok) break;
|
|
2998
|
-
const aSettled = await drainOne2(scope);
|
|
2999
|
-
completions += 1;
|
|
3000
|
-
if (aSettled.kind === "down") break;
|
|
3001
|
-
const findings = aSettled.out;
|
|
3002
|
-
if (/^\s*COMPLETE\b/i.test(findings)) break;
|
|
3003
|
-
pendingSteer = `A reviewer flagged unfinished items:
|
|
3004
|
-
${findings}
|
|
3005
|
-
|
|
3006
|
-
Address each with the tools, verify they took, then continue.`;
|
|
3007
|
-
}
|
|
3008
|
-
const final = await surface.score(task, handle);
|
|
3009
|
-
const score = final.total > 0 ? final.passes / final.total : 0;
|
|
3010
|
-
return {
|
|
3011
|
-
kind: "done",
|
|
3012
|
-
deliverable: {
|
|
3013
|
-
mode: "depth",
|
|
3014
|
-
score,
|
|
3015
|
-
resolved: final.total > 0 && final.passes === final.total,
|
|
3016
|
-
completions,
|
|
3017
|
-
progression,
|
|
3018
|
-
shots: shots + 1
|
|
3019
|
-
}
|
|
3020
|
-
};
|
|
3021
|
-
} finally {
|
|
3022
|
-
await surface.close(handle);
|
|
3023
|
-
}
|
|
3024
|
-
}
|
|
3025
|
-
};
|
|
3026
|
-
}
|
|
3027
|
-
function breadthStrategy(_surface, task, opts, cfg) {
|
|
3028
|
-
const innerTurns = opts.innerTurns ?? 4;
|
|
3029
|
-
return {
|
|
3030
|
-
name: "breadth",
|
|
3031
|
-
async act(_t, scope) {
|
|
3032
|
-
let opened = 0;
|
|
3033
|
-
for (let k = 0; k < cfg.width; k += 1) {
|
|
3034
|
-
const res = scope.spawn(leaf(`rollout:${k}`, "shot"), { task }, {
|
|
3035
|
-
budget: perChild(innerTurns),
|
|
3036
|
-
label: `rollout:${k}`
|
|
3037
|
-
});
|
|
3038
|
-
if (res.ok) opened += 1;
|
|
3039
|
-
}
|
|
3040
|
-
if (opened === 0) return { kind: "blocked", blockers: ["breadth: pool admitted no rollout"] };
|
|
3041
|
-
let best = -1;
|
|
3042
|
-
let bestResolved = false;
|
|
3043
|
-
let completions = 0;
|
|
3044
|
-
const progression = [];
|
|
3045
|
-
for (let s = await scope.next(); s !== null; s = await scope.next()) {
|
|
3046
|
-
if (s.kind === "down") continue;
|
|
3047
|
-
const out = s.out;
|
|
3048
|
-
completions += out.completions;
|
|
3049
|
-
if (out.score > best) best = out.score;
|
|
3050
|
-
if (out.total > 0 && out.passes === out.total) bestResolved = true;
|
|
3051
|
-
progression.push(best);
|
|
3052
|
-
}
|
|
3053
|
-
if (best < 0) return { kind: "blocked", blockers: ["breadth: every rollout went down"] };
|
|
3054
|
-
return {
|
|
3055
|
-
kind: "done",
|
|
3056
|
-
deliverable: {
|
|
3057
|
-
mode: "breadth",
|
|
3058
|
-
score: best,
|
|
3059
|
-
resolved: bestResolved,
|
|
3060
|
-
completions,
|
|
3061
|
-
progression,
|
|
3062
|
-
shots: opened
|
|
3063
|
-
}
|
|
3064
|
-
};
|
|
3065
|
-
}
|
|
3066
|
-
};
|
|
3067
|
-
}
|
|
3068
|
-
var sample = {
|
|
3069
|
-
name: "sample",
|
|
3070
|
-
driver: (surface, task, opts, budget) => breadthStrategy(surface, task, opts, { width: budget })
|
|
3071
|
-
};
|
|
3072
|
-
var refine = {
|
|
3073
|
-
name: "refine",
|
|
3074
|
-
driver: (surface, task, opts, budget) => depthStrategy(surface, task, opts, { maxShots: budget })
|
|
3075
|
-
};
|
|
3076
|
-
function defineStrategy(name, run) {
|
|
3077
|
-
return {
|
|
3078
|
-
name,
|
|
3079
|
-
driver: (surface, task, opts, budget) => ({
|
|
3080
|
-
name,
|
|
3081
|
-
async act(_t, scope) {
|
|
3082
|
-
let seq = 0;
|
|
3083
|
-
const innerTurns = opts.innerTurns ?? 4;
|
|
3084
|
-
let verifiedBest = 0;
|
|
3085
|
-
let verifiedResolved = false;
|
|
3086
|
-
const openHandles = /* @__PURE__ */ new Set();
|
|
3087
|
-
const ctx = {
|
|
3088
|
-
// Narrowed to open/close — the body gets no raw call()/score() access.
|
|
3089
|
-
surface: {
|
|
3090
|
-
name: surface.name,
|
|
3091
|
-
open: async (t) => {
|
|
3092
|
-
const h = await surface.open(t);
|
|
3093
|
-
openHandles.add(h.id);
|
|
3094
|
-
return h;
|
|
3095
|
-
},
|
|
3096
|
-
close: async (h) => {
|
|
3097
|
-
if (!h || !openHandles.has(h.id)) return;
|
|
3098
|
-
openHandles.delete(h.id);
|
|
3099
|
-
await surface.close(h);
|
|
3100
|
-
}
|
|
3101
|
-
},
|
|
3102
|
-
task,
|
|
3103
|
-
opts,
|
|
3104
|
-
budget,
|
|
3105
|
-
scope,
|
|
3106
|
-
async shot(spec) {
|
|
3107
|
-
const child = leaf(`shot:${seq}`, "shot");
|
|
3108
|
-
seq += 1;
|
|
3109
|
-
const res = scope.spawn(
|
|
3110
|
-
child,
|
|
3111
|
-
{
|
|
3112
|
-
task,
|
|
3113
|
-
handle: spec?.handle,
|
|
3114
|
-
messages: spec?.messages,
|
|
3115
|
-
steer: spec?.steer,
|
|
3116
|
-
persona: spec?.persona,
|
|
3117
|
-
tools: spec?.tools
|
|
3118
|
-
},
|
|
3119
|
-
{ budget: perChild(innerTurns), label: child.name }
|
|
3120
|
-
);
|
|
3121
|
-
if (!res.ok) return null;
|
|
3122
|
-
const settled = await drainOne2(scope);
|
|
3123
|
-
if (settled.kind === "down") return null;
|
|
3124
|
-
const out = settled.out;
|
|
3125
|
-
if (out.score > verifiedBest) verifiedBest = out.score;
|
|
3126
|
-
if (out.total > 0 && out.passes === out.total) verifiedResolved = true;
|
|
3127
|
-
return out;
|
|
3128
|
-
},
|
|
3129
|
-
async listTools(handle) {
|
|
3130
|
-
const tools = await surface.tools(task, handle);
|
|
3131
|
-
return tools.map((t) => ({
|
|
3132
|
-
name: t.function.name,
|
|
3133
|
-
...t.function.description ? { description: t.function.description } : {}
|
|
3134
|
-
}));
|
|
3135
|
-
},
|
|
3136
|
-
async critique(messages) {
|
|
3137
|
-
const child = leaf(`analyst:${seq}`, "analyst");
|
|
3138
|
-
seq += 1;
|
|
3139
|
-
const res = scope.spawn(
|
|
3140
|
-
child,
|
|
3141
|
-
{ task, messages },
|
|
3142
|
-
{ budget: perChild(1), label: child.name }
|
|
3143
|
-
);
|
|
3144
|
-
if (!res.ok) return null;
|
|
3145
|
-
const settled = await drainOne2(scope);
|
|
3146
|
-
if (settled.kind === "down") return null;
|
|
3147
|
-
const findings = settled.out;
|
|
3148
|
-
return /^\s*COMPLETE\b/i.test(findings) ? null : findings;
|
|
3149
|
-
},
|
|
3150
|
-
async consult(messages, instruction) {
|
|
3151
|
-
const child = leaf(`analyst:${seq}`, "analyst");
|
|
3152
|
-
seq += 1;
|
|
3153
|
-
const res = scope.spawn(
|
|
3154
|
-
child,
|
|
3155
|
-
{ task, messages, rawInstruction: instruction },
|
|
3156
|
-
{ budget: perChild(1), label: child.name }
|
|
3157
|
-
);
|
|
3158
|
-
if (!res.ok) return null;
|
|
3159
|
-
const settled = await drainOne2(scope);
|
|
3160
|
-
if (settled.kind === "down") return null;
|
|
3161
|
-
return settled.out;
|
|
3162
|
-
}
|
|
3163
|
-
};
|
|
3164
|
-
const r = await run(ctx);
|
|
3165
|
-
return {
|
|
3166
|
-
kind: "done",
|
|
3167
|
-
deliverable: {
|
|
3168
|
-
mode: name,
|
|
3169
|
-
...r,
|
|
3170
|
-
progression: Array.isArray(r.progression) ? r.progression : [],
|
|
3171
|
-
completions: typeof r.completions === "number" ? r.completions : 0,
|
|
3172
|
-
shots: typeof r.shots === "number" ? r.shots : 0,
|
|
3173
|
-
score: verifiedBest,
|
|
3174
|
-
resolved: verifiedResolved
|
|
3175
|
-
}
|
|
3176
|
-
};
|
|
3177
|
-
}
|
|
3178
|
-
})
|
|
3179
|
-
};
|
|
3180
|
-
}
|
|
3181
|
-
var adaptiveRefine = defineStrategy(
|
|
3182
|
-
"adaptiveRefine",
|
|
3183
|
-
async ({ surface, task, budget, shot, critique }) => {
|
|
3184
|
-
let handle = await surface.open(task);
|
|
3185
|
-
const progression = [];
|
|
3186
|
-
let messages;
|
|
3187
|
-
let steer;
|
|
3188
|
-
let completions = 0;
|
|
3189
|
-
let best = -1;
|
|
3190
|
-
let shots = 0;
|
|
3191
|
-
try {
|
|
3192
|
-
for (shots = 0; shots < budget; shots += 1) {
|
|
3193
|
-
const out = await shot({ handle, messages, steer });
|
|
3194
|
-
if (!out) break;
|
|
3195
|
-
completions += out.completions;
|
|
3196
|
-
progression.push(out.score);
|
|
3197
|
-
if (out.score >= 1) break;
|
|
3198
|
-
if (out.score <= best) {
|
|
3199
|
-
await surface.close(handle);
|
|
3200
|
-
handle = await surface.open(task);
|
|
3201
|
-
messages = void 0;
|
|
3202
|
-
steer = void 0;
|
|
3203
|
-
continue;
|
|
3204
|
-
}
|
|
3205
|
-
best = out.score;
|
|
3206
|
-
messages = out.messages;
|
|
3207
|
-
const findings = await critique(out.messages);
|
|
3208
|
-
completions += 1;
|
|
3209
|
-
if (!findings) break;
|
|
3210
|
-
steer = `A reviewer flagged unfinished items:
|
|
3211
|
-
${findings}
|
|
3212
|
-
|
|
3213
|
-
Address each with the tools, verify they took, then continue.`;
|
|
3214
|
-
}
|
|
3215
|
-
const score = progression.length ? Math.max(...progression) : 0;
|
|
3216
|
-
return { score, resolved: score >= 1, completions, progression, shots };
|
|
3217
|
-
} finally {
|
|
3218
|
-
await surface.close(handle);
|
|
3219
|
-
}
|
|
3220
|
-
}
|
|
3221
|
-
);
|
|
3222
|
-
var sampleThenRefine = defineStrategy(
|
|
3223
|
-
"sampleThenRefine",
|
|
3224
|
-
async ({ surface, task, budget, shot, critique }) => {
|
|
3225
|
-
const explore = Math.max(1, Math.ceil(budget / 2));
|
|
3226
|
-
const open = /* @__PURE__ */ new Set();
|
|
3227
|
-
const progression = [];
|
|
3228
|
-
let completions = 0;
|
|
3229
|
-
let shots = 0;
|
|
3230
|
-
try {
|
|
3231
|
-
let best;
|
|
3232
|
-
for (let i = 0; i < explore; i += 1) {
|
|
3233
|
-
const handle = await surface.open(task);
|
|
3234
|
-
open.add(handle);
|
|
3235
|
-
const out = await shot({ handle });
|
|
3236
|
-
if (!out) continue;
|
|
3237
|
-
shots += 1;
|
|
3238
|
-
completions += out.completions;
|
|
3239
|
-
progression.push(out.score);
|
|
3240
|
-
if (!best || out.score > best.out.score) best = { handle, out };
|
|
3241
|
-
if (out.score >= 1) break;
|
|
3242
|
-
}
|
|
3243
|
-
if (!best) return { score: 0, resolved: false, completions, progression, shots };
|
|
3244
|
-
for (const h of [...open]) {
|
|
3245
|
-
if (h !== best.handle) {
|
|
3246
|
-
await surface.close(h);
|
|
3247
|
-
open.delete(h);
|
|
3248
|
-
}
|
|
3249
|
-
}
|
|
3250
|
-
let messages = best.out.messages;
|
|
3251
|
-
let topScore = best.out.score;
|
|
3252
|
-
for (let i = explore; i < budget && topScore < 1; i += 1) {
|
|
3253
|
-
const findings = await critique(messages);
|
|
3254
|
-
completions += 1;
|
|
3255
|
-
if (!findings) break;
|
|
3256
|
-
const out = await shot({
|
|
3257
|
-
handle: best.handle,
|
|
3258
|
-
messages,
|
|
3259
|
-
steer: `A reviewer flagged unfinished items:
|
|
3260
|
-
${findings}
|
|
3261
|
-
|
|
3262
|
-
Address each with the tools, verify they took, then continue.`
|
|
3263
|
-
});
|
|
3264
|
-
if (!out) break;
|
|
3265
|
-
shots += 1;
|
|
3266
|
-
completions += out.completions;
|
|
3267
|
-
progression.push(out.score);
|
|
3268
|
-
messages = out.messages;
|
|
3269
|
-
if (out.score > topScore) topScore = out.score;
|
|
3270
|
-
}
|
|
3271
|
-
const score = progression.length ? Math.max(...progression) : 0;
|
|
3272
|
-
return { score, resolved: score >= 1, completions, progression, shots };
|
|
3273
|
-
} finally {
|
|
3274
|
-
for (const h of open) await surface.close(h);
|
|
3275
|
-
}
|
|
3276
|
-
}
|
|
3277
|
-
);
|
|
3278
|
-
async function runAgentic(opts) {
|
|
3279
|
-
const strategy = opts.strategy ?? (opts.mode === "breadth" ? sample : refine);
|
|
3280
|
-
const driver = strategy.driver(opts.surface, opts.task, opts, opts.budget);
|
|
3281
|
-
const supervisor = createSupervisor();
|
|
3282
|
-
const root = opts.rootBudget ?? {
|
|
3283
|
-
maxIterations: opts.budget * ((opts.innerTurns ?? 4) + 2),
|
|
3284
|
-
maxTokens: 1e9
|
|
3285
|
-
};
|
|
3286
|
-
const started = Date.now();
|
|
3287
|
-
const result = await supervisor.run(driver, void 0, {
|
|
3288
|
-
budget: root,
|
|
3289
|
-
runId: `agentic:${strategy.name}:${opts.task.id}`,
|
|
3290
|
-
journal: new InMemorySpawnJournal(),
|
|
3291
|
-
blobs: new InMemoryResultBlobStore(),
|
|
3292
|
-
executors: agenticRegistry(opts.surface, opts),
|
|
3293
|
-
maxDepth: 3,
|
|
3294
|
-
...opts.hooks ? { hooks: opts.hooks } : {}
|
|
3295
|
-
});
|
|
3296
|
-
if (result.kind !== "winner" || result.out.kind !== "done") {
|
|
3297
|
-
const reason = result.kind === "winner" ? `blocked: ${result.out.blockers?.join("; ")}` : `no-winner: ${result.reason}`;
|
|
3298
|
-
throw new Error(`runAgentic(${strategy.name}) produced no result \u2014 ${reason}`);
|
|
3299
|
-
}
|
|
3300
|
-
const core = result.out.deliverable;
|
|
3301
|
-
return {
|
|
3302
|
-
...core,
|
|
3303
|
-
usd: result.spentTotal.usd,
|
|
3304
|
-
tokens: result.spentTotal.tokens,
|
|
3305
|
-
ms: Date.now() - started
|
|
3306
|
-
};
|
|
3307
|
-
}
|
|
3308
|
-
|
|
3309
|
-
// src/runtime/run-benchmark.ts
|
|
3310
2501
|
async function pool(items, limit, fn) {
|
|
3311
2502
|
const out = new Array(items.length);
|
|
3312
2503
|
let next = 0;
|
|
@@ -5038,381 +4229,6 @@ function deriveTurnSignal(callerSignal, timeoutMs) {
|
|
|
5038
4229
|
};
|
|
5039
4230
|
}
|
|
5040
4231
|
|
|
5041
|
-
// src/runtime/structural-rollout.ts
|
|
5042
|
-
import { randomBytes } from "crypto";
|
|
5043
|
-
var defaultStructuralRolloutPolicy = {
|
|
5044
|
-
k: 5,
|
|
5045
|
-
repairRounds: 2,
|
|
5046
|
-
testgen: 6
|
|
5047
|
-
};
|
|
5048
|
-
function resolvePolicy(overrides) {
|
|
5049
|
-
const policy = { ...defaultStructuralRolloutPolicy, ...overrides };
|
|
5050
|
-
if (!Number.isInteger(policy.k) || policy.k < 1) {
|
|
5051
|
-
throw new Error(`structuralRollout: policy.k must be an integer >= 1, got ${policy.k}`);
|
|
5052
|
-
}
|
|
5053
|
-
if (!Number.isInteger(policy.repairRounds) || policy.repairRounds < 0) {
|
|
5054
|
-
throw new Error(
|
|
5055
|
-
`structuralRollout: policy.repairRounds must be an integer >= 0, got ${policy.repairRounds}`
|
|
5056
|
-
);
|
|
5057
|
-
}
|
|
5058
|
-
if (!Number.isInteger(policy.testgen) || policy.testgen < 0) {
|
|
5059
|
-
throw new Error(
|
|
5060
|
-
`structuralRollout: policy.testgen must be an integer >= 0, got ${policy.testgen}`
|
|
5061
|
-
);
|
|
5062
|
-
}
|
|
5063
|
-
return policy;
|
|
5064
|
-
}
|
|
5065
|
-
var authorInstruction = (count, entry) => `Read the task below. Write exactly ${count} single-line assert statements that test the function \`${entry}\`, based ONLY on the behavior the task itself describes. Each assert must be one physical line of the form \`assert ${entry}(...) == expected\` (or a True/False check). Do NOT implement the function. Do NOT copy shown examples verbatim if you can test other cases too. Output ONLY the assert lines inside a single \`\`\`python code block.`;
|
|
5066
|
-
function filterAuthoredAsserts(reply, entrySymbol, count) {
|
|
5067
|
-
const fences = [...reply.matchAll(/```(?:python|py)?\s*\n([\s\S]*?)```/gi)].map(
|
|
5068
|
-
(m) => (m[1] ?? "").trim()
|
|
5069
|
-
);
|
|
5070
|
-
const block = fences.length > 0 ? fences.join("\n") : reply;
|
|
5071
|
-
const balanced = (s) => {
|
|
5072
|
-
let d = 0;
|
|
5073
|
-
for (const ch of s) {
|
|
5074
|
-
if (ch === "(" || ch === "[" || ch === "{") d += 1;
|
|
5075
|
-
else if (ch === ")" || ch === "]" || ch === "}") d -= 1;
|
|
5076
|
-
if (d < 0) return false;
|
|
5077
|
-
}
|
|
5078
|
-
return d === 0;
|
|
5079
|
-
};
|
|
5080
|
-
return block.split("\n").map((l) => l.trim()).filter((l) => l.startsWith("assert ") && l.includes(entrySymbol) && balanced(l)).slice(0, count);
|
|
5081
|
-
}
|
|
5082
|
-
function modelAuthoredChecks(overrides = {}) {
|
|
5083
|
-
return {
|
|
5084
|
-
async generate(_task, ctx) {
|
|
5085
|
-
const count = overrides.count ?? ctx.count;
|
|
5086
|
-
if (count <= 0 || !ctx.entrySymbol) return [];
|
|
5087
|
-
const entry = ctx.entrySymbol;
|
|
5088
|
-
const reply = await ctx.consult(authorInstruction(count, entry));
|
|
5089
|
-
if (!reply) return [];
|
|
5090
|
-
return filterAuthoredAsserts(reply, entry, count).map((code) => ({
|
|
5091
|
-
code,
|
|
5092
|
-
kind: "authored"
|
|
5093
|
-
}));
|
|
5094
|
-
}
|
|
5095
|
-
};
|
|
5096
|
-
}
|
|
5097
|
-
function officialChecksFromMeta(key = "visibleChecks") {
|
|
5098
|
-
return {
|
|
5099
|
-
async generate(task) {
|
|
5100
|
-
const raw = task.meta?.[key];
|
|
5101
|
-
if (!Array.isArray(raw)) return [];
|
|
5102
|
-
return raw.filter((c) => typeof c === "string" && c.trim().length > 0).map((code) => ({ code, kind: "official" }));
|
|
5103
|
-
}
|
|
5104
|
-
};
|
|
5105
|
-
}
|
|
5106
|
-
function composeCheckSources(...sources) {
|
|
5107
|
-
return {
|
|
5108
|
-
async generate(task, ctx) {
|
|
5109
|
-
const all = [];
|
|
5110
|
-
for (const source of sources) all.push(...await source.generate(task, ctx));
|
|
5111
|
-
return all;
|
|
5112
|
-
}
|
|
5113
|
-
};
|
|
5114
|
-
}
|
|
5115
|
-
function resolveEntrySymbol(task) {
|
|
5116
|
-
const meta = task.meta?.entryPoint;
|
|
5117
|
-
if (typeof meta === "string" && meta.trim().length > 0) return meta.trim();
|
|
5118
|
-
const defs = [...task.userPrompt.matchAll(/(?:^|\n)\s*def\s+([A-Za-z_]\w*)\s*\(/g)];
|
|
5119
|
-
const last = defs[defs.length - 1];
|
|
5120
|
-
return last?.[1];
|
|
5121
|
-
}
|
|
5122
|
-
function buildCheckProgram(candidate, official, authored, nonce) {
|
|
5123
|
-
const officialB64 = Buffer.from(JSON.stringify(official), "utf8").toString("base64");
|
|
5124
|
-
const authoredB64 = Buffer.from(JSON.stringify(authored), "utf8").toString("base64");
|
|
5125
|
-
return `${candidate}
|
|
5126
|
-
|
|
5127
|
-
import base64 as _b64, json as _json, sys as _sys
|
|
5128
|
-
_official = _json.loads(_b64.b64decode("${officialB64}").decode("utf8"))
|
|
5129
|
-
_authored = _json.loads(_b64.b64decode("${authoredB64}").decode("utf8"))
|
|
5130
|
-
_lines = []
|
|
5131
|
-
def _run(_tests):
|
|
5132
|
-
_att, _fail = 0, 0
|
|
5133
|
-
for _t in _tests:
|
|
5134
|
-
_att += 1
|
|
5135
|
-
try:
|
|
5136
|
-
exec(_t, dict(globals()))
|
|
5137
|
-
except Exception as _e:
|
|
5138
|
-
_fail += 1
|
|
5139
|
-
_lines.append("CHECK FAILED: %s -> %s: %s" % (_t.strip()[:200], type(_e).__name__, str(_e)[:200]))
|
|
5140
|
-
return _att, _fail
|
|
5141
|
-
_o_att, _o_fail = _run(_official)
|
|
5142
|
-
_a_att, _a_fail = _run(_authored)
|
|
5143
|
-
print("SRCK-${nonce} official=%d/%d authored=%d/%d" % (_o_att - _o_fail, _o_att, _a_att - _a_fail, _a_att))
|
|
5144
|
-
_sys.stdout.write("\\n".join(_lines)[-1500:])
|
|
5145
|
-
_sys.exit(0 if (_o_fail + _a_fail) == 0 and (_o_att + _a_att) > 0 else 1)
|
|
5146
|
-
`;
|
|
5147
|
-
}
|
|
5148
|
-
function sandboxCheckRunner(options = {}) {
|
|
5149
|
-
const python = options.python ?? "python3";
|
|
5150
|
-
const timeoutMs = options.timeoutMs ?? 2e4;
|
|
5151
|
-
return {
|
|
5152
|
-
async run(candidate, checks, ctx) {
|
|
5153
|
-
if (checks.length === 0) {
|
|
5154
|
-
return {
|
|
5155
|
-
passedOfficial: 0,
|
|
5156
|
-
totalOfficial: 0,
|
|
5157
|
-
passedAuthored: 0,
|
|
5158
|
-
totalAuthored: 0,
|
|
5159
|
-
failureOutput: ""
|
|
5160
|
-
};
|
|
5161
|
-
}
|
|
5162
|
-
const box = ctx.box ?? options.box;
|
|
5163
|
-
if (!box) {
|
|
5164
|
-
throw new Error(
|
|
5165
|
-
"sandboxCheckRunner: no execution channel \u2014 bind one via sandboxCheckRunner({ box }) or CheckRunContext.box (ValidationCtx.box / a sandbox instance). Refusing to score without executing: a silent 0 would poison selection."
|
|
5166
|
-
);
|
|
5167
|
-
}
|
|
5168
|
-
const nonce = randomBytes(8).toString("hex");
|
|
5169
|
-
const official = checks.filter((c) => c.kind === "official").map((c) => c.code);
|
|
5170
|
-
const authored = checks.filter((c) => c.kind === "authored").map((c) => c.code);
|
|
5171
|
-
const program = buildCheckProgram(candidate, official, authored, nonce);
|
|
5172
|
-
const b64 = Buffer.from(program, "utf8").toString("base64");
|
|
5173
|
-
const r = await box.exec(`printf '%s' '${b64}' | base64 -d | ${python} -`, { timeoutMs });
|
|
5174
|
-
const summary = new RegExp(
|
|
5175
|
-
`SRCK-${nonce} official=(\\d+)/(\\d+) authored=(\\d+)/(\\d+)`
|
|
5176
|
-
).exec(r.stdout);
|
|
5177
|
-
if (!summary) {
|
|
5178
|
-
const detail = (r.stderr || r.stdout).slice(-1500) || "no output (crashed or timed out before the checks could run)";
|
|
5179
|
-
return {
|
|
5180
|
-
passedOfficial: 0,
|
|
5181
|
-
totalOfficial: 0,
|
|
5182
|
-
passedAuthored: 0,
|
|
5183
|
-
totalAuthored: 0,
|
|
5184
|
-
failureOutput: detail,
|
|
5185
|
-
crashed: true
|
|
5186
|
-
};
|
|
5187
|
-
}
|
|
5188
|
-
const failureOutput = r.stdout.replace(summary[0], "").slice(-1500).trim();
|
|
5189
|
-
return {
|
|
5190
|
-
passedOfficial: Number(summary[1]),
|
|
5191
|
-
totalOfficial: Number(summary[2]),
|
|
5192
|
-
passedAuthored: Number(summary[3]),
|
|
5193
|
-
totalAuthored: Number(summary[4]),
|
|
5194
|
-
failureOutput
|
|
5195
|
-
};
|
|
5196
|
-
}
|
|
5197
|
-
};
|
|
5198
|
-
}
|
|
5199
|
-
var frac = (passed, total) => total > 0 ? passed / total : 0;
|
|
5200
|
-
function compareCheckOutcomes(a, b) {
|
|
5201
|
-
const aCrashed = a.crashed === true;
|
|
5202
|
-
const bCrashed = b.crashed === true;
|
|
5203
|
-
if (aCrashed !== bCrashed) return aCrashed ? -1 : 1;
|
|
5204
|
-
if (aCrashed) return 0;
|
|
5205
|
-
const official = frac(a.passedOfficial, a.totalOfficial) - frac(b.passedOfficial, b.totalOfficial);
|
|
5206
|
-
if (official !== 0) return official;
|
|
5207
|
-
return frac(a.passedAuthored, a.totalAuthored) - frac(b.passedAuthored, b.totalAuthored);
|
|
5208
|
-
}
|
|
5209
|
-
function visibleCheckScore(o) {
|
|
5210
|
-
if (o.crashed) return -1;
|
|
5211
|
-
return frac(o.passedOfficial, o.totalOfficial) + 1e-3 * frac(o.passedAuthored, o.totalAuthored);
|
|
5212
|
-
}
|
|
5213
|
-
function selectBestIndex(outcomes) {
|
|
5214
|
-
let best = 0;
|
|
5215
|
-
for (let i = 1; i < outcomes.length; i += 1) {
|
|
5216
|
-
if (compareCheckOutcomes(outcomes[i], outcomes[best]) > 0) {
|
|
5217
|
-
best = i;
|
|
5218
|
-
}
|
|
5219
|
-
}
|
|
5220
|
-
return best;
|
|
5221
|
-
}
|
|
5222
|
-
function canDisplace(challenger, incumbent) {
|
|
5223
|
-
if (challenger.crashed === true) return false;
|
|
5224
|
-
if (challenger.passedOfficial < incumbent.passedOfficial) return false;
|
|
5225
|
-
return compareCheckOutcomes(challenger, incumbent) > 0;
|
|
5226
|
-
}
|
|
5227
|
-
var totalChecks = (o) => o.totalOfficial + o.totalAuthored;
|
|
5228
|
-
var passesAllChecks = (o) => o.crashed !== true && totalChecks(o) > 0 && o.passedOfficial === o.totalOfficial && o.passedAuthored === o.totalAuthored;
|
|
5229
|
-
function defaultExtractCandidate(messages) {
|
|
5230
|
-
for (let i = messages.length - 1; i >= 0; i -= 1) {
|
|
5231
|
-
const calls = messages[i]?.tool_calls;
|
|
5232
|
-
if (!calls) continue;
|
|
5233
|
-
for (let j = calls.length - 1; j >= 0; j -= 1) {
|
|
5234
|
-
const call = calls[j];
|
|
5235
|
-
if (call?.function?.name !== "submit_answer") continue;
|
|
5236
|
-
try {
|
|
5237
|
-
const args = JSON.parse(call.function.arguments ?? "{}");
|
|
5238
|
-
if (typeof args.answer === "string" && args.answer.trim()) return args.answer.trim();
|
|
5239
|
-
} catch {
|
|
5240
|
-
}
|
|
5241
|
-
}
|
|
5242
|
-
}
|
|
5243
|
-
const contents = [];
|
|
5244
|
-
for (const m of messages) {
|
|
5245
|
-
if (m.role === "assistant" && typeof m.content === "string" && m.content.trim()) {
|
|
5246
|
-
contents.push(m.content);
|
|
5247
|
-
}
|
|
5248
|
-
}
|
|
5249
|
-
const fencesOf = (text) => [...text.matchAll(/```(?:python|py)?\s*\n([\s\S]*?)```/gi)].map((m) => (m[1] ?? "").trim());
|
|
5250
|
-
for (let i = contents.length - 1; i >= 0; i -= 1) {
|
|
5251
|
-
const fences = fencesOf(contents[i]);
|
|
5252
|
-
for (let j = fences.length - 1; j >= 0; j -= 1) {
|
|
5253
|
-
if (/(^|\n)\s*def\s+\w+/.test(fences[j])) return fences[j];
|
|
5254
|
-
}
|
|
5255
|
-
}
|
|
5256
|
-
for (let i = contents.length - 1; i >= 0; i -= 1) {
|
|
5257
|
-
const fences = fencesOf(contents[i]);
|
|
5258
|
-
if (fences.length > 0) return fences[fences.length - 1];
|
|
5259
|
-
}
|
|
5260
|
-
return (contents[contents.length - 1] ?? "").trim();
|
|
5261
|
-
}
|
|
5262
|
-
var DIVERSE_LENSES = [
|
|
5263
|
-
"Answer directly and decisively from what you already know. State the single best answer without hedging.",
|
|
5264
|
-
"Decompose the question into the sub-facts it depends on. Establish each sub-fact explicitly, then compose them into the answer.",
|
|
5265
|
-
"Reason from first principles. Ignore the most obvious or popular guess; derive the answer from underlying facts and relationships.",
|
|
5266
|
-
"Name the most plausible WRONG answer and the trap that makes it tempting. Rule it out, then commit to the answer that survives."
|
|
5267
|
-
];
|
|
5268
|
-
function slotLens(slot) {
|
|
5269
|
-
const lens = DIVERSE_LENSES[slot % DIVERSE_LENSES.length];
|
|
5270
|
-
const tag = slot < DIVERSE_LENSES.length ? "" : ` (variant ${Math.floor(slot / DIVERSE_LENSES.length) + 1})`;
|
|
5271
|
-
return `${lens}${tag}`;
|
|
5272
|
-
}
|
|
5273
|
-
function repairSteer(outcome) {
|
|
5274
|
-
return [
|
|
5275
|
-
"Your latest solution failed some of the task-visible checks.",
|
|
5276
|
-
"Result of running the visible checks against it:",
|
|
5277
|
-
"```",
|
|
5278
|
-
outcome.failureOutput.trim() || "(the code crashed before the checks could run)",
|
|
5279
|
-
"```",
|
|
5280
|
-
"Fix the solution so the visible checks pass. Provide the COMPLETE corrected solution the",
|
|
5281
|
-
"same way you provided the original (same tool or format) \u2014 not a fragment or a diff."
|
|
5282
|
-
].join("\n");
|
|
5283
|
-
}
|
|
5284
|
-
function describeOutcome(label, o) {
|
|
5285
|
-
if (o.crashed) return `${label}: crashed before the checks could run`;
|
|
5286
|
-
return `${label}: official ${o.passedOfficial}/${o.totalOfficial}, authored ${o.passedAuthored}/${o.totalAuthored}`;
|
|
5287
|
-
}
|
|
5288
|
-
function structuralRollout(config = {}) {
|
|
5289
|
-
const policy = resolvePolicy(config.policy);
|
|
5290
|
-
const checkSource = config.checkSource ?? composeCheckSources(officialChecksFromMeta(), modelAuthoredChecks());
|
|
5291
|
-
const checkRunner = config.checkRunner ?? sandboxCheckRunner();
|
|
5292
|
-
const extract = config.extractCandidate ?? defaultExtractCandidate;
|
|
5293
|
-
const inner = defineStrategy(
|
|
5294
|
-
"structuralRollout",
|
|
5295
|
-
async (ctx) => {
|
|
5296
|
-
const { task, shot } = ctx;
|
|
5297
|
-
const progression = [];
|
|
5298
|
-
const receipts = [];
|
|
5299
|
-
let completions = 0;
|
|
5300
|
-
let shots = 0;
|
|
5301
|
-
const consult = async (instruction) => {
|
|
5302
|
-
const reply = await ctx.consult([], instruction);
|
|
5303
|
-
completions += 1;
|
|
5304
|
-
return reply;
|
|
5305
|
-
};
|
|
5306
|
-
const entrySymbol = resolveEntrySymbol(task);
|
|
5307
|
-
const checks = await checkSource.generate(task, {
|
|
5308
|
-
count: policy.testgen,
|
|
5309
|
-
...entrySymbol ? { entrySymbol } : {},
|
|
5310
|
-
consult
|
|
5311
|
-
});
|
|
5312
|
-
const officialChecks = checks.filter((c) => c.kind === "official").length;
|
|
5313
|
-
const authoredChecks = checks.length - officialChecks;
|
|
5314
|
-
const runCtx = { task, ...config.box ? { box: config.box } : {} };
|
|
5315
|
-
const candidates = [];
|
|
5316
|
-
for (let i = 0; i < policy.k; i += 1) {
|
|
5317
|
-
const out = await shot(policy.diverse ? { steer: slotLens(i) } : void 0);
|
|
5318
|
-
if (!out) break;
|
|
5319
|
-
shots += 1;
|
|
5320
|
-
completions += out.completions;
|
|
5321
|
-
progression.push(out.score);
|
|
5322
|
-
const outcome = await checkRunner.run(extract(out.messages), checks, runCtx);
|
|
5323
|
-
candidates.push({
|
|
5324
|
-
index: candidates.length,
|
|
5325
|
-
messages: out.messages,
|
|
5326
|
-
outcome,
|
|
5327
|
-
shotScore: out.score,
|
|
5328
|
-
shotResolved: out.total > 0 && out.passes === out.total
|
|
5329
|
-
});
|
|
5330
|
-
}
|
|
5331
|
-
if (candidates.length === 0) {
|
|
5332
|
-
return {
|
|
5333
|
-
score: 0,
|
|
5334
|
-
resolved: false,
|
|
5335
|
-
completions,
|
|
5336
|
-
progression,
|
|
5337
|
-
shots,
|
|
5338
|
-
selection: receipts,
|
|
5339
|
-
repairStop: "no-candidates",
|
|
5340
|
-
officialChecks,
|
|
5341
|
-
authoredChecks
|
|
5342
|
-
};
|
|
5343
|
-
}
|
|
5344
|
-
let best = candidates[selectBestIndex(candidates.map((c) => c.outcome))];
|
|
5345
|
-
for (const c of candidates) {
|
|
5346
|
-
receipts.push({
|
|
5347
|
-
candidateIndex: c.index,
|
|
5348
|
-
selected: false,
|
|
5349
|
-
score: visibleCheckScore(c.outcome),
|
|
5350
|
-
reason: describeOutcome("sample", c.outcome),
|
|
5351
|
-
selector: "driver"
|
|
5352
|
-
});
|
|
5353
|
-
}
|
|
5354
|
-
let seq = candidates.length;
|
|
5355
|
-
let repairStop = "already-passing";
|
|
5356
|
-
if (!passesAllChecks(best.outcome)) {
|
|
5357
|
-
if (best.outcome.crashed !== true && totalChecks(best.outcome) === 0) {
|
|
5358
|
-
repairStop = "no-signal";
|
|
5359
|
-
} else {
|
|
5360
|
-
repairStop = "rounds-exhausted";
|
|
5361
|
-
for (let r = 0; r < policy.repairRounds; r += 1) {
|
|
5362
|
-
const out = await shot({ messages: best.messages, steer: repairSteer(best.outcome) });
|
|
5363
|
-
if (!out) break;
|
|
5364
|
-
shots += 1;
|
|
5365
|
-
completions += out.completions;
|
|
5366
|
-
progression.push(out.score);
|
|
5367
|
-
const outcome = await checkRunner.run(extract(out.messages), checks, runCtx);
|
|
5368
|
-
const displaced = canDisplace(outcome, best.outcome);
|
|
5369
|
-
const label = displaced ? "repair (displaced the incumbent)" : outcome.crashed !== true && outcome.passedOfficial < best.outcome.passedOfficial ? "repair (held out: passes fewer official checks than the incumbent)" : "repair (held out: no improvement)";
|
|
5370
|
-
receipts.push({
|
|
5371
|
-
candidateIndex: seq,
|
|
5372
|
-
selected: false,
|
|
5373
|
-
score: visibleCheckScore(outcome),
|
|
5374
|
-
reason: describeOutcome(label, outcome),
|
|
5375
|
-
selector: "driver"
|
|
5376
|
-
});
|
|
5377
|
-
if (displaced) {
|
|
5378
|
-
best = {
|
|
5379
|
-
index: seq,
|
|
5380
|
-
messages: out.messages,
|
|
5381
|
-
outcome,
|
|
5382
|
-
shotScore: out.score,
|
|
5383
|
-
shotResolved: out.total > 0 && out.passes === out.total
|
|
5384
|
-
};
|
|
5385
|
-
}
|
|
5386
|
-
seq += 1;
|
|
5387
|
-
if (passesAllChecks(best.outcome)) {
|
|
5388
|
-
repairStop = "repaired-pass";
|
|
5389
|
-
break;
|
|
5390
|
-
}
|
|
5391
|
-
}
|
|
5392
|
-
}
|
|
5393
|
-
}
|
|
5394
|
-
const winner = receipts.find((r) => r.candidateIndex === best.index);
|
|
5395
|
-
if (winner) winner.selected = true;
|
|
5396
|
-
return {
|
|
5397
|
-
score: best.shotScore,
|
|
5398
|
-
resolved: best.shotResolved,
|
|
5399
|
-
completions,
|
|
5400
|
-
progression,
|
|
5401
|
-
shots,
|
|
5402
|
-
selection: receipts,
|
|
5403
|
-
repairStop,
|
|
5404
|
-
officialChecks,
|
|
5405
|
-
authoredChecks
|
|
5406
|
-
};
|
|
5407
|
-
}
|
|
5408
|
-
);
|
|
5409
|
-
if (policy.temperature === void 0) return inner;
|
|
5410
|
-
return {
|
|
5411
|
-
name: inner.name,
|
|
5412
|
-
driver: (surface, task, opts, budget) => inner.driver(surface, task, { ...opts, temperature: policy.temperature }, budget)
|
|
5413
|
-
};
|
|
5414
|
-
}
|
|
5415
|
-
|
|
5416
4232
|
// src/runtime/supervise/detector-monitor.ts
|
|
5417
4233
|
import {
|
|
5418
4234
|
argHash,
|
|
@@ -6229,30 +5045,6 @@ export {
|
|
|
6229
5045
|
createSandboxPromptBackend,
|
|
6230
5046
|
createOpenAICompatibleBackend,
|
|
6231
5047
|
normalizeBackendStreamEvent,
|
|
6232
|
-
defaultAnalystInstruction,
|
|
6233
|
-
observe,
|
|
6234
|
-
renderReport,
|
|
6235
|
-
depthStrategy,
|
|
6236
|
-
breadthStrategy,
|
|
6237
|
-
sample,
|
|
6238
|
-
refine,
|
|
6239
|
-
defineStrategy,
|
|
6240
|
-
adaptiveRefine,
|
|
6241
|
-
sampleThenRefine,
|
|
6242
|
-
runAgentic,
|
|
6243
|
-
defaultStructuralRolloutPolicy,
|
|
6244
|
-
filterAuthoredAsserts,
|
|
6245
|
-
modelAuthoredChecks,
|
|
6246
|
-
officialChecksFromMeta,
|
|
6247
|
-
composeCheckSources,
|
|
6248
|
-
resolveEntrySymbol,
|
|
6249
|
-
sandboxCheckRunner,
|
|
6250
|
-
compareCheckOutcomes,
|
|
6251
|
-
visibleCheckScore,
|
|
6252
|
-
selectBestIndex,
|
|
6253
|
-
canDisplace,
|
|
6254
|
-
defaultExtractCandidate,
|
|
6255
|
-
structuralRollout,
|
|
6256
5048
|
anytimeReport,
|
|
6257
5049
|
renderAnytimeTable,
|
|
6258
5050
|
defaultAuditorInstruction,
|
|
@@ -6332,6 +5124,6 @@ export {
|
|
|
6332
5124
|
jjWorkspace,
|
|
6333
5125
|
runInWorkspace,
|
|
6334
5126
|
computeFindingId,
|
|
6335
|
-
|
|
5127
|
+
makeFinding
|
|
6336
5128
|
};
|
|
6337
|
-
//# sourceMappingURL=chunk-
|
|
5129
|
+
//# sourceMappingURL=chunk-I7WVPJBZ.js.map
|