@tangle-network/agent-runtime 0.126.0 → 0.131.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +70 -20
- package/dist/{activation-DhWJ3p8N.js → activation-BCzMOTaV.js} +3 -3
- package/dist/{activation-DhWJ3p8N.js.map → activation-BCzMOTaV.js.map} +1 -1
- package/dist/agent.d.ts +2 -3
- package/dist/agent.js +4 -5
- package/dist/agent.js.map +1 -1
- package/dist/{analyst-loop-DvSciOfB.js → analyst-loop-BE8cDs5Q.js} +2 -2
- package/dist/{analyst-loop-DvSciOfB.js.map → analyst-loop-BE8cDs5Q.js.map} +1 -1
- package/dist/analyst-loop.d.ts +1 -1
- package/dist/analyst-loop.js +1 -1
- package/dist/authoring-CvHwo1oW.js +163 -0
- package/dist/authoring-CvHwo1oW.js.map +1 -0
- package/dist/candidate-execution/index.js +4 -4
- package/dist/{candidate-execution-BFpq-Xi6.js → candidate-execution-BNxKr-Bu.js} +5 -5
- package/dist/{candidate-execution-BFpq-Xi6.js.map → candidate-execution-BNxKr-Bu.js.map} +1 -1
- package/dist/{conversation-BpLQZGPH.js → conversation-DNtxaJ1Z.js} +113 -31
- package/dist/conversation-DNtxaJ1Z.js.map +1 -0
- package/dist/conversation.d.ts +2 -2
- package/dist/conversation.js +2 -2
- package/dist/environment-provider-CxvSd1W6.d.ts +86 -0
- package/dist/{environment-provider-DqFS6FSZ.js → environment-provider-Dyg8DtLK.js} +8 -4
- package/dist/environment-provider-Dyg8DtLK.js.map +1 -0
- package/dist/environment-provider.d.ts +1 -1
- package/dist/environment-provider.js +1 -1
- package/dist/graph-BJTxGOFB.js +471 -0
- package/dist/graph-BJTxGOFB.js.map +1 -0
- package/dist/{improvement-cycle-IJgCbKWQ.js → improvement-cycle-Csp38cWg.js} +133 -224
- package/dist/improvement-cycle-Csp38cWg.js.map +1 -0
- package/dist/{index-EdjCQBV9.d.ts → index-CoO7atyo.d.ts} +640 -1184
- package/dist/{index-DIV33AF5.d.ts → index-DwGtu9nc.d.ts} +7 -9
- package/dist/{index-Efjb3nrQ.d.ts → index-qYHpsmG2.d.ts} +36 -18
- package/dist/index.d.ts +353 -11
- package/dist/index.js +111 -354
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +6 -6
- package/dist/intelligence.js +9 -8
- package/dist/intelligence.js.map +1 -1
- package/dist/kernel.d.ts +7 -5
- package/dist/kernel.js +13 -9
- package/dist/{knowledge-EnuEqm_Y.js → knowledge-ce0_uKCl.js} +19 -17
- package/dist/knowledge-ce0_uKCl.js.map +1 -0
- package/dist/knowledge.d.ts +1 -1
- package/dist/knowledge.js +1 -1
- package/dist/{loop-runner-bin-Bo29_fiD.d.ts → loop-runner-bin-BwgZ8m_7.d.ts} +6 -3
- package/dist/{loop-runner-bin-qwT_4F5I.js → loop-runner-bin-DSbuDDqM.js} +5 -27
- package/dist/loop-runner-bin-DSbuDDqM.js.map +1 -0
- package/dist/loop-runner-bin.d.ts +1 -1
- package/dist/loop-runner-bin.js +1 -1
- package/dist/materialization-COJ1UYQ-.js +272 -0
- package/dist/materialization-COJ1UYQ-.js.map +1 -0
- package/dist/mcp/bin.js +39 -47
- package/dist/mcp/bin.js.map +1 -1
- package/dist/mcp/index.d.ts +24 -30
- package/dist/mcp/index.js +66 -83
- package/dist/mcp/index.js.map +1 -1
- package/dist/mcp/memory-bin.js +1 -1
- package/dist/{memory-server-DL6cE2Ag.js → memory-server-5HEJH672.js} +2 -2
- package/dist/{memory-server-DL6cE2Ag.js.map → memory-server-5HEJH672.js.map} +1 -1
- package/dist/model-policy-CqziaqS1.js +232 -0
- package/dist/model-policy-CqziaqS1.js.map +1 -0
- package/dist/{openai-tools-B68JaOCx.d.ts → openai-tools-D3oMyVGl.d.ts} +2 -2
- package/dist/{openai-tools-Bp1KSkP6.js → openai-tools-ru75mLjq.js} +2 -2
- package/dist/openai-tools-ru75mLjq.js.map +1 -0
- package/dist/{prepare--8EvLqCr.js → prepare-DYWjVcPx.js} +169 -153
- package/dist/prepare-DYWjVcPx.js.map +1 -0
- package/dist/primeintellect/index.d.ts +7 -6
- package/dist/primeintellect/index.js +9 -11
- package/dist/primeintellect/index.js.map +1 -1
- package/dist/profiles.d.ts +21 -174
- package/dist/profiles.js +67 -276
- package/dist/profiles.js.map +1 -1
- package/dist/{protected-model-port-CXVfOUu_.js → protected-model-port-48ALLGxT.js} +2 -2
- package/dist/{protected-model-port-CXVfOUu_.js.map → protected-model-port-48ALLGxT.js.map} +1 -1
- package/dist/{redact-PQzmE1Jn.d.ts → redact-9_Gf-8m3.d.ts} +58 -69
- package/dist/{researcher-CoVqNhfI.js → researcher-Skz5-Uc8.js} +50 -26
- package/dist/researcher-Skz5-Uc8.js.map +1 -0
- package/dist/{run-layout-C2FGmZ3v.js → run-layout-CeJsEAom.js} +2 -2
- package/dist/{run-layout-C2FGmZ3v.js.map → run-layout-CeJsEAom.js.map} +1 -1
- package/dist/runtime-D-QfLbSd.d.ts +893 -0
- package/dist/{runtime-BzXz7OjS.js → runtime-hiAABiTk.js} +329 -854
- package/dist/runtime-hiAABiTk.js.map +1 -0
- package/dist/{sandbox-events-Yhd1GYWl.js → sandbox-events-CRDwc5WN.js} +42 -5
- package/dist/sandbox-events-CRDwc5WN.js.map +1 -0
- package/dist/snapshot-CXiiuHhL.js +21 -0
- package/dist/snapshot-CXiiuHhL.js.map +1 -0
- package/dist/{spawn-journal-DsZKDqeh.js → spawn-journal-saHQzqYi.js} +15 -21
- package/dist/spawn-journal-saHQzqYi.js.map +1 -0
- package/dist/stream-agent-turn-C852AgMT.d.ts +128 -0
- package/dist/stream-agent-turn-rYgaOLO0.js +910 -0
- package/dist/stream-agent-turn-rYgaOLO0.js.map +1 -0
- package/dist/{structural-rollout-zY0oqQzO.js → structural-rollout-3uxVGcg2.js} +459 -235
- package/dist/structural-rollout-3uxVGcg2.js.map +1 -0
- package/dist/{supervise-Ds8FtyI9.js → supervise-iPN27pO0.js} +932 -4771
- package/dist/supervise-iPN27pO0.js.map +1 -0
- package/dist/{supervisor-DpjO0Gmy.js → supervisor-CV6Jh28D.js} +7439 -2897
- package/dist/supervisor-CV6Jh28D.js.map +1 -0
- package/dist/testing.d.ts +3 -1
- package/dist/testing.js +271 -221
- package/dist/testing.js.map +1 -1
- package/dist/{top-app-5unxqovu.js → top-app-G4M18b6i.js} +3 -3
- package/dist/{top-app-5unxqovu.js.map → top-app-G4M18b6i.js.map} +1 -1
- package/dist/tui/bin.js +1 -1
- package/dist/tui/index.js +1 -1
- package/dist/{environment-provider-PM9PeW_J.d.ts → types-C6Q-J0Dt.d.ts} +57 -114
- package/dist/{types-DnNGJ5Gz.d.ts → types-ebIY0dMG.d.ts} +556 -30
- package/dist/{util-MVgdwuIS.js → util-Bw6srryQ.js} +3 -2
- package/dist/{util-MVgdwuIS.js.map → util-Bw6srryQ.js.map} +1 -1
- package/dist/{workspace-archive-BQxvkypI.js → workspace-archive-aOfJ47ms.js} +3 -3
- package/dist/{workspace-archive-BQxvkypI.js.map → workspace-archive-aOfJ47ms.js.map} +1 -1
- package/package.json +12 -15
- package/skills/agent-graphs/IMPROVE.md +58 -0
- package/skills/agent-graphs/SKILL.md +139 -0
- package/skills/agent-graphs/cases/artifact-mission-release-notes.json +10 -0
- package/skills/agent-graphs/cases/audited-single-writer.json +9 -0
- package/skills/agent-graphs/cases/cap-as-stop-mistake.json +8 -0
- package/skills/agent-graphs/cases/mission-in-deliverable.json +8 -0
- package/skills/agent-graphs/cases/review-pipeline.json +13 -0
- package/skills/agent-graphs/cases/runtime-discovered-fanout.json +8 -0
- package/skills/agent-graphs/cases/single-agent-suffices.json +7 -0
- package/skills/agent-graphs/cases/steer-heavy-drafting.json +9 -0
- package/skills/agent-graphs/cases/unmeasured-harness.json +7 -0
- package/skills/agent-graphs/generations/gen1-baseline.json +248 -0
- package/skills/agent-graphs/generations/gen2.json +375 -0
- package/skills/agent-graphs/generations/gen3.json +702 -0
- package/skills/build-with-agent-runtime/SKILL.md +1 -0
- package/dist/backends-CiOCyRHb.js +0 -743
- package/dist/backends-CiOCyRHb.js.map +0 -1
- package/dist/conversation-BpLQZGPH.js.map +0 -1
- package/dist/environment-provider-DqFS6FSZ.js.map +0 -1
- package/dist/improvement-cycle-IJgCbKWQ.js.map +0 -1
- package/dist/index-D_M4d1_B.d.ts +0 -545
- package/dist/knowledge-EnuEqm_Y.js.map +0 -1
- package/dist/local-harness-BIajef4A.d.ts +0 -465
- package/dist/loop-runner-bin-qwT_4F5I.js.map +0 -1
- package/dist/model-resolution-Btd9iIKV.js +0 -98
- package/dist/model-resolution-Btd9iIKV.js.map +0 -1
- package/dist/openai-tools-Bp1KSkP6.js.map +0 -1
- package/dist/prepare--8EvLqCr.js.map +0 -1
- package/dist/researcher-CoVqNhfI.js.map +0 -1
- package/dist/runtime-BzXz7OjS.js.map +0 -1
- package/dist/sandbox-events-Yhd1GYWl.js.map +0 -1
- package/dist/spawn-journal-DsZKDqeh.js.map +0 -1
- package/dist/structural-rollout-zY0oqQzO.js.map +0 -1
- package/dist/supervise-Ds8FtyI9.js.map +0 -1
- package/dist/supervisor-DpjO0Gmy.js.map +0 -1
- package/dist/types-C9j4qg6l.d.ts +0 -500
|
@@ -1,19 +1,24 @@
|
|
|
1
|
-
import { n as AnalystError,
|
|
2
|
-
import
|
|
3
|
-
import { i as InMemorySpawnJournal, r as InMemoryResultBlobStore, v as isTraceAnalysisStore } from "./spawn-journal-
|
|
4
|
-
import { a as randomSuffix, c as stringifySafe, f as zeroTokenUsage, r as isAbortError, s as sleep, t as addTokenUsage } from "./util-
|
|
1
|
+
import { n as AnalystError, s as PlannerError, u as ValidationError } from "./errors-DEAvWQPy.js";
|
|
2
|
+
import "./stream-agent-turn-rYgaOLO0.js";
|
|
3
|
+
import { i as InMemorySpawnJournal, r as InMemoryResultBlobStore, v as isTraceAnalysisStore, x as contentAddress } from "./spawn-journal-saHQzqYi.js";
|
|
4
|
+
import { a as randomSuffix, c as stringifySafe, f as zeroTokenUsage, r as isAbortError, s as sleep, t as addTokenUsage } from "./util-Bw6srryQ.js";
|
|
5
5
|
import { i as redactProtectedValue, r as redactProtectedReason } from "./protected-redaction--F3v1oo8.js";
|
|
6
|
-
import {
|
|
7
|
-
import {
|
|
8
|
-
import {
|
|
9
|
-
import {
|
|
10
|
-
import
|
|
11
|
-
import {
|
|
6
|
+
import { a as notifySandboxEventObserver } from "./sandbox-events-CRDwc5WN.js";
|
|
7
|
+
import { c as profileProviderModel, i as concreteModelId, s as profileModelExecutionSettings, t as assertExecutableAgentProfile } from "./model-policy-CqziaqS1.js";
|
|
8
|
+
import { v as executableAgentProfileSnapshot, y as executableAgentSpecSnapshot } from "./materialization-COJ1UYQ-.js";
|
|
9
|
+
import { $ as createSandboxLineage, I as createWorktreeCliExecutor, N as createExecutor, P as createExecutorRegistry, Q as runAgentRounds, U as createPushTraceSource, Z as defaultSelectWinner, et as probeSandboxCapabilities, l as withDriverExecutor, m as settledToIteration, n as createSupervisor, nt as routerBrain, rt as runBrainLoop, w as rollingDispatch } from "./supervisor-CV6Jh28D.js";
|
|
10
|
+
import "./environment-provider-Dyg8DtLK.js";
|
|
11
|
+
import { i as notifyRuntimeHookEvent } from "./runtime-hooks-C7iJOWm3.js";
|
|
12
|
+
import { C as observe, O as strategyAuthorMethod, T as profileChatClient, b as sample, v as refine, x as sampleThenRefine, y as runAgentic } from "./structural-rollout-3uxVGcg2.js";
|
|
13
|
+
import { It as gateOnDeliverable, Lt as mapExecutorResult, n as supervise } from "./supervise-iPN27pO0.js";
|
|
14
|
+
import "./authoring-CvHwo1oW.js";
|
|
15
|
+
import "./graph-BJTxGOFB.js";
|
|
16
|
+
import { CODING_HARNESSES, InMemoryTraceStore, OUTPUT_VALUE, benjaminiHochberg, buildTrajectory, computeFindingId as computeFindingId$1, confidenceInterval, expandProfileAxes, makeFinding as makeFinding$1, pairedBootstrap, paretoFrontier, wilcoxonSignedRank, wilson } from "@tangle-network/agent-eval";
|
|
12
17
|
import { heldoutSignificance, runProfileMatrix } from "@tangle-network/agent-eval/campaign";
|
|
13
|
-
import { agentProfileSchema,
|
|
18
|
+
import { agentProfileSchema, canonicalAgentProfileDigest, validateAgentProfileSecurity } from "@tangle-network/agent-interface";
|
|
19
|
+
import { dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
|
|
14
20
|
import { mkdir, readFile, writeFile } from "node:fs/promises";
|
|
15
21
|
import { appendFileSync, chmodSync, constants, copyFileSync, existsSync, lstatSync, mkdirSync, mkdtempSync, readFileSync, readlinkSync, rmSync, symlinkSync, writeFileSync } from "node:fs";
|
|
16
|
-
import { dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
|
|
17
22
|
import { execFileSync, spawn, spawnSync } from "node:child_process";
|
|
18
23
|
import { tmpdir } from "node:os";
|
|
19
24
|
import { createInterface } from "node:readline";
|
|
@@ -530,13 +535,13 @@ const auditSchema = {
|
|
|
530
535
|
};
|
|
531
536
|
/** The route-rigor analyst: compare declared vs revealed vs user intent over a trajectory and return aligned / drifting / diverged with evidence and one recommended intervention. */
|
|
532
537
|
async function auditIntent(input, opts) {
|
|
533
|
-
const res = await
|
|
534
|
-
|
|
538
|
+
const res = await profileChatClient({
|
|
539
|
+
profile: opts.profile,
|
|
540
|
+
executor: opts.executor,
|
|
541
|
+
context: "intent auditor"
|
|
542
|
+
}).chat({
|
|
535
543
|
jsonSchema: auditSchema,
|
|
536
544
|
messages: [{
|
|
537
|
-
role: "system",
|
|
538
|
-
content: opts.auditorInstruction ?? "You audit whether an AI agent is on the RIGHT ROUTE — not whether it works hard, but whether its actions serve the stated intents. Infer the REVEALED intent from the action pattern (what the trajectory is actually optimizing). Compare against the declared task intent, the user intent when given, and the meta-intent when given. Flawless execution down the wrong route is DIVERGED. Busy-work that neither advances nor harms is DRIFTING. Judge only from the trajectory — be specific about which actions ground your verdict. Recommend abort only when continuing cannot serve the intent."
|
|
539
|
-
}, {
|
|
540
545
|
role: "user",
|
|
541
546
|
content: `DECLARED INTENT (the task):\n${input.declaredIntent}\n\n` + (input.userIntent ? `USER INTENT (the principal's actual goal):\n${input.userIntent}\n\n` : "") + (input.metaIntent ? `META-INTENT (what the whole run is for):\n${input.metaIntent}\n\n` : "") + `TRAJECTORY (in order):\n${summarize(input.trace, opts.maxTraceLines ?? 80)}\n\nAudit the route: revealed intent, verdict, evidence, one recommendation.`
|
|
542
547
|
}]
|
|
@@ -1035,7 +1040,10 @@ function loopCostReceipt(result, model) {
|
|
|
1035
1040
|
model,
|
|
1036
1041
|
inputTokens: result.tokenUsage.input,
|
|
1037
1042
|
outputTokens: result.tokenUsage.output,
|
|
1038
|
-
...result.
|
|
1043
|
+
...result.tokenUsage.tokensKnown === false ? { usageUnknown: true } : {},
|
|
1044
|
+
...result.costUsdKnown !== false ? { actualCostUsd: result.costUsd } : {},
|
|
1045
|
+
...result.costUsdKnown === false ? { costUnknown: true } : {},
|
|
1046
|
+
...result.estimatedCostUsd !== void 0 ? { estimatedCostUsd: result.estimatedCostUsd } : {}
|
|
1039
1047
|
};
|
|
1040
1048
|
}
|
|
1041
1049
|
function modelFromLoopOptions(options) {
|
|
@@ -1070,6 +1078,20 @@ function loopDispatch(opts) {
|
|
|
1070
1078
|
}
|
|
1071
1079
|
//#endregion
|
|
1072
1080
|
//#region src/runtime/inline-sandbox-client.ts
|
|
1081
|
+
/**
|
|
1082
|
+
* The ONE pseudo-box adapter: present any one-shot `Executor` (router / bridge /
|
|
1083
|
+
* BYO) as a `SandboxClient` so the round-synchronous `runAgentRounds` can drive it
|
|
1084
|
+
* without each call site re-faking a box. This is the single shell that
|
|
1085
|
+
* `bench/src/router-executor.ts`, generate-eval's old `bridgeSandboxClient`, and
|
|
1086
|
+
* the search-bench bridge transport were each re-implementing.
|
|
1087
|
+
*
|
|
1088
|
+
* It is deliberately for NON-box executors only — a real sandbox harness already
|
|
1089
|
+
* IS a `SandboxClient` (boxes, sessions, fs, fork are real there). Here each
|
|
1090
|
+
* `streamPrompt` runs the executor once and emits the terminal
|
|
1091
|
+
* `{type:'result', data:{finalText, tokenUsage, costUsd}}` event that
|
|
1092
|
+
* `answerOutput`/the kernel's cost ledger already parse — no sessions, no fs,
|
|
1093
|
+
* no fork (those degrade gracefully via the optional `SandboxClient` methods).
|
|
1094
|
+
*/
|
|
1073
1095
|
function isAsyncIterable$1(v) {
|
|
1074
1096
|
return typeof v === "object" && v !== null && Symbol.asyncIterator in v;
|
|
1075
1097
|
}
|
|
@@ -1087,7 +1109,8 @@ async function settle(exec, task, signal) {
|
|
|
1087
1109
|
* instantiated fresh per `streamPrompt` (mirrors the per-spawn executor lifecycle):
|
|
1088
1110
|
* run once on the prompt, emit the terminal result event, tear down.
|
|
1089
1111
|
*/
|
|
1090
|
-
function inlineSandboxClient(factory) {
|
|
1112
|
+
function inlineSandboxClient(factory, defaults = {}) {
|
|
1113
|
+
const capturedDefaultProfile = defaults.profile === void 0 ? void 0 : agentProfileSchema.parse(structuredClone(defaults.profile));
|
|
1091
1114
|
let seq = 0;
|
|
1092
1115
|
return { async create(options) {
|
|
1093
1116
|
const id = `inline-${seq++}`;
|
|
@@ -1100,8 +1123,11 @@ function inlineSandboxClient(factory) {
|
|
|
1100
1123
|
const onAbort = () => controller.abort(callerSignal?.reason ?? /* @__PURE__ */ new Error("prompt aborted"));
|
|
1101
1124
|
if (callerSignal) if (callerSignal.aborted) onAbort();
|
|
1102
1125
|
else callerSignal.addEventListener("abort", onAbort, { once: true });
|
|
1126
|
+
const requestedProfile = (options?.backend && typeof options.backend === "object" ? options.backend.profile : void 0) ?? capturedDefaultProfile;
|
|
1127
|
+
const parsedProfile = agentProfileSchema.safeParse(requestedProfile);
|
|
1128
|
+
if (!parsedProfile.success) throw new Error("inlineSandboxClient: an exact AgentProfile is required; pass defaults.profile or create({ backend: { profile } })");
|
|
1103
1129
|
const exec = factory({
|
|
1104
|
-
profile:
|
|
1130
|
+
profile: parsedProfile.data,
|
|
1105
1131
|
harness: null
|
|
1106
1132
|
}, {
|
|
1107
1133
|
signal: controller.signal,
|
|
@@ -1113,23 +1139,43 @@ function inlineSandboxClient(factory) {
|
|
|
1113
1139
|
const tokensIn = artifact.spent.tokens.input;
|
|
1114
1140
|
const tokensOut = artifact.spent.tokens.output;
|
|
1115
1141
|
const costUsd = artifact.spent.usd;
|
|
1116
|
-
|
|
1142
|
+
const estimatedCostUsd = out?.estimatedCostUsd;
|
|
1143
|
+
if (artifact.spent.iterations > 0 || artifact.spent.tokensKnown === false || artifact.spent.usdKnown === false || tokensIn > 0 || tokensOut > 0 || costUsd > 0 || estimatedCostUsd !== void 0) yield {
|
|
1117
1144
|
type: "llm_call",
|
|
1118
1145
|
data: {
|
|
1119
|
-
|
|
1120
|
-
|
|
1121
|
-
|
|
1146
|
+
...artifact.spent.tokensKnown === false ? {} : {
|
|
1147
|
+
tokensIn,
|
|
1148
|
+
tokensOut
|
|
1149
|
+
},
|
|
1150
|
+
...artifact.spent.usdKnown !== false ? { costUsd } : {},
|
|
1151
|
+
...artifact.spent.tokensKnown === false ? { tokensKnown: false } : {},
|
|
1152
|
+
...artifact.spent.usdKnown === false ? { costKnown: false } : {},
|
|
1153
|
+
...artifact.spent.usdKnown !== false ? {
|
|
1154
|
+
costKnown: true,
|
|
1155
|
+
costProvenance: "provider-receipt"
|
|
1156
|
+
} : {},
|
|
1157
|
+
...estimatedCostUsd !== void 0 ? { estimatedCostUsd } : {},
|
|
1158
|
+
...out?.promptCache ? { promptCache: out.promptCache } : {}
|
|
1122
1159
|
}
|
|
1123
1160
|
};
|
|
1124
1161
|
yield {
|
|
1125
1162
|
type: "result",
|
|
1126
1163
|
data: {
|
|
1127
1164
|
finalText: out?.content ?? "",
|
|
1128
|
-
tokenUsage: {
|
|
1165
|
+
...artifact.spent.tokensKnown === false ? { tokensKnown: false } : { tokenUsage: {
|
|
1129
1166
|
inputTokens: tokensIn,
|
|
1130
1167
|
outputTokens: tokensOut
|
|
1168
|
+
} },
|
|
1169
|
+
...artifact.spent.usdKnown === false ? {
|
|
1170
|
+
costKnown: false,
|
|
1171
|
+
...costUsd > 0 ? { costUsd } : {}
|
|
1172
|
+
} : {
|
|
1173
|
+
costUsd,
|
|
1174
|
+
costKnown: true,
|
|
1175
|
+
costProvenance: "provider-receipt"
|
|
1131
1176
|
},
|
|
1132
|
-
|
|
1177
|
+
...estimatedCostUsd !== void 0 ? { estimatedCostUsd } : {},
|
|
1178
|
+
...out?.promptCache ? { promptCache: out.promptCache } : {}
|
|
1133
1179
|
}
|
|
1134
1180
|
};
|
|
1135
1181
|
} finally {
|
|
@@ -1158,36 +1204,52 @@ function inlineSandboxClient(factory) {
|
|
|
1158
1204
|
* local MCP process applies only when its full canonical bytes match the fixed
|
|
1159
1205
|
* constructor profile; a different generated profile is refused.
|
|
1160
1206
|
*
|
|
1161
|
-
* Event protocol matches `inlineSandboxClient`: one `llm_call
|
|
1162
|
-
*
|
|
1207
|
+
* Event protocol matches `inlineSandboxClient`: known token usage is emitted as one `llm_call`;
|
|
1208
|
+
* Router catalog cost remains a separately-labelled estimate, never billed spend.
|
|
1163
1209
|
*/
|
|
1164
1210
|
/** A same-host `SandboxClient` adapter with no process isolation. Local MCP is
|
|
1165
1211
|
* refused unless the caller explicitly supplies a policy that allows it. */
|
|
1166
1212
|
function localSandboxClient(opts) {
|
|
1167
|
-
|
|
1168
|
-
const
|
|
1169
|
-
|
|
1213
|
+
const defaultProfile = opts.profile === void 0 ? void 0 : executableAgentProfileSnapshot(opts.profile, "localSandboxClient default profile");
|
|
1214
|
+
const router = Object.freeze({ ...opts.router });
|
|
1215
|
+
if (opts.profileSecurityPolicy?.allowLocalMcp && defaultProfile === void 0) throw new ValidationError("localSandboxClient: allowLocalMcp requires a fixed author-controlled profile; dynamic profiles need a real sandbox");
|
|
1216
|
+
const trustedProfileDigest = opts.profileSecurityPolicy?.allowLocalMcp && defaultProfile !== void 0 ? canonicalAgentProfileDigest(defaultProfile) : void 0;
|
|
1170
1217
|
let seq = 0;
|
|
1171
1218
|
return { async create(options) {
|
|
1172
|
-
const profile = (options?.backend)?.profile ??
|
|
1173
|
-
const
|
|
1219
|
+
const profile = executableAgentProfileSnapshot((options?.backend)?.profile ?? defaultProfile, "localSandboxClient");
|
|
1220
|
+
const model = profileProviderModel(profile);
|
|
1221
|
+
const settings = profileModelExecutionSettings(profile, "localSandboxClient");
|
|
1222
|
+
const policyApplies = opts.profileSecurityPolicy !== void 0 && (!opts.profileSecurityPolicy.allowLocalMcp || trustedProfileDigest !== void 0 && canonicalAgentProfileDigest(profile) === trustedProfileDigest);
|
|
1174
1223
|
const mcp = await materializeLocalMcp(profile, {
|
|
1175
1224
|
...opts.keys ? { keys: opts.keys } : {},
|
|
1176
1225
|
...policyApplies ? { profileSecurityPolicy: opts.profileSecurityPolicy } : {}
|
|
1177
1226
|
});
|
|
1178
1227
|
const brain = routerBrain({
|
|
1179
|
-
routerBaseUrl:
|
|
1180
|
-
routerKey:
|
|
1181
|
-
model
|
|
1182
|
-
|
|
1228
|
+
routerBaseUrl: router.baseUrl,
|
|
1229
|
+
routerKey: router.key,
|
|
1230
|
+
model,
|
|
1231
|
+
...settings.retry !== void 0 ? { retry: settings.retry } : {},
|
|
1232
|
+
...settings.maxTokens !== void 0 ? { maxTokens: settings.maxTokens } : {},
|
|
1233
|
+
...settings.stream !== void 0 ? { stream: settings.stream } : {}
|
|
1234
|
+
}, {
|
|
1235
|
+
...settings.temperature !== void 0 ? { temperature: settings.temperature } : {},
|
|
1236
|
+
...settings.seed !== void 0 ? { seed: settings.seed } : {},
|
|
1237
|
+
...settings.toolChoice !== void 0 ? { toolChoice: settings.toolChoice } : {},
|
|
1238
|
+
...settings.extraBody !== void 0 ? { extraBody: settings.extraBody } : {},
|
|
1239
|
+
...profile.model?.reasoningEffort ? { reasoningEffort: profile.model.reasoningEffort } : {}
|
|
1240
|
+
});
|
|
1183
1241
|
const system = [profile.prompt?.systemPrompt, ...profile.prompt?.instructions ?? []].filter((s) => typeof s === "string" && s.trim().length > 0).join("\n\n");
|
|
1184
1242
|
return {
|
|
1185
1243
|
id: `local-${seq++}`,
|
|
1186
1244
|
async *streamPrompt(message, popts) {
|
|
1187
|
-
let
|
|
1245
|
+
let estimatedCostUsd = 0;
|
|
1246
|
+
let sawEstimatedCost = false;
|
|
1188
1247
|
const chat = async (messages, tools) => {
|
|
1189
1248
|
const r = await brain(messages, tools);
|
|
1190
|
-
if (r.
|
|
1249
|
+
if (r.costProvenance === "catalog-estimate" && r.costUsd !== void 0) {
|
|
1250
|
+
estimatedCostUsd += r.costUsd;
|
|
1251
|
+
sawEstimatedCost = true;
|
|
1252
|
+
}
|
|
1191
1253
|
return r;
|
|
1192
1254
|
};
|
|
1193
1255
|
const r = await runBrainLoop({
|
|
@@ -1201,26 +1263,31 @@ function localSandboxClient(opts) {
|
|
|
1201
1263
|
role: "user",
|
|
1202
1264
|
content: message
|
|
1203
1265
|
}],
|
|
1204
|
-
maxTurns,
|
|
1266
|
+
maxTurns: settings.maxTurns ?? 0,
|
|
1205
1267
|
hooks: { stopBefore: () => popts?.signal?.aborted === true }
|
|
1206
1268
|
});
|
|
1207
|
-
if (r.
|
|
1269
|
+
if (r.turns > 0) yield {
|
|
1208
1270
|
type: "llm_call",
|
|
1209
1271
|
data: {
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
1272
|
+
model,
|
|
1273
|
+
...r.tokensKnown === false ? { tokensKnown: false } : {
|
|
1274
|
+
tokensIn: r.usage.input,
|
|
1275
|
+
tokensOut: r.usage.output
|
|
1276
|
+
},
|
|
1277
|
+
costKnown: false,
|
|
1278
|
+
...sawEstimatedCost ? { estimatedCostUsd } : {}
|
|
1213
1279
|
}
|
|
1214
1280
|
};
|
|
1215
1281
|
yield {
|
|
1216
1282
|
type: "result",
|
|
1217
1283
|
data: {
|
|
1218
1284
|
finalText: r.final,
|
|
1219
|
-
tokenUsage: {
|
|
1285
|
+
...r.tokensKnown === false ? { tokensKnown: false } : { tokenUsage: {
|
|
1220
1286
|
inputTokens: r.usage.input,
|
|
1221
1287
|
outputTokens: r.usage.output
|
|
1222
|
-
},
|
|
1223
|
-
|
|
1288
|
+
} },
|
|
1289
|
+
costKnown: false,
|
|
1290
|
+
...sawEstimatedCost ? { estimatedCostUsd } : {}
|
|
1224
1291
|
}
|
|
1225
1292
|
};
|
|
1226
1293
|
},
|
|
@@ -1268,28 +1335,26 @@ function resolveSandboxClient(opts) {
|
|
|
1268
1335
|
return opts.sandboxClient;
|
|
1269
1336
|
case "bridge": {
|
|
1270
1337
|
const bridge = opts.bridge;
|
|
1271
|
-
if (!bridge?.bearer
|
|
1338
|
+
if (!bridge?.bearer) throw new Error("resolveSandboxClient: backend 'bridge' requires bridge.bearer");
|
|
1272
1339
|
return inlineSandboxClient(createExecutor({
|
|
1273
1340
|
backend: "bridge",
|
|
1274
1341
|
bridgeUrl: bridge.url ?? "http://127.0.0.1:3355",
|
|
1275
1342
|
bridgeBearer: bridge.bearer,
|
|
1276
|
-
model: bridge.model,
|
|
1277
1343
|
timeoutMs: bridge.timeoutMs
|
|
1278
1344
|
}));
|
|
1279
1345
|
}
|
|
1280
1346
|
case "router": {
|
|
1281
1347
|
const router = opts.router;
|
|
1282
|
-
if (!router?.baseUrl || !router.key
|
|
1348
|
+
if (!router?.baseUrl || !router.key) throw new Error("resolveSandboxClient: backend 'router' requires router.baseUrl and router.key");
|
|
1283
1349
|
return inlineSandboxClient(createExecutor({
|
|
1284
1350
|
backend: "router",
|
|
1285
1351
|
routerBaseUrl: router.baseUrl,
|
|
1286
|
-
routerKey: router.key
|
|
1287
|
-
model: router.model
|
|
1352
|
+
routerKey: router.key
|
|
1288
1353
|
}));
|
|
1289
1354
|
}
|
|
1290
1355
|
case "local": {
|
|
1291
1356
|
const local = opts.local;
|
|
1292
|
-
if (!local?.router?.baseUrl || !local.router.key
|
|
1357
|
+
if (!local?.router?.baseUrl || !local.router.key) throw new Error("resolveSandboxClient: backend 'local' requires local.router.baseUrl and local.router.key");
|
|
1293
1358
|
return localSandboxClient(local);
|
|
1294
1359
|
}
|
|
1295
1360
|
}
|
|
@@ -1313,7 +1378,7 @@ function resolveSandboxClient(opts) {
|
|
|
1313
1378
|
*
|
|
1314
1379
|
* - LEVEL 0 (declarative): `cases` / `prompt` / `score` / `axis`.
|
|
1315
1380
|
* - LEVEL 1 (seams): `backends`, `flags`, `parseOutput`, `onCellEvents`,
|
|
1316
|
-
* `resolveModel`, `setup`/`teardown`, `export`, `
|
|
1381
|
+
* `resolveModel`, `setup`/`teardown`, `export`, `matrix`
|
|
1317
1382
|
* passthrough.
|
|
1318
1383
|
* - LEVEL 2 (replacement): `dispatch` and `judges` swap out the whole
|
|
1319
1384
|
* loop wiring or scoring; `runProfileMatrix` itself stays public as the
|
|
@@ -1372,10 +1437,6 @@ function splitList(v) {
|
|
|
1372
1437
|
function withSnapshot(model, snapshot) {
|
|
1373
1438
|
return model.includes("@") ? model : `${model}@${snapshot}`;
|
|
1374
1439
|
}
|
|
1375
|
-
/** The bare model id the backend actually serves (identity snapshot stripped). */
|
|
1376
|
-
function bareModel(model) {
|
|
1377
|
-
return model.split("@")[0] ?? model;
|
|
1378
|
-
}
|
|
1379
1440
|
function gitSha() {
|
|
1380
1441
|
try {
|
|
1381
1442
|
return execFileSync("git", ["rev-parse", "HEAD"], { encoding: "utf8" }).trim();
|
|
@@ -1457,13 +1518,11 @@ function defineLeaderboard(spec) {
|
|
|
1457
1518
|
if (!rawModels || rawModels.length === 0) throw new Error(`defineLeaderboard(${spec.name}): no models — pass --models, set spec.axis.models, or give spec.baseProfile a model.default`);
|
|
1458
1519
|
const models = rawModels.map((m) => withSnapshot(m, snapshot));
|
|
1459
1520
|
const profiles = expandProfileAxes({
|
|
1460
|
-
base: spec.baseProfile
|
|
1461
|
-
name: spec.name,
|
|
1462
|
-
model: { default: bareModel(models[0] ?? "") }
|
|
1463
|
-
},
|
|
1521
|
+
base: spec.baseProfile,
|
|
1464
1522
|
harnesses,
|
|
1465
1523
|
models
|
|
1466
1524
|
});
|
|
1525
|
+
for (const profile of profiles) assertExecutableAgentProfile(profile, `defineLeaderboard(${spec.name})`);
|
|
1467
1526
|
const ctx = {
|
|
1468
1527
|
name: spec.name,
|
|
1469
1528
|
backend: backendName,
|
|
@@ -1488,7 +1547,6 @@ function defineLeaderboard(spec) {
|
|
|
1488
1547
|
bridge: {
|
|
1489
1548
|
url: process.env.CLI_BRIDGE_URL,
|
|
1490
1549
|
bearer,
|
|
1491
|
-
model: bareModel(models[0] ?? ""),
|
|
1492
1550
|
timeoutMs: 9e5
|
|
1493
1551
|
}
|
|
1494
1552
|
});
|
|
@@ -1518,21 +1576,11 @@ function defineLeaderboard(spec) {
|
|
|
1518
1576
|
return [...served][0] ?? cellProfile.model?.default;
|
|
1519
1577
|
},
|
|
1520
1578
|
toLoopOptions: (cellScenario, cellProfile) => {
|
|
1521
|
-
const axis = harnessAxisOf(cellProfile);
|
|
1522
|
-
const modelId = bareModel(axis?.model ?? models[0] ?? "");
|
|
1523
|
-
const backendModel = {
|
|
1524
|
-
...spec.modelBackend,
|
|
1525
|
-
...!isHarnessNativeModel(modelId) || backendName === "cli-bridge" ? { model: modelId } : {}
|
|
1526
|
-
};
|
|
1527
1579
|
return {
|
|
1528
1580
|
driver: naiveRetryDriver(shots),
|
|
1529
1581
|
agentRun: {
|
|
1530
1582
|
profile: cellProfile,
|
|
1531
|
-
taskToPrompt: (s) => `${promptOf(s)}\n\n<!-- independent-attempt:${shotNonce++}
|
|
1532
|
-
...axis ? { sandboxOverrides: { backend: {
|
|
1533
|
-
type: axis.harness,
|
|
1534
|
-
...Object.keys(backendModel).length > 0 ? { model: backendModel } : {}
|
|
1535
|
-
} } } : {}
|
|
1583
|
+
taskToPrompt: (s) => `${promptOf(s)}\n\n<!-- independent-attempt:${shotNonce++} -->`
|
|
1536
1584
|
},
|
|
1537
1585
|
output: { parse: (events) => spec.parseOutput ? spec.parseOutput(events, cellScenario.case) : collectAgentResponseText(events) ?? "" },
|
|
1538
1586
|
validator: { validate: async (output) => {
|
|
@@ -1653,11 +1701,10 @@ async function harvestCorpus(opts) {
|
|
|
1653
1701
|
if (opts.signal?.aborted) return;
|
|
1654
1702
|
try {
|
|
1655
1703
|
const obs = await observe(input, {
|
|
1656
|
-
|
|
1657
|
-
|
|
1704
|
+
profile: opts.profile,
|
|
1705
|
+
executor: opts.executor,
|
|
1658
1706
|
corpus: opts.corpus,
|
|
1659
1707
|
tags: opts.tags ?? [],
|
|
1660
|
-
...opts.analystInstruction ? { analystInstruction: opts.analystInstruction } : {},
|
|
1661
1708
|
...opts.signal ? { signal: opts.signal } : {}
|
|
1662
1709
|
});
|
|
1663
1710
|
report.runsObserved += 1;
|
|
@@ -1941,7 +1988,7 @@ function observedBestScore(settledSoFar) {
|
|
|
1941
1988
|
* CONCRETE blocker (never an eager over-fan, never a silent drop), and a `blocked` outcome always
|
|
1942
1989
|
* names at least one blocker (a shape that cannot finish MUST say why — `blocked([])` throws).
|
|
1943
1990
|
*
|
|
1944
|
-
* @
|
|
1991
|
+
* @stable
|
|
1945
1992
|
*/
|
|
1946
1993
|
/**
|
|
1947
1994
|
* The single content-free valid-only winner selector. Among the gated-VALID children only
|
|
@@ -1976,6 +2023,8 @@ function selectValidWinner(opts) {
|
|
|
1976
2023
|
* pool would not admit, or a stage whose `collect` chose to block) short-circuits — its blockers
|
|
1977
2024
|
* ARE the pipeline's blockers, never coerced past a failed stage. The terminal stage's `done`
|
|
1978
2025
|
* deliverable is the pipeline's deliverable.
|
|
2026
|
+
*
|
|
2027
|
+
* @stable
|
|
1979
2028
|
*/
|
|
1980
2029
|
function pipeline(stages) {
|
|
1981
2030
|
if (stages.length === 0) throw new ValidationError("pipeline: at least one stage is required");
|
|
@@ -2012,6 +2061,8 @@ function pipeline(stages) {
|
|
|
2012
2061
|
* `opts.width` swaps the single round for `rollingDispatch`: at most `width` items live at once,
|
|
2013
2062
|
* refilled the instant one settles. Selection, blockers, and the conserved pool are unchanged —
|
|
2014
2063
|
* the refill behavior lives in the existing combinator rather than in a rival primitive.
|
|
2064
|
+
*
|
|
2065
|
+
* @stable
|
|
2015
2066
|
*/
|
|
2016
2067
|
function fanout(items, opts) {
|
|
2017
2068
|
if (opts.synthesize && opts.selectWinner) throw new ValidationError("fanout: pass at most one of `synthesize` or `selectWinner`");
|
|
@@ -2095,6 +2146,8 @@ function fanout(items, opts) {
|
|
|
2095
2146
|
* `until` on the resulting trace-derived findings (the analyst spawns into THIS scope, so its
|
|
2096
2147
|
* compute is conserved-pooled — equal-k holds by construction). Absent an analyst the findings
|
|
2097
2148
|
* argument is the empty array — never a fabricated finding (fail-loud honesty over a silent default).
|
|
2149
|
+
*
|
|
2150
|
+
* @stable
|
|
2098
2151
|
*/
|
|
2099
2152
|
function loopUntil(seed, spec) {
|
|
2100
2153
|
return (ctx) => ({
|
|
@@ -2143,6 +2196,8 @@ function loopUntil(seed, spec) {
|
|
|
2143
2196
|
* reaches another judge's task; the merge never spawns or re-ranks). A `down` judge carries no
|
|
2144
2197
|
* verdict and is excluded from the merge denominator. A panel that admitted no judge is a
|
|
2145
2198
|
* concrete blocker before `merge` is consulted.
|
|
2199
|
+
*
|
|
2200
|
+
* @stable
|
|
2146
2201
|
*/
|
|
2147
2202
|
function panel(spec) {
|
|
2148
2203
|
if (spec.judges.length === 0) throw new ValidationError("panel: at least one judge is required");
|
|
@@ -2192,6 +2247,8 @@ function panel(spec) {
|
|
|
2192
2247
|
* it; only a `valid` verifier verdict ships. Any other outcome (implement down, verifier down,
|
|
2193
2248
|
* verifier verdict absent or not `valid`) is a concrete blocker carrying the failure verbatim —
|
|
2194
2249
|
* never a coerced "done". The implement child does not grade itself.
|
|
2250
|
+
*
|
|
2251
|
+
* @stable
|
|
2195
2252
|
*/
|
|
2196
2253
|
function verify(spec) {
|
|
2197
2254
|
return (ctx) => ({
|
|
@@ -2236,6 +2293,8 @@ function verify(spec) {
|
|
|
2236
2293
|
* the widen loop sees it. The shipped default (`flatWidenGate`) never widens, so no widen child is
|
|
2237
2294
|
* ever live when the analyst runs and the wire is exact; a non-flat gate must drive the analyst on
|
|
2238
2295
|
* a scope whose siblings are quiesced, or read findings without the shared-cursor drain.
|
|
2296
|
+
*
|
|
2297
|
+
* @stable
|
|
2239
2298
|
*/
|
|
2240
2299
|
function widen(spec) {
|
|
2241
2300
|
return (ctx) => ({
|
|
@@ -2685,19 +2744,22 @@ function registerShape(name, factory) {
|
|
|
2685
2744
|
* receive a ctx with the persona seams merged in — so a persona never has to pre-close its
|
|
2686
2745
|
* factories by hand. A persona may instead supply a fully-built `registry` and skip the wrap.
|
|
2687
2746
|
*
|
|
2688
|
-
* @
|
|
2747
|
+
* @stable
|
|
2689
2748
|
*/
|
|
2690
2749
|
/**
|
|
2691
2750
|
* Build a frozen `Persona`. Fails loud on the executors-supplied invariant: a persona with
|
|
2692
2751
|
* neither a pre-built registry nor a seam bag cannot resolve its built-in runtimes, so it is
|
|
2693
2752
|
* unrunnable — refuse it at definition time, not at the first spawn. Pure; no I/O.
|
|
2753
|
+
*
|
|
2754
|
+
* @stable
|
|
2694
2755
|
*/
|
|
2695
2756
|
function definePersona(input) {
|
|
2696
2757
|
if (!input.executors.registry && !input.executors.seams) throw new ValidationError(`definePersona("${input.name}"): executors must supply a registry or a seams bag (built-in runtimes read their seams off ExecutorContext; neither was provided)`);
|
|
2697
2758
|
if (!input.root || typeof input.root !== "object" || !("harness" in input.root)) throw new ValidationError(`definePersona("${input.name}"): root must be an AgentSpec`);
|
|
2759
|
+
const root = executableAgentSpecSnapshot(input.root, `definePersona("${input.name}")`);
|
|
2698
2760
|
return Object.freeze({
|
|
2699
2761
|
name: input.name,
|
|
2700
|
-
root
|
|
2762
|
+
root,
|
|
2701
2763
|
directive: input.directive,
|
|
2702
2764
|
context: input.context,
|
|
2703
2765
|
executors: input.executors,
|
|
@@ -2740,6 +2802,8 @@ function createShapeContext(persona, budget, analyst) {
|
|
|
2740
2802
|
* `ShapeContext`, and runs the resulting root `Agent` to a typed `SupervisedResult<Outcome>`.
|
|
2741
2803
|
* Fail loud on an unknown shape name or an unresolvable persona registry — never a silent
|
|
2742
2804
|
* default-shape fallback.
|
|
2805
|
+
*
|
|
2806
|
+
* @stable
|
|
2743
2807
|
*/
|
|
2744
2808
|
async function runPersonified(options) {
|
|
2745
2809
|
const { persona } = options;
|
|
@@ -3290,22 +3354,34 @@ async function pool(items, limit, fn) {
|
|
|
3290
3354
|
async function preflightModels(cfg) {
|
|
3291
3355
|
if (cfg.modelPreflight === false) return;
|
|
3292
3356
|
if (cfg.worker.complete && !cfg.modelPreflight) return;
|
|
3293
|
-
const
|
|
3357
|
+
const profiles = [cfg.worker.workerProfile, cfg.worker.analystProfile ?? cfg.worker.workerProfile];
|
|
3358
|
+
const profilesByModel = /* @__PURE__ */ new Map();
|
|
3359
|
+
for (const [index, profile] of profiles.entries()) {
|
|
3360
|
+
const model = concreteModelId(profile.model?.default);
|
|
3361
|
+
if (!model) throw new Error(`Benchmark ${index === 0 ? "worker" : "analyst"} AgentProfile.model.default must name an exact model`);
|
|
3362
|
+
if (!profilesByModel.has(model)) profilesByModel.set(model, profile);
|
|
3363
|
+
}
|
|
3364
|
+
const models = [...profilesByModel.keys()];
|
|
3294
3365
|
const timeoutMs = cfg.modelPreflightTimeoutMs ?? 3e4;
|
|
3295
3366
|
if (!Number.isFinite(timeoutMs) || timeoutMs <= 0) throw new Error("modelPreflightTimeoutMs must be a positive finite number");
|
|
3296
3367
|
const check = cfg.modelPreflight ?? (async (model, worker, signal) => {
|
|
3297
|
-
|
|
3298
|
-
|
|
3299
|
-
|
|
3300
|
-
|
|
3301
|
-
|
|
3302
|
-
|
|
3303
|
-
|
|
3304
|
-
|
|
3305
|
-
|
|
3306
|
-
|
|
3307
|
-
|
|
3308
|
-
|
|
3368
|
+
const profile = profilesByModel.get(model);
|
|
3369
|
+
if (!profile) throw new Error(`Benchmark preflight has no AgentProfile for model ${model}`);
|
|
3370
|
+
await profileChatClient({
|
|
3371
|
+
profile,
|
|
3372
|
+
context: `runBenchmark model preflight (${model})`,
|
|
3373
|
+
executor: {
|
|
3374
|
+
backend: "router",
|
|
3375
|
+
routerBaseUrl: worker.routerBaseUrl,
|
|
3376
|
+
routerKey: worker.routerKey
|
|
3377
|
+
}
|
|
3378
|
+
}).chat({
|
|
3379
|
+
model,
|
|
3380
|
+
messages: [{
|
|
3381
|
+
role: "user",
|
|
3382
|
+
content: "Reply OK."
|
|
3383
|
+
}]
|
|
3384
|
+
}, { signal });
|
|
3309
3385
|
});
|
|
3310
3386
|
const failures = (await Promise.allSettled(models.map(async (model) => {
|
|
3311
3387
|
const controller = new AbortController();
|
|
@@ -3361,8 +3437,10 @@ async function runBenchmark(cfg) {
|
|
|
3361
3437
|
resolved: r.resolved,
|
|
3362
3438
|
progression: r.progression,
|
|
3363
3439
|
usd: r.usd,
|
|
3440
|
+
usdKnown: r.usdKnown,
|
|
3364
3441
|
ms: r.ms,
|
|
3365
|
-
tokens: r.tokens
|
|
3442
|
+
tokens: r.tokens,
|
|
3443
|
+
tokensKnown: r.tokensKnown
|
|
3366
3444
|
};
|
|
3367
3445
|
} catch (e) {
|
|
3368
3446
|
errors[s.name] = e instanceof Error ? e.message.slice(0, 300) : String(e);
|
|
@@ -3371,11 +3449,13 @@ async function runBenchmark(cfg) {
|
|
|
3371
3449
|
resolved: false,
|
|
3372
3450
|
progression: [],
|
|
3373
3451
|
usd: 0,
|
|
3452
|
+
usdKnown: true,
|
|
3374
3453
|
ms: 0,
|
|
3375
3454
|
tokens: {
|
|
3376
3455
|
input: 0,
|
|
3377
3456
|
output: 0
|
|
3378
|
-
}
|
|
3457
|
+
},
|
|
3458
|
+
tokensKnown: true
|
|
3379
3459
|
};
|
|
3380
3460
|
}
|
|
3381
3461
|
row = {
|
|
@@ -3402,6 +3482,7 @@ async function runBenchmark(cfg) {
|
|
|
3402
3482
|
score: mean(cells.map((c) => c.score)),
|
|
3403
3483
|
resolved: mean(cells.map((c) => c.resolved ? 1 : 0)),
|
|
3404
3484
|
usd: mean(cells.map((c) => c.usd)),
|
|
3485
|
+
usdKnownRate: mean(cells.map((c) => c.usdKnown ? 1 : 0)),
|
|
3405
3486
|
ms: mean(cells.map((c) => c.ms))
|
|
3406
3487
|
};
|
|
3407
3488
|
}
|
|
@@ -3793,6 +3874,8 @@ export default defineStrategy('your-strategy-name', async ({ surface, task, budg
|
|
|
3793
3874
|
// your composition (listTools comes from the destructured context — it is NOT a global)
|
|
3794
3875
|
})
|
|
3795
3876
|
`;
|
|
3877
|
+
/** Standing behavior callers put in the strategy-author AgentProfile. */
|
|
3878
|
+
const strategyAuthorSystemPrompt = "You are a senior researcher authoring optimization strategies for agent loops: you read per-task losses like experimental data, form a mechanism-level hypothesis, and author the one composition that tests it. Output exactly one fenced ```ts code block and nothing else.";
|
|
3796
3879
|
/** Static CONTRACT lint over an authored strategy module — the module-boundary
|
|
3797
3880
|
* enforcement of the harness's two measurement invariants:
|
|
3798
3881
|
* - author blindness: the only import allowed is the kernel surface. A body that could
|
|
@@ -3819,21 +3902,17 @@ function assertStrategyContract(code) {
|
|
|
3819
3902
|
}
|
|
3820
3903
|
/** One authoring attempt: chat with the given model, extract the fenced module. Throws
|
|
3821
3904
|
* when the reply carries no code block. */
|
|
3822
|
-
async function requestAuthoredCode(opts,
|
|
3823
|
-
const res = await
|
|
3824
|
-
|
|
3825
|
-
|
|
3826
|
-
|
|
3827
|
-
|
|
3828
|
-
|
|
3829
|
-
|
|
3830
|
-
|
|
3831
|
-
role: "user",
|
|
3832
|
-
content: `${opts.contract ?? strategyAuthorContract}\n\nBASELINE RESULTS on the "${opts.environmentName}" environment (budget=${opts.budget}) — the per-task losses are your gradient:\n${opts.lossesJson}\n\nAuthor ONE new strategy that you expect to beat the baselines on THIS environment at the same budget.\n${strategyAuthorMethod}\n\nOutput only the module code block.`
|
|
3833
|
-
}]
|
|
3834
|
-
}, { ...opts.signal ? { signal: opts.signal } : {} });
|
|
3905
|
+
async function requestAuthoredCode(opts, profile) {
|
|
3906
|
+
const res = await profileChatClient({
|
|
3907
|
+
profile,
|
|
3908
|
+
executor: opts.executor,
|
|
3909
|
+
context: "strategy author"
|
|
3910
|
+
}).chat({ messages: [{
|
|
3911
|
+
role: "user",
|
|
3912
|
+
content: `${opts.contract ?? strategyAuthorContract}\n\nBASELINE RESULTS on the "${opts.environmentName}" environment (budget=${opts.budget}) — the per-task losses are your gradient:\n${opts.lossesJson}\n\nAuthor ONE new strategy that you expect to beat the baselines on THIS environment at the same budget.\n${strategyAuthorMethod}\n\nOutput only the module code block.`
|
|
3913
|
+
}] }, { ...opts.signal ? { signal: opts.signal } : {} });
|
|
3835
3914
|
const match = res.content.match(/```(?:ts|typescript)?\s*\n([\s\S]*?)```/);
|
|
3836
|
-
if (!match?.[1]) throw new Error(`authorStrategy: no code block in the author's reply
|
|
3915
|
+
if (!match?.[1]) throw new Error(`authorStrategy: no code block in the author's reply: ${res.content.slice(0, 300)}`);
|
|
3837
3916
|
return match[1];
|
|
3838
3917
|
}
|
|
3839
3918
|
/** Author + load a strategy from losses. Throws when the author emits no loadable module;
|
|
@@ -3841,10 +3920,10 @@ async function requestAuthoredCode(opts, model) {
|
|
|
3841
3920
|
async function authorStrategy(opts) {
|
|
3842
3921
|
let code;
|
|
3843
3922
|
try {
|
|
3844
|
-
code = await requestAuthoredCode(opts, opts.
|
|
3923
|
+
code = await requestAuthoredCode(opts, opts.profile);
|
|
3845
3924
|
} catch (primaryError) {
|
|
3846
|
-
if (!opts.
|
|
3847
|
-
code = await requestAuthoredCode(opts, opts.
|
|
3925
|
+
if (!opts.fallbackProfile) throw primaryError;
|
|
3926
|
+
code = await requestAuthoredCode(opts, opts.fallbackProfile);
|
|
3848
3927
|
}
|
|
3849
3928
|
assertStrategyContract(code);
|
|
3850
3929
|
mkdirSync(opts.outDir, { recursive: true });
|
|
@@ -3882,6 +3961,8 @@ async function authorStrategy(opts) {
|
|
|
3882
3961
|
* Lineage fields (`parent`, `generation`) are recorded on every archive node so a
|
|
3883
3962
|
* descendant-productivity parent-selection policy can be added without changing the
|
|
3884
3963
|
* report schema; the v1 search authors from the latest tournament's losses.
|
|
3964
|
+
*
|
|
3965
|
+
* @experimental
|
|
3885
3966
|
*/
|
|
3886
3967
|
/** Strategy means recomputed over the DISCRIMINATING tasks only — tasks where the field
|
|
3887
3968
|
* strategies did not all score identically. Zero-spread tasks (everyone 1.0, everyone
|
|
@@ -4083,11 +4164,9 @@ async function runStrategyEvolution(cfg) {
|
|
|
4083
4164
|
const contract = `${strategyAuthorContract}${cfg.objective === "cost" ? `\n\nYOUR OBJECTIVE: match or exceed the incumbent's SCORE while spending LESS (the losses include usd per task). Promotion requires proven score non-inferiority PLUS significant cost savings — a strategy that ties the score at half the cost WINS; a cheaper strategy that loses score by more than ${((cfg.scoreTolerance ?? .05) * 100).toFixed(0)}pp LOSES.` : ""}\n\nEXAMPLE TOOLS FROM ONE TASK (tool sets VARY per task on this domain — a strategy MUST select tool names from await listTools(handle) at runtime; hardcoding these example names will zero your score on most tasks):\n${toolCatalog}\n\nSTRATEGIES ALREADY IN THE TOURNAMENT (author something MEANINGFULLY different — a new composition, not a rename):\n${fieldSummary(archive)}\n\nYou are authoring candidate ${i + 1} of ${populationSize} this generation; explore a distinct region of the strategy space from your siblings.`;
|
|
4084
4165
|
try {
|
|
4085
4166
|
const authored = await authorStrategy({
|
|
4086
|
-
|
|
4087
|
-
|
|
4088
|
-
...cfg.author.
|
|
4089
|
-
...cfg.author.temperature !== void 0 ? { temperature: cfg.author.temperature } : {},
|
|
4090
|
-
...cfg.author.maxTokens !== void 0 ? { maxTokens: cfg.author.maxTokens } : {},
|
|
4167
|
+
profile: cfg.author.profile,
|
|
4168
|
+
executor: cfg.author.executor,
|
|
4169
|
+
...cfg.author.fallbackProfile ? { fallbackProfile: cfg.author.fallbackProfile } : {},
|
|
4091
4170
|
contract,
|
|
4092
4171
|
environmentName: cfg.environment.name,
|
|
4093
4172
|
lossesJson,
|
|
@@ -4224,24 +4303,24 @@ async function runStrategyEvolution(cfg) {
|
|
|
4224
4303
|
const tolerance = cfg.reproducerCheck.tolerance ?? .05;
|
|
4225
4304
|
const championHoldoutScore = holdout.perStrategy[incumbent.name]?.score ?? 0;
|
|
4226
4305
|
try {
|
|
4227
|
-
const summary = (await
|
|
4228
|
-
|
|
4229
|
-
|
|
4230
|
-
|
|
4231
|
-
|
|
4232
|
-
|
|
4233
|
-
|
|
4234
|
-
},
|
|
4235
|
-
|
|
4236
|
-
|
|
4237
|
-
|
|
4238
|
-
|
|
4306
|
+
const summary = (await profileChatClient({
|
|
4307
|
+
profile: {
|
|
4308
|
+
...cfg.author.profile,
|
|
4309
|
+
prompt: {
|
|
4310
|
+
...cfg.author.profile.prompt,
|
|
4311
|
+
systemPrompt: `Summarize the optimization strategy implemented by this code in at most ${words} words. Describe the COMPOSITION (shots, critique, artifact handling, restarts, stopping) — not the code. Output only the summary.`
|
|
4312
|
+
}
|
|
4313
|
+
},
|
|
4314
|
+
executor: cfg.author.executor,
|
|
4315
|
+
context: "strategy reproducer summary"
|
|
4316
|
+
}).chat({ messages: [{
|
|
4317
|
+
role: "user",
|
|
4318
|
+
content: championCode
|
|
4319
|
+
}] })).content.trim();
|
|
4239
4320
|
const reproduced = await authorStrategy({
|
|
4240
|
-
|
|
4241
|
-
|
|
4242
|
-
...cfg.author.
|
|
4243
|
-
...cfg.author.maxTokens !== void 0 ? { maxTokens: cfg.author.maxTokens } : {},
|
|
4244
|
-
temperature: .2,
|
|
4321
|
+
profile: cfg.author.profile,
|
|
4322
|
+
executor: cfg.author.executor,
|
|
4323
|
+
...cfg.author.fallbackProfile ? { fallbackProfile: cfg.author.fallbackProfile } : {},
|
|
4245
4324
|
contract: `${strategyAuthorContract}\n\nIMPLEMENT EXACTLY THIS STRATEGY (a colleague's description — do not invent a different approach):\n${summary}`,
|
|
4246
4325
|
environmentName: cfg.environment.name,
|
|
4247
4326
|
lossesJson: "[]",
|
|
@@ -4288,737 +4367,126 @@ async function runStrategyEvolution(cfg) {
|
|
|
4288
4367
|
};
|
|
4289
4368
|
}
|
|
4290
4369
|
//#endregion
|
|
4291
|
-
//#region src/runtime/
|
|
4292
|
-
/**
|
|
4293
|
-
* `streamAgentTurn` — the ONE run-a-turn event-stream contract over every
|
|
4294
|
-
* execution substrate: a sandbox box (`SandboxInstance.streamPrompt`), a
|
|
4295
|
-
* one-shot `Executor` (cli-bridge / router / BYO, via `ExecutorFactory`), and
|
|
4296
|
-
* an in-process `AgentExecutionBackend` (the `resolveAgentBackend` output).
|
|
4297
|
-
*
|
|
4298
|
-
* One function, one vocabulary: every backend kind yields the existing
|
|
4299
|
-
* `RuntimeStreamEvent` union incrementally and ALWAYS terminates with a
|
|
4300
|
-
* `final` event whose `text` is the turn's final text and whose
|
|
4301
|
-
* `metadata.tokenUsage` / `metadata.costUsd` / `metadata.model` carry the
|
|
4302
|
-
* turn's metered usage. `collectAgentTurn` drains a stream into that terminal
|
|
4303
|
-
* summary plus the full event list.
|
|
4304
|
-
*
|
|
4305
|
-
* This is a UNIFICATION seam, not a new stream parser — each kind is a thin
|
|
4306
|
-
* adapter over code that already exists and is already hardened:
|
|
4307
|
-
* - `box` — `mapSandboxEvent` + `extractLlmCallEvent` (sandbox-events.ts)
|
|
4308
|
-
* project the sandbox event stream; nothing is re-mapped here.
|
|
4309
|
-
* - `executor` — `inlineSandboxClient` (the ONE executor→box adapter) turns
|
|
4310
|
-
* the factory into a box, then the box path drives it. The
|
|
4311
|
-
* executor's settle/teardown lifecycle stays in that adapter.
|
|
4312
|
-
* - `chat` — the backend's own `stream()` surface, normalized by
|
|
4313
|
-
* `normalizeBackendStreamEvent` (the same projection
|
|
4314
|
-
* `runAgentTaskStream` applies).
|
|
4315
|
-
*
|
|
4316
|
-
* Distinct from `openSandboxRun` (box-only, session resume over one persistent
|
|
4317
|
-
* artifact, raw `SandboxEvent` deliverables) and from `runAgentTaskStream`
|
|
4318
|
-
* (full task lifecycle: knowledge preflight, session store, resume). This is
|
|
4319
|
-
* the minimal turn primitive underneath both worlds: prompt in, one normalized
|
|
4320
|
-
* event stream out, terminal result+usage guaranteed on every non-thrown path.
|
|
4321
|
-
*
|
|
4322
|
-
* Stream envelope: `backend_start` → incremental events → (`backend_error` on
|
|
4323
|
-
* failure) → `final`. A caller-initiated abort terminates with
|
|
4324
|
-
* `final.status: 'aborted'`; an expired `timeoutMs` deadline with
|
|
4325
|
-
* `final.status: 'failed'` — so cancellation stays distinguishable from a
|
|
4326
|
-
* blown deadline.
|
|
4327
|
-
*
|
|
4328
|
-
* Mid-stream lifecycle work needs NO extra API: the generator is pull-based,
|
|
4329
|
-
* so the producer is suspended between yields and resumes only when the caller
|
|
4330
|
-
* pulls again. A consumer can therefore run arbitrary async work between
|
|
4331
|
-
* events — sync state on each `tool_result`, decide a no-op retry after
|
|
4332
|
-
* draining, run a pre-`done` flush when it receives `final` and BEFORE it
|
|
4333
|
-
* forwards its own terminal event downstream. The interleaving is guaranteed
|
|
4334
|
-
* (and locked by test): nothing is produced past the event the caller is
|
|
4335
|
-
* holding.
|
|
4336
|
-
*
|
|
4337
|
-
* @experimental
|
|
4338
|
-
*/
|
|
4370
|
+
//#region src/runtime/supervise/chat-transport-executor.ts
|
|
4339
4371
|
/**
|
|
4340
|
-
*
|
|
4341
|
-
* `RuntimeStreamEvent` vocabulary incrementally and always ends with a `final`
|
|
4342
|
-
* event carrying the turn's text and usage (`metadata.tokenUsage`,
|
|
4343
|
-
* `metadata.costUsd?`, `metadata.model?`) — on success, failure, abort, and
|
|
4344
|
-
* timeout alike. The generator never throws; failures surface in-band as
|
|
4345
|
-
* `backend_error` + `final` with a typed `error` detail.
|
|
4372
|
+
* A session-owning composition over Runtime's canonical Router tool-loop executor.
|
|
4346
4373
|
*
|
|
4347
|
-
*
|
|
4348
|
-
|
|
4349
|
-
|
|
4350
|
-
const label = backend.kind === "chat" ? backend.backend.kind : backend.kind;
|
|
4351
|
-
const task = {
|
|
4352
|
-
id: `turn-${crypto.randomUUID()}`,
|
|
4353
|
-
intent: prompt
|
|
4354
|
-
};
|
|
4355
|
-
const acc = {
|
|
4356
|
-
deltaText: "",
|
|
4357
|
-
input: 0,
|
|
4358
|
-
output: 0,
|
|
4359
|
-
costUsd: 0
|
|
4360
|
-
};
|
|
4361
|
-
const deadline = deriveTurnSignal(opts.signal, opts.timeoutMs ?? 0);
|
|
4362
|
-
let session;
|
|
4363
|
-
try {
|
|
4364
|
-
session = await startTurnSession(backend, task, prompt, deadline.signal, label);
|
|
4365
|
-
yield {
|
|
4366
|
-
type: "backend_start",
|
|
4367
|
-
task,
|
|
4368
|
-
session,
|
|
4369
|
-
backend: label,
|
|
4370
|
-
timestamp: nowIso()
|
|
4371
|
-
};
|
|
4372
|
-
const inner = backend.kind === "chat" ? driveChatTurn(backend.backend, task, session, prompt, deadline.signal, acc) : driveBoxTurn(backend.kind === "executor" ? await inlineSandboxClient(backend.factory).create() : backend.box, prompt, deadline.signal, backend.agentRunName ?? "agent", acc, {
|
|
4373
|
-
...backend.kind !== "executor" && backend.options ? { options: backend.options } : {},
|
|
4374
|
-
preserveToolParts: opts.preserveToolParts === true,
|
|
4375
|
-
...opts.onRawEvent ? { onRawEvent: opts.onRawEvent } : {}
|
|
4376
|
-
});
|
|
4377
|
-
for await (const event of inner) {
|
|
4378
|
-
yield event;
|
|
4379
|
-
throwIfAborted(deadline.signal);
|
|
4380
|
-
}
|
|
4381
|
-
yield buildFinalEvent(task, session, acc, {
|
|
4382
|
-
status: "completed",
|
|
4383
|
-
reason: "turn completed"
|
|
4384
|
-
});
|
|
4385
|
-
} catch (err) {
|
|
4386
|
-
const callerAborted = opts.signal?.aborted === true;
|
|
4387
|
-
const status = callerAborted ? "aborted" : "failed";
|
|
4388
|
-
const message = err instanceof Error ? err.message : String(err);
|
|
4389
|
-
const error = err instanceof BackendTransportError ? {
|
|
4390
|
-
kind: "transport",
|
|
4391
|
-
message,
|
|
4392
|
-
status: err.status,
|
|
4393
|
-
body: err.body
|
|
4394
|
-
} : {
|
|
4395
|
-
kind: "backend",
|
|
4396
|
-
message
|
|
4397
|
-
};
|
|
4398
|
-
yield {
|
|
4399
|
-
type: "backend_error",
|
|
4400
|
-
task,
|
|
4401
|
-
...session ? { session } : {},
|
|
4402
|
-
backend: label,
|
|
4403
|
-
message,
|
|
4404
|
-
recoverable: !callerAborted,
|
|
4405
|
-
error,
|
|
4406
|
-
timestamp: nowIso()
|
|
4407
|
-
};
|
|
4408
|
-
yield buildFinalEvent(task, session, acc, {
|
|
4409
|
-
status,
|
|
4410
|
-
reason: message,
|
|
4411
|
-
error
|
|
4412
|
-
});
|
|
4413
|
-
} finally {
|
|
4414
|
-
deadline.dispose();
|
|
4415
|
-
}
|
|
4416
|
-
}
|
|
4417
|
-
/**
|
|
4418
|
-
* Drain a `streamAgentTurn` stream (or any `RuntimeStreamEvent` stream that
|
|
4419
|
-
* honors its terminal contract) into the turn summary plus the full event
|
|
4420
|
-
* list. Fail-loud: throws when the stream ends without a terminal `final`
|
|
4421
|
-
* event — a stream that violates the contract must not read as an empty turn.
|
|
4374
|
+
* This module adds conversation persistence for graph-edge `resume` continuity. It does not own
|
|
4375
|
+
* model selection, prompts, generation controls, retries, tool policy, or provider accounting:
|
|
4376
|
+
* those are lowered from one exact `AgentProfile` by `createExecutor({ backend: 'router-tools' })`.
|
|
4422
4377
|
*
|
|
4423
4378
|
* @experimental
|
|
4424
4379
|
*/
|
|
4425
|
-
|
|
4426
|
-
|
|
4427
|
-
|
|
4428
|
-
const final = events.at(-1);
|
|
4429
|
-
if (final?.type !== "final") throw new Error(`collectAgentTurn: stream ended without a terminal 'final' event (last: ${final ? final.type : "none"})`);
|
|
4430
|
-
const metadata = final.metadata ?? {};
|
|
4431
|
-
const tokenUsage = metadata.tokenUsage && typeof metadata.tokenUsage === "object" ? metadata.tokenUsage : {};
|
|
4432
|
-
const usage = {
|
|
4433
|
-
input: finiteNumber(tokenUsage.input) ?? 0,
|
|
4434
|
-
output: finiteNumber(tokenUsage.output) ?? 0
|
|
4435
|
-
};
|
|
4436
|
-
const costUsd = finiteNumber(metadata.costUsd);
|
|
4437
|
-
if (costUsd !== void 0) usage.costUsd = costUsd;
|
|
4438
|
-
if (typeof metadata.model === "string" && metadata.model.length > 0) usage.model = metadata.model;
|
|
4380
|
+
/** In-memory, process-local conversation store with detached reads and writes. */
|
|
4381
|
+
function createChatSessionStore() {
|
|
4382
|
+
const sessions = /* @__PURE__ */ new Map();
|
|
4439
4383
|
return {
|
|
4440
|
-
|
|
4441
|
-
|
|
4442
|
-
|
|
4443
|
-
status: final.status,
|
|
4444
|
-
...final.error ? { error: final.error } : {}
|
|
4445
|
-
};
|
|
4446
|
-
}
|
|
4447
|
-
/** Start the backend's session when it owns one (`chat` kind); mint a local
|
|
4448
|
-
* correlation session otherwise. Box/executor turns carry no server session
|
|
4449
|
-
* here — resume lives in `openSandboxRun`/`SandboxLineage`, not this primitive. */
|
|
4450
|
-
async function startTurnSession(backend, task, prompt, signal, label) {
|
|
4451
|
-
if (backend.kind === "chat" && backend.backend.start) return backend.backend.start({
|
|
4452
|
-
task,
|
|
4453
|
-
message: prompt
|
|
4454
|
-
}, {
|
|
4455
|
-
task,
|
|
4456
|
-
knowledge: emptyReadiness(task),
|
|
4457
|
-
signal
|
|
4458
|
-
});
|
|
4459
|
-
return newRuntimeSession(label);
|
|
4460
|
-
}
|
|
4461
|
-
/**
|
|
4462
|
-
* One turn over a box: `box.streamPrompt` projected through the existing
|
|
4463
|
-
* `mapSandboxEvent` (text/reasoning deltas +
|
|
4464
|
-
* cost-bearing `llm_call`s), plus the opt-in `mapSandboxToolEvent` tool-part
|
|
4465
|
-
* projection. Usage accumulates off the mapped `llm_call` events — the same
|
|
4466
|
-
* fold `sumSandboxUsage` applies. Final text prefers the terminal
|
|
4467
|
-
* `result`/`done`/`final` payload over concatenated deltas, because the
|
|
4468
|
-
* sandbox `message.part.updated` fallback may carry running accumulations.
|
|
4469
|
-
*/
|
|
4470
|
-
async function* driveBoxTurn(box, prompt, signal, agentRunName, acc, cfg) {
|
|
4471
|
-
const callOptions = {
|
|
4472
|
-
...cfg.options ?? {},
|
|
4473
|
-
signal
|
|
4474
|
-
};
|
|
4475
|
-
const stream = box.streamPrompt(prompt, callOptions);
|
|
4476
|
-
const toolParts = cfg.preserveToolParts ? createSandboxToolPartState() : void 0;
|
|
4477
|
-
for await (const event of stream) {
|
|
4478
|
-
if (cfg.onRawEvent) await cfg.onRawEvent(event);
|
|
4479
|
-
const terminalText = terminalTextFromSandboxEvent(event);
|
|
4480
|
-
if (terminalText !== void 0) acc.terminalText = terminalText;
|
|
4481
|
-
if (toolParts) for (const toolEvent of mapSandboxToolEvent(event, toolParts)) yield toolEvent;
|
|
4482
|
-
const mapped = mapSandboxEvent(event, { agentRunName });
|
|
4483
|
-
if (!mapped) continue;
|
|
4484
|
-
foldEvent(mapped, acc, agentRunName);
|
|
4485
|
-
yield mapped;
|
|
4486
|
-
}
|
|
4487
|
-
}
|
|
4488
|
-
/** One turn over an in-process backend: its own `stream()` surface, projected
|
|
4489
|
-
* through the same `normalizeBackendStreamEvent` the task lifecycle applies. */
|
|
4490
|
-
async function* driveChatTurn(backend, task, session, prompt, signal, acc) {
|
|
4491
|
-
const input = {
|
|
4492
|
-
task,
|
|
4493
|
-
message: prompt
|
|
4494
|
-
};
|
|
4495
|
-
const context = {
|
|
4496
|
-
task,
|
|
4497
|
-
knowledge: emptyReadiness(task),
|
|
4498
|
-
session,
|
|
4499
|
-
signal
|
|
4500
|
-
};
|
|
4501
|
-
for await (const raw of backend.stream(input, context)) {
|
|
4502
|
-
const event = normalizeBackendStreamEvent(raw, task, session);
|
|
4503
|
-
foldEvent(event, acc);
|
|
4504
|
-
yield event;
|
|
4505
|
-
}
|
|
4506
|
-
}
|
|
4507
|
-
/** Fold one normalized event into the turn accumulator (text + usage).
|
|
4508
|
-
* `fallbackModelLabel` — a mapper-stamped run label to exclude from
|
|
4509
|
-
* `usage.model` (it is not a backend-reported model). */
|
|
4510
|
-
function foldEvent(event, acc, fallbackModelLabel) {
|
|
4511
|
-
if (event.type === "text_delta") {
|
|
4512
|
-
acc.deltaText += event.text;
|
|
4513
|
-
return;
|
|
4514
|
-
}
|
|
4515
|
-
if (event.type === "llm_call") {
|
|
4516
|
-
acc.input += event.tokensIn ?? 0;
|
|
4517
|
-
acc.output += event.tokensOut ?? 0;
|
|
4518
|
-
acc.costUsd += event.costUsd ?? 0;
|
|
4519
|
-
if (event.model && event.model !== fallbackModelLabel) acc.model = event.model;
|
|
4520
|
-
}
|
|
4521
|
-
}
|
|
4522
|
-
/** Read the final text off a terminal sandbox event, when present. */
|
|
4523
|
-
function terminalTextFromSandboxEvent(event) {
|
|
4524
|
-
if (!event || typeof event !== "object") return void 0;
|
|
4525
|
-
const type = String(event.type ?? "");
|
|
4526
|
-
if (type !== "result" && type !== "done" && type !== "final") return void 0;
|
|
4527
|
-
const data = event.data && typeof event.data === "object" ? event.data : {};
|
|
4528
|
-
for (const key of [
|
|
4529
|
-
"finalText",
|
|
4530
|
-
"text",
|
|
4531
|
-
"response",
|
|
4532
|
-
"content"
|
|
4533
|
-
]) {
|
|
4534
|
-
const value = data[key];
|
|
4535
|
-
if (typeof value === "string") return value;
|
|
4536
|
-
}
|
|
4537
|
-
}
|
|
4538
|
-
function buildFinalEvent(task, session, acc, outcome) {
|
|
4539
|
-
const finalText = acc.terminalText ?? acc.deltaText;
|
|
4540
|
-
return {
|
|
4541
|
-
type: "final",
|
|
4542
|
-
task,
|
|
4543
|
-
...session ? { session } : {},
|
|
4544
|
-
status: outcome.status,
|
|
4545
|
-
reason: outcome.reason,
|
|
4546
|
-
...finalText ? { text: finalText } : {},
|
|
4547
|
-
metadata: {
|
|
4548
|
-
tokenUsage: {
|
|
4549
|
-
input: acc.input,
|
|
4550
|
-
output: acc.output
|
|
4551
|
-
},
|
|
4552
|
-
...acc.costUsd > 0 ? { costUsd: acc.costUsd } : {},
|
|
4553
|
-
...acc.model ? { model: acc.model } : {}
|
|
4384
|
+
load(workerId) {
|
|
4385
|
+
const messages = sessions.get(workerId);
|
|
4386
|
+
return messages === void 0 ? void 0 : structuredClone(messages);
|
|
4554
4387
|
},
|
|
4555
|
-
|
|
4556
|
-
|
|
4388
|
+
save(workerId, messages) {
|
|
4389
|
+
sessions.set(workerId, structuredClone(messages));
|
|
4390
|
+
}
|
|
4557
4391
|
};
|
|
4558
4392
|
}
|
|
4559
|
-
|
|
4560
|
-
|
|
4561
|
-
|
|
4562
|
-
|
|
4563
|
-
|
|
4564
|
-
|
|
4565
|
-
|
|
4566
|
-
|
|
4567
|
-
|
|
4568
|
-
|
|
4569
|
-
|
|
4570
|
-
|
|
4571
|
-
|
|
4572
|
-
|
|
4573
|
-
|
|
4574
|
-
|
|
4575
|
-
|
|
4576
|
-
|
|
4577
|
-
|
|
4578
|
-
* >=20.3 — the package floor is >=20).
|
|
4579
|
-
*/
|
|
4580
|
-
function deriveTurnSignal(callerSignal, timeoutMs) {
|
|
4581
|
-
const controller = new AbortController();
|
|
4582
|
-
const timer = timeoutMs > 0 ? setTimeout(() => controller.abort(/* @__PURE__ */ new Error(`agent turn timed out after ${timeoutMs}ms`)), timeoutMs) : void 0;
|
|
4583
|
-
if (timer && typeof timer.unref === "function") timer.unref();
|
|
4584
|
-
const onCallerAbort = () => controller.abort(callerSignal?.reason ?? /* @__PURE__ */ new Error("agent turn aborted"));
|
|
4585
|
-
if (callerSignal) if (callerSignal.aborted) onCallerAbort();
|
|
4586
|
-
else callerSignal.addEventListener("abort", onCallerAbort, { once: true });
|
|
4393
|
+
function exactProfile(profile, context) {
|
|
4394
|
+
const parsed = agentProfileSchema.safeParse(profile);
|
|
4395
|
+
if (!parsed.success) throw new ValidationError(`${context}: invalid AgentProfile: ${parsed.error.message}`);
|
|
4396
|
+
assertExecutableAgentProfile(parsed.data, context);
|
|
4397
|
+
return parsed.data;
|
|
4398
|
+
}
|
|
4399
|
+
function initialMessages(opts) {
|
|
4400
|
+
if (!opts.resume) return void 0;
|
|
4401
|
+
if (!opts.sessions) throw new ValidationError("chat transport: a 'resume' spawn needs the session store holding the prior conversation");
|
|
4402
|
+
const prior = opts.sessions.load(opts.resume.ofWorker);
|
|
4403
|
+
if (prior === void 0) throw new ValidationError(`chat transport: no recorded conversation for worker '${opts.resume.ofWorker}'`);
|
|
4404
|
+
return prior;
|
|
4405
|
+
}
|
|
4406
|
+
function executorConfig(opts) {
|
|
4407
|
+
const profile = exactProfile(opts.profile, "chat transport");
|
|
4408
|
+
if (!opts.complete && !opts.url) throw new ValidationError("chat transport: url required unless complete is injected");
|
|
4409
|
+
const tools = opts.tools ?? [];
|
|
4410
|
+
for (const tool of tools) if (!tool.spec.function.name || typeof tool.execute !== "function") throw new ValidationError("chat transport: every tool needs spec.function.name and execute");
|
|
4411
|
+
const resumed = initialMessages(opts);
|
|
4587
4412
|
return {
|
|
4588
|
-
|
|
4589
|
-
|
|
4590
|
-
|
|
4591
|
-
|
|
4413
|
+
profile,
|
|
4414
|
+
config: {
|
|
4415
|
+
backend: "router-tools",
|
|
4416
|
+
routerBaseUrl: opts.url ?? "http://injected.invalid",
|
|
4417
|
+
routerKey: opts.bearer ?? (opts.complete ? "injected-transport" : ""),
|
|
4418
|
+
tools: tools.map((tool) => tool.spec),
|
|
4419
|
+
executeToolCall: async (name, args, task) => {
|
|
4420
|
+
const tool = tools.find((candidate) => candidate.spec.function.name === name);
|
|
4421
|
+
if (!tool) throw new ValidationError(`chat transport: unknown tool ${JSON.stringify(name)}`);
|
|
4422
|
+
return tool.execute(args, task);
|
|
4423
|
+
},
|
|
4424
|
+
...opts.complete ? { complete: opts.complete } : {},
|
|
4425
|
+
...resumed ? { initialMessages: resumed } : {},
|
|
4426
|
+
...opts.sessions && opts.sessionKey ? { onMessages: (messages) => {
|
|
4427
|
+
opts.sessions?.save(opts.sessionKey, messages);
|
|
4428
|
+
} } : {}
|
|
4592
4429
|
}
|
|
4593
4430
|
};
|
|
4594
4431
|
}
|
|
4595
|
-
|
|
4596
|
-
|
|
4597
|
-
|
|
4598
|
-
|
|
4599
|
-
|
|
4600
|
-
*
|
|
4601
|
-
* A topology is PLAIN DATA an agent can author in a few lines: nodes are canonical
|
|
4602
|
-
* `AgentProfile`s (the ONLY way a node is described — no role-builder functions), edges are typed
|
|
4603
|
-
* values carrying versioned {@link PromptHandle} directives, `deliverable` (termination) and
|
|
4604
|
-
* `budget` (one conserved pool) are mandatory. Driver↔worker is the two-node cyclic instance;
|
|
4605
|
-
* "agent 3 analyzes 1 and 2 and reports to 1" is ONE edge, not a framework.
|
|
4606
|
-
*
|
|
4607
|
-
* NOT A SECOND SCHEDULER. `runGraph` is an interpretation layer over what already runs:
|
|
4608
|
-
* `supervise()` is the execution core — the same `supervisorAgent`/`driverAgent` machinery,
|
|
4609
|
-
* `makeWorkerAgent` seam, conserved-pool budget, and deliverable-gated settlement every
|
|
4610
|
-
* supervised run uses. (`runLoop` is a deprecated alias of `runAgentRounds` and is deliberately
|
|
4611
|
-
* NOT the substrate here.) What the graph layer ADDS is exactly what a bespoke driver loop never
|
|
4612
|
-
* had:
|
|
4613
|
-
*
|
|
4614
|
-
* 1. **Node pinning** — a spawn names a node (`profile.name` = node id) and the node's canonical
|
|
4615
|
-
* profile is what runs; a driver cannot smuggle capabilities into a worker it did not define.
|
|
4616
|
-
* 2. **Observable edges** — every delegates/analyzes traversal lands in an EDGE LEDGER
|
|
4617
|
-
* (`delivered | stripped | empty | unpropagated`, with byte counts), in memory on
|
|
4618
|
-
* the result AND as `edge` events in the run journal. The motivating incident: a filter
|
|
4619
|
-
* silently replaced 1,700-char steering with 241 chars of boilerplate for three rounds and
|
|
4620
|
-
* NO artifact said so — an unobservable edge cannot be trusted and its directive cannot be
|
|
4621
|
-
* optimized.
|
|
4622
|
-
* 3. **Directives as data** — edge text lives in the prompt registry (`<surface>/v<n>`), so every
|
|
4623
|
-
* edge is a versioned optimization target, never prose hardcoded in a builder function.
|
|
4624
|
-
* 4. **Per-edge traversal caps** — the cyclic-graph backstop. A delegates edge whose cap is
|
|
4625
|
-
* exhausted REFUSES further traversals (fail loud), so a cycle cannot spin the pool dry.
|
|
4626
|
-
*
|
|
4627
|
-
* ORACLES ARE ENVIRONMENT, NEVER WORKERS. Graders/verifiers must not be spawnable in the graph —
|
|
4628
|
-
* a delegates edge to them leaks the rubric. An `analyzes` edge names its analyst in one of two
|
|
4629
|
-
* forms: a LENS id from the environment's registry (a pure function over trace evidence), or the
|
|
4630
|
-
* id of a graph NODE — a tool-equipped analyst AGENT spawned on each matching settle with the
|
|
4631
|
-
* node's pinned profile, whose settle output IS the findings. Either way the oracle doctrine
|
|
4632
|
-
* holds: an analyst node can never be a delegates target (refused loudly), so no driver can hand
|
|
4633
|
-
* it work, and an id living in both the registry and the nodes is refused as ambiguous.
|
|
4634
|
-
*
|
|
4635
|
-
* @experimental
|
|
4636
|
-
*/
|
|
4637
|
-
/** Default per-edge traversal cap — the cyclic-graph backstop when an edge names none. */
|
|
4638
|
-
const defaultEdgeTraversalCap = 32;
|
|
4639
|
-
/** A delegates edge exhausted its traversal cap and the run produced no winner: the cap, not the
|
|
4640
|
-
* task, ended it. Carries the full evidence so failing loud loses nothing. */
|
|
4641
|
-
var GraphEdgeCapError = class extends Error {
|
|
4642
|
-
exhaustedEdges;
|
|
4643
|
-
ledger;
|
|
4644
|
-
result;
|
|
4645
|
-
constructor(exhaustedEdges, ledger, result) {
|
|
4646
|
-
super(`runGraph: edge traversal cap exhausted on ${exhaustedEdges.join(", ")} and the run delivered no winner — the cap (the cyclic-graph backstop), not the task, ended this run. Raise maxTraversals on the edge or fix the cycle; the full edge ledger and the supervised result ride on this error.`);
|
|
4647
|
-
this.name = "GraphEdgeCapError";
|
|
4648
|
-
this.exhaustedEdges = exhaustedEdges;
|
|
4649
|
-
this.ledger = ledger;
|
|
4650
|
-
this.result = result;
|
|
4651
|
-
}
|
|
4652
|
-
};
|
|
4653
|
-
function edgeId(edge) {
|
|
4654
|
-
return edge.kind === "delegates" ? `delegates:${edge.from}->${edge.to}` : `analyzes:${edge.analyst}:${edge.over.join("+")}->${edge.to}`;
|
|
4655
|
-
}
|
|
4656
|
-
/** Validate the graph and resolve every directive BEFORE any compute is spent — an invalid
|
|
4657
|
-
* topology or an unknown directive is a configuration fault, never a mid-run surprise. */
|
|
4658
|
-
function validateGraph(graph, registry, analysts) {
|
|
4659
|
-
if (!Array.isArray(graph.nodes) || graph.nodes.length === 0) throw new ValidationError("runGraph: graph.nodes must be a non-empty array");
|
|
4660
|
-
if (!Array.isArray(graph.edges) || graph.edges.length === 0) throw new ValidationError("runGraph: graph.edges must be a non-empty array");
|
|
4661
|
-
if (typeof graph.deliverable?.check !== "function") throw new ValidationError("runGraph: graph.deliverable is mandatory (termination oracle)");
|
|
4662
|
-
if (typeof graph.budget !== "object" || graph.budget === null) throw new ValidationError("runGraph: graph.budget is mandatory (the conserved pool)");
|
|
4663
|
-
const byId = /* @__PURE__ */ new Map();
|
|
4664
|
-
for (const node of graph.nodes) {
|
|
4665
|
-
if (typeof node.id !== "string" || node.id.length === 0) throw new ValidationError("runGraph: every node needs a non-empty string id");
|
|
4666
|
-
if (byId.has(node.id)) throw new ValidationError(`runGraph: duplicate node id '${node.id}'`);
|
|
4667
|
-
const parsed = agentProfileSchema.safeParse(node.profile);
|
|
4668
|
-
if (!parsed.success) throw new ValidationError(`runGraph: node '${node.id}' has an invalid AgentProfile: ${parsed.error.message}`);
|
|
4669
|
-
if (node.profile.name !== node.id) throw new ValidationError(`runGraph: node '${node.id}' has profile.name ${JSON.stringify(node.profile.name)} — profile.name IS the node identity (node pinning and analyst routing match on it) and must equal the node id`);
|
|
4670
|
-
byId.set(node.id, node);
|
|
4671
|
-
}
|
|
4672
|
-
const requireNode = (id, where) => {
|
|
4673
|
-
const node = byId.get(id);
|
|
4674
|
-
if (!node) throw new ValidationError(`runGraph: ${where} references unknown node '${id}'`);
|
|
4675
|
-
return node;
|
|
4432
|
+
function buildChatTransportExecutor(opts, context) {
|
|
4433
|
+
const { config, profile } = executorConfig(opts);
|
|
4434
|
+
const spec = {
|
|
4435
|
+
profile,
|
|
4436
|
+
harness: null
|
|
4676
4437
|
};
|
|
4677
|
-
|
|
4678
|
-
|
|
4679
|
-
|
|
4680
|
-
|
|
4681
|
-
|
|
4682
|
-
|
|
4683
|
-
|
|
4684
|
-
|
|
4685
|
-
|
|
4686
|
-
|
|
4687
|
-
|
|
4688
|
-
|
|
4689
|
-
|
|
4690
|
-
for (const edge of delegates) if (edge.from !== root.id) throw new ValidationError(`runGraph: ${edgeId(edge)} delegates from a non-root node — P0 executes one driver over its workers (the 2-node cyclic case, star-generalized); deeper delegation is P3`);
|
|
4691
|
-
const analystIds = /* @__PURE__ */ new Set();
|
|
4692
|
-
const analystNodes = /* @__PURE__ */ new Map();
|
|
4693
|
-
for (const edge of analyzes) {
|
|
4694
|
-
if (analystIds.has(edge.analyst)) throw new ValidationError(`runGraph: two analyzes edges share analyst '${edge.analyst}' — one analyzes edge per analyst lens (traversals are ledgered by analyst id; a second edge would silently absorb the first's). Register the lens under a second id for a second edge.`);
|
|
4695
|
-
analystIds.add(edge.analyst);
|
|
4696
|
-
const analystNode = byId.get(edge.analyst);
|
|
4697
|
-
const inRegistry = analysts?.kinds.some((kind) => kind.id === edge.analyst) === true;
|
|
4698
|
-
if (analystNode !== void 0 && inRegistry) throw new ValidationError(`runGraph: ${edgeId(edge)} analyst '${edge.analyst}' is BOTH a graph node and a lens in the analysts registry — the id alone distinguishes the two analyst forms, so this is ambiguous; rename the node or register the lens under another id`);
|
|
4699
|
-
if (analystNode !== void 0) {
|
|
4700
|
-
if (analystNode.id === root.id) throw new ValidationError(`runGraph: ${edgeId(edge)} names the ROOT as its analyst — the root is the driver; give the analyst its own node with no delegates edge pointing at it`);
|
|
4701
|
-
if (delegatedTo.has(analystNode.id)) throw new ValidationError(`runGraph: ${edgeId(edge)} names node '${edge.analyst}' as its analyst, but that node is a delegates target — oracle doctrine: an analyst is never delegated to. An analyst NODE is legal only with NO delegates edge pointing at it; give the analyst its own delegates-free node or pass a lens id from RunGraphOptions.analysts.`);
|
|
4702
|
-
analystNodes.set(analystNode.id, analystNode);
|
|
4703
|
-
} else if (!analysts) throw new ValidationError(`runGraph: ${edgeId(edge)} analyst '${edge.analyst}' is not a graph node, and no RunGraphOptions.analysts registry was provided to resolve it as a lens`);
|
|
4704
|
-
else if (!inRegistry) throw new ValidationError(`runGraph: ${edgeId(edge)} analyst '${edge.analyst}' is neither a graph node nor in the analysts registry (known lenses: ${analysts.kinds.map((kind) => kind.id).join(", ") || "none"})`);
|
|
4705
|
-
if (edge.over.length === 0) throw new ValidationError(`runGraph: ${edgeId(edge)} must analyze at least one node`);
|
|
4706
|
-
for (const over of edge.over) requireNode(over, edgeId(edge));
|
|
4707
|
-
requireNode(edge.to, edgeId(edge));
|
|
4708
|
-
}
|
|
4709
|
-
for (const edge of analyzes) for (const over of edge.over) if (analystNodes.has(over)) throw new ValidationError(`runGraph: ${edgeId(edge)} analyzes '${over}', which is an analyst node — an analyst run settles as a finding, never as a worker, so this edge would silently never fire; analyst nodes are not analyzable`);
|
|
4710
|
-
const workers = /* @__PURE__ */ new Map();
|
|
4711
|
-
const delegatesByWorker = /* @__PURE__ */ new Map();
|
|
4712
|
-
for (const edge of delegates) {
|
|
4713
|
-
if (delegatesByWorker.has(edge.to)) throw new ValidationError(`runGraph: node '${edge.to}' is the target of two delegates edges — one delegation directive per worker node (version the directive instead of forking the edge)`);
|
|
4714
|
-
delegatesByWorker.set(edge.to, edge);
|
|
4715
|
-
workers.set(edge.to, requireNode(edge.to, edgeId(edge)));
|
|
4716
|
-
}
|
|
4717
|
-
for (const node of graph.nodes) if (node.id !== root.id && !workers.has(node.id) && !analystNodes.has(node.id)) throw new ValidationError(`runGraph: node '${node.id}' has no delegates edge to it — an unreachable node never runs`);
|
|
4718
|
-
return {
|
|
4719
|
-
root,
|
|
4720
|
-
workers,
|
|
4721
|
-
delegatesByWorker,
|
|
4722
|
-
analyzes,
|
|
4723
|
-
analystNodes
|
|
4724
|
-
};
|
|
4725
|
-
}
|
|
4726
|
-
const byteLength = (text) => Buffer.byteLength(text, "utf8");
|
|
4727
|
-
function stringifyPayload(payload) {
|
|
4728
|
-
if (typeof payload === "string") return payload;
|
|
4729
|
-
try {
|
|
4730
|
-
return JSON.stringify(payload) ?? String(payload);
|
|
4731
|
-
} catch {
|
|
4732
|
-
return String(payload);
|
|
4733
|
-
}
|
|
4438
|
+
return mapExecutorResult(createExecutor(config)(spec, context), (result) => {
|
|
4439
|
+
const raw = result.out;
|
|
4440
|
+
const content = typeof raw?.content === "string" ? raw.content : "";
|
|
4441
|
+
return {
|
|
4442
|
+
outRef: contentAddress({
|
|
4443
|
+
kind: "chat-transport",
|
|
4444
|
+
profile,
|
|
4445
|
+
content
|
|
4446
|
+
}),
|
|
4447
|
+
out: content,
|
|
4448
|
+
...result.verdict ? { verdict: result.verdict } : {}
|
|
4449
|
+
};
|
|
4450
|
+
});
|
|
4734
4451
|
}
|
|
4735
4452
|
/**
|
|
4736
|
-
*
|
|
4737
|
-
*
|
|
4738
|
-
* `profile: { name: '<node id>' }`; the node's canonical profile is pinned by the graph), each
|
|
4739
|
-
* delegates directive is appended to the worker profile's `prompt.instructions` per traversal,
|
|
4740
|
-
* and each analyzes edge becomes an analyst-on-settle route with a real DESTINATION. Every
|
|
4741
|
-
* traversal is ledgered and journaled.
|
|
4453
|
+
* Build one exact profile-driven chat executor through `createExecutor`.
|
|
4454
|
+
* Prefer `chatWorkerSeam` for supervised work because it supplies trusted node identity.
|
|
4742
4455
|
*/
|
|
4743
|
-
function
|
|
4744
|
-
|
|
4745
|
-
|
|
4746
|
-
|
|
4747
|
-
const journal = opts.journal ?? new InMemorySpawnJournal();
|
|
4748
|
-
const blobs = opts.blobs ?? new InMemoryResultBlobStore();
|
|
4749
|
-
const runId = opts.runId ?? `graph-${canonicalCandidateDigest(graph.nodes.map((n) => n.id)).slice(7, 19)}`;
|
|
4750
|
-
const now = opts.now ?? Date.now;
|
|
4751
|
-
const ledger = [];
|
|
4752
|
-
const journaled = /* @__PURE__ */ new Set();
|
|
4753
|
-
const traversalCounts = /* @__PURE__ */ new Map();
|
|
4754
|
-
const exhausted = /* @__PURE__ */ new Set();
|
|
4755
|
-
const exhaustedDelegates = /* @__PURE__ */ new Set();
|
|
4756
|
-
const journalWrites = [];
|
|
4757
|
-
let ledgerSeq = 0;
|
|
4758
|
-
const appendJournal = (entry, nodeIdForEvent) => {
|
|
4759
|
-
if (journaled.has(entry)) return Promise.resolve();
|
|
4760
|
-
journaled.add(entry);
|
|
4761
|
-
const write = journal.appendEvent(runId, {
|
|
4762
|
-
kind: "edge",
|
|
4763
|
-
id: nodeIdForEvent,
|
|
4764
|
-
edge: {
|
|
4765
|
-
kind: entry.kind,
|
|
4766
|
-
from: entry.from,
|
|
4767
|
-
to: entry.to,
|
|
4768
|
-
directive: entry.directive
|
|
4769
|
-
},
|
|
4770
|
-
traversal: entry.traversal,
|
|
4771
|
-
outcome: entry.outcome,
|
|
4772
|
-
bytes: entry.bytes,
|
|
4773
|
-
...entry.reason !== void 0 ? { reason: entry.reason } : {},
|
|
4774
|
-
seq: ledgerSeq++,
|
|
4775
|
-
at: new Date(now()).toISOString()
|
|
4776
|
-
});
|
|
4777
|
-
journalWrites.push(write);
|
|
4778
|
-
return write;
|
|
4779
|
-
};
|
|
4780
|
-
const record = (entry, journalNow) => {
|
|
4781
|
-
const count = (traversalCounts.get(entry.edge) ?? 0) + 1;
|
|
4782
|
-
traversalCounts.set(entry.edge, count);
|
|
4783
|
-
const row = {
|
|
4784
|
-
...entry,
|
|
4785
|
-
traversal: count
|
|
4786
|
-
};
|
|
4787
|
-
ledger.push(row);
|
|
4788
|
-
if (journalNow) appendJournal(row, row.workerId ?? `graph:${row.to}`);
|
|
4789
|
-
return row;
|
|
4790
|
-
};
|
|
4791
|
-
const makeLeaf = opts.makeWorkerAgent ?? workerFromBackend(opts.backend, graph.deliverable);
|
|
4792
|
-
const nodeByWorkerId = /* @__PURE__ */ new Map();
|
|
4793
|
-
const pendingByAssignment = /* @__PURE__ */ new Map();
|
|
4794
|
-
const graphWorker = (authoredProfile, spawnContext) => {
|
|
4795
|
-
const requested = typeof authoredProfile?.name === "string" ? authoredProfile.name : void 0;
|
|
4796
|
-
if (spawnContext?.analyst !== void 0) {
|
|
4797
|
-
const analystNode = analystNodes.get(spawnContext.analyst);
|
|
4798
|
-
if (!analystNode || requested !== analystNode.id) throw new ValidationError(`runGraph: analyst run for ${JSON.stringify(spawnContext.analyst)} does not name an analyst node of this graph (analyst nodes: ${[...analystNodes.keys()].join(", ") || "none"})`);
|
|
4799
|
-
return makeLeaf(analystNode.profile, spawnContext);
|
|
4800
|
-
}
|
|
4801
|
-
const node = requested !== void 0 ? workers.get(requested) : void 0;
|
|
4802
|
-
if (!node) throw new ValidationError(`runGraph: spawn_agent named profile ${JSON.stringify(requested)} which is not a worker node of this graph (nodes: ${[...workers.keys()].join(", ")}). Spawn by node id: profile.name selects the node; the node profile itself is pinned by the graph.`);
|
|
4803
|
-
const edge = delegatesByWorker.get(node.id);
|
|
4804
|
-
const id = edgeId(edge);
|
|
4805
|
-
const cap = edge.maxTraversals ?? 32;
|
|
4806
|
-
if ((traversalCounts.get(id) ?? 0) >= cap) {
|
|
4807
|
-
exhausted.add(id);
|
|
4808
|
-
exhaustedDelegates.add(id);
|
|
4809
|
-
record({
|
|
4810
|
-
edge: id,
|
|
4811
|
-
kind: "delegates",
|
|
4812
|
-
from: edge.from,
|
|
4813
|
-
to: edge.to,
|
|
4814
|
-
directive: formatPromptHandle(edge.directive),
|
|
4815
|
-
outcome: "unpropagated",
|
|
4816
|
-
bytes: 0,
|
|
4817
|
-
reason: `traversal-cap-exhausted (max ${cap})`
|
|
4818
|
-
}, true);
|
|
4819
|
-
throw new ValidationError(`runGraph: delegates edge ${id} exhausted its traversal cap (${cap}) — the cyclic-graph backstop refused this spawn`);
|
|
4820
|
-
}
|
|
4821
|
-
const directiveText = registry.resolve(edge.directive).text;
|
|
4822
|
-
const taskText = stringifyPayload(spawnContext?.task);
|
|
4823
|
-
const bytes = byteLength(directiveText) + byteLength(taskText);
|
|
4824
|
-
const row = record({
|
|
4825
|
-
edge: id,
|
|
4826
|
-
kind: "delegates",
|
|
4827
|
-
from: edge.from,
|
|
4828
|
-
to: edge.to,
|
|
4829
|
-
directive: formatPromptHandle(edge.directive),
|
|
4830
|
-
outcome: bytes === 0 ? "empty" : "delivered",
|
|
4831
|
-
bytes,
|
|
4832
|
-
...bytes === 0 ? { reason: "no directive text and no task payload" } : {}
|
|
4833
|
-
}, false);
|
|
4834
|
-
if (spawnContext?.assignmentId !== void 0) pendingByAssignment.set(spawnContext.assignmentId, row);
|
|
4835
|
-
else appendJournal(row, `graph:${row.to}`);
|
|
4836
|
-
const pinned = directiveText.length === 0 ? node.profile : {
|
|
4837
|
-
...node.profile,
|
|
4838
|
-
prompt: {
|
|
4839
|
-
...node.profile.prompt ?? {},
|
|
4840
|
-
instructions: [...node.profile.prompt?.instructions ?? [], directiveText]
|
|
4841
|
-
}
|
|
4842
|
-
};
|
|
4843
|
-
return makeLeaf(pinned, spawnContext);
|
|
4844
|
-
};
|
|
4845
|
-
const routes = analyzes.map((edge) => {
|
|
4846
|
-
const analystNode = analystNodes.get(edge.analyst);
|
|
4847
|
-
if (analystNode) return {
|
|
4848
|
-
kind: edge.analyst,
|
|
4849
|
-
over: edge.over,
|
|
4850
|
-
agent: analystNode.profile,
|
|
4851
|
-
directive: registry.resolve(edge.directive).text,
|
|
4852
|
-
...edge.to === root.id ? {} : { to: edge.to }
|
|
4853
|
-
};
|
|
4854
|
-
return edge.to === root.id ? {
|
|
4855
|
-
kind: edge.analyst,
|
|
4856
|
-
over: edge.over
|
|
4857
|
-
} : {
|
|
4858
|
-
kind: edge.analyst,
|
|
4859
|
-
over: edge.over,
|
|
4860
|
-
to: edge.to,
|
|
4861
|
-
directive: registry.resolve(edge.directive).text
|
|
4862
|
-
};
|
|
4456
|
+
function chatTransportExecutor(opts) {
|
|
4457
|
+
return buildChatTransportExecutor(opts, {
|
|
4458
|
+
signal: new AbortController().signal,
|
|
4459
|
+
seams: {}
|
|
4863
4460
|
});
|
|
4864
|
-
|
|
4865
|
-
|
|
4866
|
-
|
|
4867
|
-
|
|
4868
|
-
|
|
4869
|
-
|
|
4870
|
-
|
|
4871
|
-
const description = typeof node.profile.description === "string" && node.profile.description.length > 0 ? ` — ${node.profile.description}` : "";
|
|
4872
|
-
return `- '${node.id}'${description} (delegation cap: ${cap} traversals)`;
|
|
4873
|
-
}),
|
|
4874
|
-
...driverAnalyzesBriefs.length > 0 ? ["", ...driverAnalyzesBriefs] : []
|
|
4875
|
-
].join("\n");
|
|
4876
|
-
const rootProfile = {
|
|
4877
|
-
...root.profile,
|
|
4878
|
-
prompt: {
|
|
4879
|
-
...root.profile.prompt ?? {},
|
|
4880
|
-
instructions: [...root.profile.prompt?.instructions ?? [], graphBrief]
|
|
4881
|
-
}
|
|
4882
|
-
};
|
|
4883
|
-
const strippedByDigest = /* @__PURE__ */ new Map();
|
|
4884
|
-
const authorizeMessage = opts.authorizeMessage ? (input) => {
|
|
4885
|
-
const decision = opts.authorizeMessage(input);
|
|
4886
|
-
if (decision.instruction !== input.instruction) strippedByDigest.set(canonicalCandidateDigest(decision.instruction), { composedBytes: byteLength(input.instruction) });
|
|
4887
|
-
return decision;
|
|
4888
|
-
} : void 0;
|
|
4889
|
-
const routedAnalyzesByAnalyst = /* @__PURE__ */ new Map();
|
|
4890
|
-
const driverAnalyzesByAnalyst = /* @__PURE__ */ new Map();
|
|
4891
|
-
for (const edge of analyzes) (edge.to === root.id ? driverAnalyzesByAnalyst : routedAnalyzesByAnalyst).set(edge.analyst, edge);
|
|
4892
|
-
const analyzesCapReached = (edge) => {
|
|
4893
|
-
const cap = edge.maxTraversals ?? 32;
|
|
4894
|
-
if ((traversalCounts.get(edgeId(edge)) ?? 0) < cap) return false;
|
|
4895
|
-
exhausted.add(edgeId(edge));
|
|
4896
|
-
return true;
|
|
4897
|
-
};
|
|
4898
|
-
const ledgerAnalyzes = (edge, outcome, bytes, reason, workerId) => {
|
|
4899
|
-
const capped = analyzesCapReached(edge);
|
|
4900
|
-
record({
|
|
4901
|
-
edge: edgeId(edge),
|
|
4902
|
-
kind: "analyzes",
|
|
4903
|
-
from: edge.over.join("+"),
|
|
4904
|
-
to: edge.to,
|
|
4905
|
-
directive: formatPromptHandle(edge.directive),
|
|
4906
|
-
outcome: capped ? "unpropagated" : outcome,
|
|
4907
|
-
bytes,
|
|
4908
|
-
...capped ? { reason: `traversal-cap-exhausted (max ${edge.maxTraversals ?? 32})` } : reason !== void 0 ? { reason } : {},
|
|
4909
|
-
...workerId !== void 0 ? { workerId } : {}
|
|
4910
|
-
}, true);
|
|
4911
|
-
};
|
|
4912
|
-
const onCoordinationEvent = async (_context, _eventId, recordEnvelope) => {
|
|
4913
|
-
const event = recordEnvelope.event;
|
|
4914
|
-
if (event.type === "finding") {
|
|
4915
|
-
const edge = driverAnalyzesByAnalyst.get(event.finding.analyst);
|
|
4916
|
-
if (!edge) return;
|
|
4917
|
-
const sourceNode = nodeByWorkerId.get(event.finding.fromWorker);
|
|
4918
|
-
if (sourceNode === void 0 || !edge.over.includes(sourceNode)) return;
|
|
4919
|
-
const findingsText = event.finding.findings === void 0 ? "" : stringifyPayload(event.finding.findings);
|
|
4920
|
-
const directiveBytes = byteLength(registry.resolve(edge.directive).text);
|
|
4921
|
-
const empty = findingsText.length === 0;
|
|
4922
|
-
ledgerAnalyzes(edge, empty ? "empty" : "delivered", directiveBytes + byteLength(findingsText), empty ? "analyst returned no findings" : void 0, event.finding.fromWorker);
|
|
4923
|
-
return;
|
|
4924
|
-
}
|
|
4925
|
-
if (event.type === "steer") {
|
|
4926
|
-
const down = event.down;
|
|
4927
|
-
if (event.analyst !== void 0) {
|
|
4928
|
-
const edge = routedAnalyzesByAnalyst.get(event.analyst);
|
|
4929
|
-
if (!edge) return;
|
|
4930
|
-
ledgerAnalyzes(edge, down.delivered ? "delivered" : "unpropagated", byteLength(down.instruction), down.delivered ? void 0 : down.outcome, down.toWorker);
|
|
4931
|
-
return;
|
|
4932
|
-
}
|
|
4933
|
-
const nodeId = nodeByWorkerId.get(down.toWorker);
|
|
4934
|
-
if (nodeId === void 0) return;
|
|
4935
|
-
const edge = delegatesByWorker.get(nodeId);
|
|
4936
|
-
if (!edge) return;
|
|
4937
|
-
const stripped = strippedByDigest.get(down.instructionDigest);
|
|
4938
|
-
record({
|
|
4939
|
-
edge: edgeId(edge),
|
|
4940
|
-
kind: "delegates",
|
|
4941
|
-
from: edge.from,
|
|
4942
|
-
to: edge.to,
|
|
4943
|
-
directive: formatPromptHandle(edge.directive),
|
|
4944
|
-
outcome: !down.delivered ? "unpropagated" : stripped ? "stripped" : "delivered",
|
|
4945
|
-
bytes: byteLength(down.instruction),
|
|
4946
|
-
...!down.delivered ? { reason: down.outcome } : stripped ? { reason: `authorization narrowed ${stripped.composedBytes} composed bytes` } : {},
|
|
4947
|
-
workerId: down.toWorker
|
|
4948
|
-
}, true);
|
|
4949
|
-
}
|
|
4950
|
-
};
|
|
4951
|
-
const hooks = composeRuntimeHooks({ onEvent: (event) => {
|
|
4952
|
-
if (event.target !== "agent.spawn" || event.phase !== "after") return;
|
|
4953
|
-
const payload = event.payload;
|
|
4954
|
-
if (typeof payload?.childId !== "string" || typeof payload.assignmentId !== "string") return;
|
|
4955
|
-
const pending = pendingByAssignment.get(payload.assignmentId);
|
|
4956
|
-
if (!pending) return;
|
|
4957
|
-
pendingByAssignment.delete(payload.assignmentId);
|
|
4958
|
-
const bound = {
|
|
4959
|
-
...pending,
|
|
4960
|
-
workerId: payload.childId
|
|
4961
|
-
};
|
|
4962
|
-
ledger[ledger.indexOf(pending)] = bound;
|
|
4963
|
-
nodeByWorkerId.set(payload.childId, bound.to);
|
|
4964
|
-
return appendJournal(bound, payload.childId);
|
|
4965
|
-
} }, opts.hooks);
|
|
4966
|
-
const start = async () => {
|
|
4967
|
-
const result = await supervise(rootProfile, graphTask(graph, root), {
|
|
4968
|
-
budget: graph.budget,
|
|
4969
|
-
deliverable: graph.deliverable,
|
|
4970
|
-
makeWorkerAgent: graphWorker,
|
|
4971
|
-
journal,
|
|
4972
|
-
blobs,
|
|
4973
|
-
runId,
|
|
4974
|
-
hooks,
|
|
4975
|
-
onCoordinationEvent,
|
|
4976
|
-
...routes.length > 0 ? {
|
|
4977
|
-
analyzeOnSettle: routes,
|
|
4978
|
-
...opts.analysts ? { analysts: opts.analysts } : {}
|
|
4979
|
-
} : {},
|
|
4980
|
-
...opts.watchWorkers ? { watchWorkers: opts.watchWorkers } : {},
|
|
4981
|
-
...opts.router ? { router: opts.router } : {},
|
|
4982
|
-
...opts.brain ? { brain: opts.brain } : {},
|
|
4983
|
-
...authorizeMessage ? { authorizeMessage } : {},
|
|
4984
|
-
...opts.perWorker ? { perWorker: opts.perWorker } : {},
|
|
4985
|
-
...opts.maxTurns !== void 0 ? { maxTurns: opts.maxTurns } : {},
|
|
4986
|
-
...opts.maxLiveWorkers !== void 0 ? { maxLiveWorkers: opts.maxLiveWorkers } : {},
|
|
4987
|
-
...opts.signal ? { signal: opts.signal } : {},
|
|
4988
|
-
...opts.now ? { now: opts.now } : {},
|
|
4989
|
-
...opts.otel ? { otel: opts.otel } : {},
|
|
4990
|
-
...opts.stallAfterMs !== void 0 ? { stallAfterMs: opts.stallAfterMs } : {},
|
|
4991
|
-
...opts.allowedModels ? { allowedModels: opts.allowedModels } : {}
|
|
4992
|
-
});
|
|
4993
|
-
for (const pending of pendingByAssignment.values()) {
|
|
4994
|
-
const refused = {
|
|
4995
|
-
...pending,
|
|
4996
|
-
outcome: "unpropagated",
|
|
4997
|
-
bytes: 0,
|
|
4998
|
-
reason: `no-live-worker-bound (spawn refused after the factory, or a keyed re-spawn deduplicated to a completed result; ${pending.bytes} composed bytes never crossed)`
|
|
4999
|
-
};
|
|
5000
|
-
ledger[ledger.indexOf(pending)] = refused;
|
|
5001
|
-
await appendJournal(refused, `graph:${refused.to}`);
|
|
5002
|
-
}
|
|
5003
|
-
pendingByAssignment.clear();
|
|
5004
|
-
await Promise.all(journalWrites);
|
|
5005
|
-
const exhaustedEdges = Object.freeze([...exhausted]);
|
|
5006
|
-
const frozenLedger = Object.freeze(ledger.map((row) => Object.freeze({ ...row })));
|
|
5007
|
-
const lifecycleEnded = result.kind === "no-winner" && (result.reason === "aborted" || result.reason === "budget-exhausted");
|
|
5008
|
-
if (result.kind !== "winner" && !lifecycleEnded && exhaustedDelegates.size > 0) throw new GraphEdgeCapError(Object.freeze([...exhaustedDelegates]), frozenLedger, result);
|
|
4461
|
+
}
|
|
4462
|
+
/** Session-owning worker factory for graph continuity. */
|
|
4463
|
+
function chatWorkerSeam(opts) {
|
|
4464
|
+
if (!opts.complete && !opts.url) throw new ValidationError("chatWorkerSeam: url required unless complete is injected");
|
|
4465
|
+
const sessions = opts.sessions ?? createChatSessionStore();
|
|
4466
|
+
return (rawProfile, spawnContext) => {
|
|
4467
|
+
const profile = exactProfile(rawProfile, "chatWorkerSeam");
|
|
5009
4468
|
return {
|
|
5010
|
-
|
|
5011
|
-
|
|
5012
|
-
|
|
5013
|
-
|
|
4469
|
+
name: profile.name ?? "chat-worker",
|
|
4470
|
+
act: async () => void 0,
|
|
4471
|
+
executorSpec: {
|
|
4472
|
+
profile,
|
|
4473
|
+
harness: null,
|
|
4474
|
+
executorFactory: (executorSpec, context) => {
|
|
4475
|
+
const executor = buildChatTransportExecutor({
|
|
4476
|
+
profile: executorSpec.profile,
|
|
4477
|
+
...opts.url ? { url: opts.url } : {},
|
|
4478
|
+
...opts.bearer ? { bearer: opts.bearer } : {},
|
|
4479
|
+
...opts.tools ? { tools: opts.tools } : {},
|
|
4480
|
+
...opts.complete ? { complete: opts.complete } : {},
|
|
4481
|
+
sessions,
|
|
4482
|
+
...context.node?.nodeId ? { sessionKey: context.node.nodeId } : {},
|
|
4483
|
+
...spawnContext?.resume ? { resume: spawnContext.resume } : {}
|
|
4484
|
+
}, context);
|
|
4485
|
+
return opts.deliverable ? gateOnDeliverable(executor, opts.deliverable) : executor;
|
|
4486
|
+
}
|
|
4487
|
+
}
|
|
5014
4488
|
};
|
|
5015
4489
|
};
|
|
5016
|
-
return start();
|
|
5017
|
-
}
|
|
5018
|
-
/** The root task: the graph's own framing. The deliverable (mandatory) is the termination; the
|
|
5019
|
-
* task names what the topology exists to produce. */
|
|
5020
|
-
function graphTask(graph, root) {
|
|
5021
|
-
return graph.deliverable.describe ?? `Deliver the graph's deliverable by driving your worker nodes (root: '${root.id}').`;
|
|
5022
4490
|
}
|
|
5023
4491
|
//#endregion
|
|
5024
4492
|
//#region src/runtime/supervise/patch-checks.ts
|
|
@@ -5459,7 +4927,6 @@ function worktreeFanout(options) {
|
|
|
5459
4927
|
return gateOnDeliverable(createWorktreeCliExecutor({
|
|
5460
4928
|
repoRoot: options.repoRoot,
|
|
5461
4929
|
profile: item.profile,
|
|
5462
|
-
harness: item.harness,
|
|
5463
4930
|
taskPrompt: options.taskPrompt,
|
|
5464
4931
|
executionAttemptId: ctx.node.attemptId,
|
|
5465
4932
|
...item.budgetExempt !== void 0 ? { budgetExempt: item.budgetExempt } : {},
|
|
@@ -5632,7 +5099,7 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
|
|
|
5632
5099
|
const guidance = typeof brief === "string" ? brief.trim() : brief ? JSON.stringify(brief) : "";
|
|
5633
5100
|
const attemptTask = guidance ? {
|
|
5634
5101
|
...task,
|
|
5635
|
-
|
|
5102
|
+
userPrompt: `${task.userPrompt}\n\n— Supervisor guidance for THIS attempt (incorporate it; do not just repeat a prior approach) —\n${guidance}`
|
|
5636
5103
|
} : task;
|
|
5637
5104
|
const r = await runAgentic({
|
|
5638
5105
|
surface: traced.surface,
|
|
@@ -5641,8 +5108,8 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
|
|
|
5641
5108
|
budget: worker.budget ?? 1,
|
|
5642
5109
|
routerBaseUrl: worker.routerBaseUrl,
|
|
5643
5110
|
routerKey: worker.routerKey,
|
|
5644
|
-
|
|
5645
|
-
...worker.
|
|
5111
|
+
workerProfile: worker.profile,
|
|
5112
|
+
...worker.analystProfile ? { analystProfile: worker.analystProfile } : {},
|
|
5646
5113
|
...worker.innerTurns !== void 0 ? { innerTurns: worker.innerTurns } : {}
|
|
5647
5114
|
});
|
|
5648
5115
|
const out = {
|
|
@@ -5655,7 +5122,9 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
|
|
|
5655
5122
|
const spent = {
|
|
5656
5123
|
iterations: r.completions,
|
|
5657
5124
|
tokens: r.tokens,
|
|
5125
|
+
...r.tokensKnown ? {} : { tokensKnown: false },
|
|
5658
5126
|
usd: r.usd,
|
|
5127
|
+
...r.usdKnown ? {} : { usdKnown: false },
|
|
5659
5128
|
ms: r.ms
|
|
5660
5129
|
};
|
|
5661
5130
|
artifact = {
|
|
@@ -5685,13 +5154,13 @@ async function superviseSurface(profile, task, opts) {
|
|
|
5685
5154
|
const innerTurns = opts.worker.innerTurns ?? 6;
|
|
5686
5155
|
const router = opts.router ?? {
|
|
5687
5156
|
routerBaseUrl: opts.worker.routerBaseUrl,
|
|
5688
|
-
routerKey: opts.worker.routerKey
|
|
5689
|
-
model: opts.worker.model
|
|
5157
|
+
routerKey: opts.worker.routerKey
|
|
5690
5158
|
};
|
|
5691
5159
|
const budget = opts.budget ?? {
|
|
5692
5160
|
maxIterations: (innerTurns + 2) * 5 + 16,
|
|
5693
|
-
maxTokens:
|
|
5161
|
+
maxTokens: 1e9
|
|
5694
5162
|
};
|
|
5163
|
+
const workerMaxTokens = profileMaxTokens(opts.worker.profile) ?? Math.max(1, Math.floor(budget.maxTokens / 8));
|
|
5695
5164
|
const makeWorkerAgent = (rawProfile) => {
|
|
5696
5165
|
const p = rawProfile ?? {};
|
|
5697
5166
|
return {
|
|
@@ -5716,7 +5185,7 @@ async function superviseSurface(profile, task, opts) {
|
|
|
5716
5185
|
maxLiveWorkers: opts.maxLiveWorkers ?? 1,
|
|
5717
5186
|
perWorker: {
|
|
5718
5187
|
maxIterations: innerTurns + 2,
|
|
5719
|
-
maxTokens:
|
|
5188
|
+
maxTokens: workerMaxTokens
|
|
5720
5189
|
},
|
|
5721
5190
|
router,
|
|
5722
5191
|
...analysts ? {
|
|
@@ -5736,6 +5205,12 @@ async function superviseSurface(profile, task, opts) {
|
|
|
5736
5205
|
completions: sp.iterations
|
|
5737
5206
|
};
|
|
5738
5207
|
}
|
|
5208
|
+
function profileMaxTokens(profile) {
|
|
5209
|
+
const value = profile.model?.metadata?.maxTokens;
|
|
5210
|
+
if (value === void 0) return void 0;
|
|
5211
|
+
if (!Number.isSafeInteger(value) || value < 1) throw new Error("superviseSurface: AgentProfile.model.metadata.maxTokens must be a positive safe integer");
|
|
5212
|
+
return value;
|
|
5213
|
+
}
|
|
5739
5214
|
//#endregion
|
|
5740
5215
|
//#region src/runtime/verifier-environment.ts
|
|
5741
5216
|
const submitTool = {
|
|
@@ -6150,6 +5625,6 @@ function tail(s) {
|
|
|
6150
5625
|
return s.slice(-400);
|
|
6151
5626
|
}
|
|
6152
5627
|
//#endregion
|
|
6153
|
-
export {
|
|
5628
|
+
export { pipeline as $, assertStrategyContract as A, createMcpEnvironment as At, trajectoryReport as B, chatTransportExecutor as C, renderLeaderboardSvg as Ct, pickChampion as D, McpSpawnFault as Dt, discriminatingMeans as E, defaultAuditorInstruction as Et, openSandboxRun as F, resolveSecretEnv as Ft, registerShape as G, runPersonified as H, printBenchmarkReport as I, secretEnvOfMcpServer as It, renderCorpusToInstructions as J, FileCorpus as K, runBenchmark as L, strategyAuthorContract as M, envKeyProvider as Mt, strategyAuthorSystemPrompt as N, mcpSecretEnvMetadataKey as Nt, runStrategyEvolution as O, connectStdioMcp as Ot, SandboxRunAbortError as P, resolveMcpServerLaunch as Pt, panel as Q, promotionGate as R, runCoderChecks as S, renderLeaderboardMarkdown as St, createChatSessionStore as T, auditIntent as Tt, builtinShapes as U, definePersona as V, createShapeRegistry as W, flatWidenGate as X, fanout as Y, loopUntil as Z, settledWorkerOut as _, sentinelCompletion as _t, localShell as a, createScopeAnalyst as at, analyzeTrace as b, pairwiseSignificance as bt, createVerifierEnvironment as c, harvestCorpus as ct, worktreeFanout as d, localSandboxClient as dt, selectValidWinner as et, EVIDENCE_MAX_CHARS as f, inlineSandboxClient as ft, composeWorkerEvidence as g, deterministicCompletion as gt, closingWorkerNote as h, completionAuthorizes as ht, jjWorkspace as i, buildSteerContext as it, authorStrategy as j, sanitizeMcpToolSchema as jt, selectChampion as k, materializeLocalMcp as kt, failuresAnalyst as l, defineLeaderboard as lt, VERIFY_TAIL_CHARS as m, loopDispatch as mt, makeFinding$1 as n, widen as nt, runInWorkspace as o, registryScopeAnalyst as ot, NOTE_MAX_CHARS as p, loopCampaignDispatch as pt, InMemoryCorpus as q, gitWorkspace as r, assertTraceDerivedFindings as rt, createWaterfallCollector as s, inProcessSandboxClient as st, computeFindingId$1 as t, verify as tt, superviseSurface as u, resolveSandboxClient as ut, copyUntrackedIntoClone as v, stopSentinel as vt, chatWorkerSeam as w, renderPairwiseMarkdown as wt, patchDelivered as x, renderLeaderboardHtml as xt, withUntrackedArtifacts as y, leaderboard as yt, equalKOnCost as z };
|
|
6154
5629
|
|
|
6155
|
-
//# sourceMappingURL=runtime-
|
|
5630
|
+
//# sourceMappingURL=runtime-hiAABiTk.js.map
|