@tangle-network/agent-runtime 0.128.0 → 0.131.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +70 -20
- package/dist/{activation-DhWJ3p8N.js → activation-BCzMOTaV.js} +3 -3
- package/dist/{activation-DhWJ3p8N.js.map → activation-BCzMOTaV.js.map} +1 -1
- package/dist/agent.d.ts +2 -3
- package/dist/agent.js +4 -5
- package/dist/agent.js.map +1 -1
- package/dist/{analyst-loop-DvSciOfB.js → analyst-loop-BE8cDs5Q.js} +2 -2
- package/dist/{analyst-loop-DvSciOfB.js.map → analyst-loop-BE8cDs5Q.js.map} +1 -1
- package/dist/analyst-loop.d.ts +1 -1
- package/dist/analyst-loop.js +1 -1
- package/dist/authoring-CvHwo1oW.js +163 -0
- package/dist/authoring-CvHwo1oW.js.map +1 -0
- package/dist/candidate-execution/index.js +4 -4
- package/dist/{candidate-execution-BFpq-Xi6.js → candidate-execution-BNxKr-Bu.js} +5 -5
- package/dist/{candidate-execution-BFpq-Xi6.js.map → candidate-execution-BNxKr-Bu.js.map} +1 -1
- package/dist/{conversation-BpLQZGPH.js → conversation-DNtxaJ1Z.js} +113 -31
- package/dist/conversation-DNtxaJ1Z.js.map +1 -0
- package/dist/conversation.d.ts +2 -2
- package/dist/conversation.js +2 -2
- package/dist/environment-provider-CxvSd1W6.d.ts +86 -0
- package/dist/{environment-provider-DqFS6FSZ.js → environment-provider-Dyg8DtLK.js} +8 -4
- package/dist/environment-provider-Dyg8DtLK.js.map +1 -0
- package/dist/environment-provider.d.ts +1 -1
- package/dist/environment-provider.js +1 -1
- package/dist/graph-BJTxGOFB.js +471 -0
- package/dist/graph-BJTxGOFB.js.map +1 -0
- package/dist/{improvement-cycle-IJgCbKWQ.js → improvement-cycle-Csp38cWg.js} +133 -224
- package/dist/improvement-cycle-Csp38cWg.js.map +1 -0
- package/dist/{index-BhZhQw77.d.ts → index-CoO7atyo.d.ts} +556 -1278
- package/dist/{index-BhuzfG2r.d.ts → index-DwGtu9nc.d.ts} +7 -9
- package/dist/{index-Efjb3nrQ.d.ts → index-qYHpsmG2.d.ts} +36 -18
- package/dist/index.d.ts +353 -11
- package/dist/index.js +111 -354
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +6 -6
- package/dist/intelligence.js +9 -8
- package/dist/intelligence.js.map +1 -1
- package/dist/kernel.d.ts +7 -5
- package/dist/kernel.js +13 -9
- package/dist/{knowledge-DF63xPr4.js → knowledge-ce0_uKCl.js} +19 -17
- package/dist/knowledge-ce0_uKCl.js.map +1 -0
- package/dist/knowledge.d.ts +1 -1
- package/dist/knowledge.js +1 -1
- package/dist/{loop-runner-bin-Ckp_9tmD.d.ts → loop-runner-bin-BwgZ8m_7.d.ts} +6 -3
- package/dist/{loop-runner-bin-CWqOpCEw.js → loop-runner-bin-DSbuDDqM.js} +5 -27
- package/dist/loop-runner-bin-DSbuDDqM.js.map +1 -0
- package/dist/loop-runner-bin.d.ts +1 -1
- package/dist/loop-runner-bin.js +1 -1
- package/dist/materialization-COJ1UYQ-.js +272 -0
- package/dist/materialization-COJ1UYQ-.js.map +1 -0
- package/dist/mcp/bin.js +39 -47
- package/dist/mcp/bin.js.map +1 -1
- package/dist/mcp/index.d.ts +24 -26
- package/dist/mcp/index.js +66 -83
- package/dist/mcp/index.js.map +1 -1
- package/dist/mcp/memory-bin.js +1 -1
- package/dist/{memory-server-DL6cE2Ag.js → memory-server-5HEJH672.js} +2 -2
- package/dist/{memory-server-DL6cE2Ag.js.map → memory-server-5HEJH672.js.map} +1 -1
- package/dist/model-policy-CqziaqS1.js +232 -0
- package/dist/model-policy-CqziaqS1.js.map +1 -0
- package/dist/{openai-tools-B68JaOCx.d.ts → openai-tools-D3oMyVGl.d.ts} +2 -2
- package/dist/{openai-tools-D3XfrrQ6.js → openai-tools-ru75mLjq.js} +2 -2
- package/dist/openai-tools-ru75mLjq.js.map +1 -0
- package/dist/{prepare--8EvLqCr.js → prepare-DYWjVcPx.js} +169 -153
- package/dist/prepare-DYWjVcPx.js.map +1 -0
- package/dist/primeintellect/index.d.ts +7 -6
- package/dist/primeintellect/index.js +9 -11
- package/dist/primeintellect/index.js.map +1 -1
- package/dist/profiles.d.ts +21 -174
- package/dist/profiles.js +67 -276
- package/dist/profiles.js.map +1 -1
- package/dist/{protected-model-port-CXVfOUu_.js → protected-model-port-48ALLGxT.js} +2 -2
- package/dist/{protected-model-port-CXVfOUu_.js.map → protected-model-port-48ALLGxT.js.map} +1 -1
- package/dist/{redact-PQzmE1Jn.d.ts → redact-9_Gf-8m3.d.ts} +58 -69
- package/dist/{researcher-CoVqNhfI.js → researcher-Skz5-Uc8.js} +50 -26
- package/dist/researcher-Skz5-Uc8.js.map +1 -0
- package/dist/{run-layout-C2FGmZ3v.js → run-layout-CeJsEAom.js} +2 -2
- package/dist/{run-layout-C2FGmZ3v.js.map → run-layout-CeJsEAom.js.map} +1 -1
- package/dist/runtime-D-QfLbSd.d.ts +893 -0
- package/dist/{runtime-5uDVVfER.js → runtime-hiAABiTk.js} +315 -1191
- package/dist/runtime-hiAABiTk.js.map +1 -0
- package/dist/{sandbox-events-Yhd1GYWl.js → sandbox-events-CRDwc5WN.js} +42 -5
- package/dist/sandbox-events-CRDwc5WN.js.map +1 -0
- package/dist/snapshot-CXiiuHhL.js +21 -0
- package/dist/snapshot-CXiiuHhL.js.map +1 -0
- package/dist/{spawn-journal-DsZKDqeh.js → spawn-journal-saHQzqYi.js} +15 -21
- package/dist/spawn-journal-saHQzqYi.js.map +1 -0
- package/dist/stream-agent-turn-C852AgMT.d.ts +128 -0
- package/dist/stream-agent-turn-rYgaOLO0.js +910 -0
- package/dist/stream-agent-turn-rYgaOLO0.js.map +1 -0
- package/dist/{structural-rollout-zY0oqQzO.js → structural-rollout-3uxVGcg2.js} +459 -235
- package/dist/structural-rollout-3uxVGcg2.js.map +1 -0
- package/dist/{supervise-CsTKbH9R.js → supervise-iPN27pO0.js} +864 -4785
- package/dist/supervise-iPN27pO0.js.map +1 -0
- package/dist/{supervisor-DpjO0Gmy.js → supervisor-CV6Jh28D.js} +7439 -2897
- package/dist/supervisor-CV6Jh28D.js.map +1 -0
- package/dist/testing.d.ts +3 -1
- package/dist/testing.js +271 -221
- package/dist/testing.js.map +1 -1
- package/dist/{top-app-5unxqovu.js → top-app-G4M18b6i.js} +3 -3
- package/dist/{top-app-5unxqovu.js.map → top-app-G4M18b6i.js.map} +1 -1
- package/dist/tui/bin.js +1 -1
- package/dist/tui/index.js +1 -1
- package/dist/{environment-provider-CUFsyymu.d.ts → types-C6Q-J0Dt.d.ts} +51 -114
- package/dist/{types-DnNGJ5Gz.d.ts → types-ebIY0dMG.d.ts} +556 -30
- package/dist/{util-MVgdwuIS.js → util-Bw6srryQ.js} +3 -2
- package/dist/{util-MVgdwuIS.js.map → util-Bw6srryQ.js.map} +1 -1
- package/dist/{workspace-archive-BQxvkypI.js → workspace-archive-aOfJ47ms.js} +3 -3
- package/dist/{workspace-archive-BQxvkypI.js.map → workspace-archive-aOfJ47ms.js.map} +1 -1
- package/package.json +12 -15
- package/skills/agent-graphs/IMPROVE.md +3 -3
- package/skills/agent-graphs/SKILL.md +4 -5
- package/skills/agent-graphs/cases/review-pipeline.json +1 -2
- package/skills/agent-graphs/cases/unmeasured-harness.json +2 -4
- package/dist/backends-CiOCyRHb.js +0 -743
- package/dist/backends-CiOCyRHb.js.map +0 -1
- package/dist/conversation-BpLQZGPH.js.map +0 -1
- package/dist/environment-provider-DqFS6FSZ.js.map +0 -1
- package/dist/improvement-cycle-IJgCbKWQ.js.map +0 -1
- package/dist/index-DLM0W1h1.d.ts +0 -545
- package/dist/knowledge-DF63xPr4.js.map +0 -1
- package/dist/local-harness-BIajef4A.d.ts +0 -465
- package/dist/loop-runner-bin-CWqOpCEw.js.map +0 -1
- package/dist/model-resolution-Btd9iIKV.js +0 -98
- package/dist/model-resolution-Btd9iIKV.js.map +0 -1
- package/dist/openai-tools-D3XfrrQ6.js.map +0 -1
- package/dist/prepare--8EvLqCr.js.map +0 -1
- package/dist/researcher-CoVqNhfI.js.map +0 -1
- package/dist/runtime-5uDVVfER.js.map +0 -1
- package/dist/sandbox-events-Yhd1GYWl.js.map +0 -1
- package/dist/spawn-journal-DsZKDqeh.js.map +0 -1
- package/dist/structural-rollout-zY0oqQzO.js.map +0 -1
- package/dist/supervise-CsTKbH9R.js.map +0 -1
- package/dist/supervisor-DpjO0Gmy.js.map +0 -1
- package/dist/types-C9j4qg6l.d.ts +0 -500
- package/skills/agent-graphs/cases/floor-trap-pi.json +0 -11
|
@@ -1,20 +1,24 @@
|
|
|
1
|
-
import { n as AnalystError,
|
|
2
|
-
import
|
|
3
|
-
import { i as InMemorySpawnJournal, r as InMemoryResultBlobStore, v as isTraceAnalysisStore, x as contentAddress } from "./spawn-journal-
|
|
4
|
-
import { a as randomSuffix, c as stringifySafe, f as zeroTokenUsage, r as isAbortError, s as sleep, t as addTokenUsage } from "./util-
|
|
1
|
+
import { n as AnalystError, s as PlannerError, u as ValidationError } from "./errors-DEAvWQPy.js";
|
|
2
|
+
import "./stream-agent-turn-rYgaOLO0.js";
|
|
3
|
+
import { i as InMemorySpawnJournal, r as InMemoryResultBlobStore, v as isTraceAnalysisStore, x as contentAddress } from "./spawn-journal-saHQzqYi.js";
|
|
4
|
+
import { a as randomSuffix, c as stringifySafe, f as zeroTokenUsage, r as isAbortError, s as sleep, t as addTokenUsage } from "./util-Bw6srryQ.js";
|
|
5
5
|
import { i as redactProtectedValue, r as redactProtectedReason } from "./protected-redaction--F3v1oo8.js";
|
|
6
|
-
import {
|
|
7
|
-
import {
|
|
8
|
-
import {
|
|
9
|
-
import {
|
|
10
|
-
import
|
|
11
|
-
import {
|
|
6
|
+
import { a as notifySandboxEventObserver } from "./sandbox-events-CRDwc5WN.js";
|
|
7
|
+
import { c as profileProviderModel, i as concreteModelId, s as profileModelExecutionSettings, t as assertExecutableAgentProfile } from "./model-policy-CqziaqS1.js";
|
|
8
|
+
import { v as executableAgentProfileSnapshot, y as executableAgentSpecSnapshot } from "./materialization-COJ1UYQ-.js";
|
|
9
|
+
import { $ as createSandboxLineage, I as createWorktreeCliExecutor, N as createExecutor, P as createExecutorRegistry, Q as runAgentRounds, U as createPushTraceSource, Z as defaultSelectWinner, et as probeSandboxCapabilities, l as withDriverExecutor, m as settledToIteration, n as createSupervisor, nt as routerBrain, rt as runBrainLoop, w as rollingDispatch } from "./supervisor-CV6Jh28D.js";
|
|
10
|
+
import "./environment-provider-Dyg8DtLK.js";
|
|
11
|
+
import { i as notifyRuntimeHookEvent } from "./runtime-hooks-C7iJOWm3.js";
|
|
12
|
+
import { C as observe, O as strategyAuthorMethod, T as profileChatClient, b as sample, v as refine, x as sampleThenRefine, y as runAgentic } from "./structural-rollout-3uxVGcg2.js";
|
|
13
|
+
import { It as gateOnDeliverable, Lt as mapExecutorResult, n as supervise } from "./supervise-iPN27pO0.js";
|
|
14
|
+
import "./authoring-CvHwo1oW.js";
|
|
15
|
+
import "./graph-BJTxGOFB.js";
|
|
16
|
+
import { CODING_HARNESSES, InMemoryTraceStore, OUTPUT_VALUE, benjaminiHochberg, buildTrajectory, computeFindingId as computeFindingId$1, confidenceInterval, expandProfileAxes, makeFinding as makeFinding$1, pairedBootstrap, paretoFrontier, wilcoxonSignedRank, wilson } from "@tangle-network/agent-eval";
|
|
12
17
|
import { heldoutSignificance, runProfileMatrix } from "@tangle-network/agent-eval/campaign";
|
|
13
|
-
import { agentProfileSchema,
|
|
14
|
-
import {
|
|
18
|
+
import { agentProfileSchema, canonicalAgentProfileDigest, validateAgentProfileSecurity } from "@tangle-network/agent-interface";
|
|
19
|
+
import { dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
|
|
15
20
|
import { mkdir, readFile, writeFile } from "node:fs/promises";
|
|
16
21
|
import { appendFileSync, chmodSync, constants, copyFileSync, existsSync, lstatSync, mkdirSync, mkdtempSync, readFileSync, readlinkSync, rmSync, symlinkSync, writeFileSync } from "node:fs";
|
|
17
|
-
import { dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
|
|
18
22
|
import { execFileSync, spawn, spawnSync } from "node:child_process";
|
|
19
23
|
import { tmpdir } from "node:os";
|
|
20
24
|
import { createInterface } from "node:readline";
|
|
@@ -531,13 +535,13 @@ const auditSchema = {
|
|
|
531
535
|
};
|
|
532
536
|
/** The route-rigor analyst: compare declared vs revealed vs user intent over a trajectory and return aligned / drifting / diverged with evidence and one recommended intervention. */
|
|
533
537
|
async function auditIntent(input, opts) {
|
|
534
|
-
const res = await
|
|
535
|
-
|
|
538
|
+
const res = await profileChatClient({
|
|
539
|
+
profile: opts.profile,
|
|
540
|
+
executor: opts.executor,
|
|
541
|
+
context: "intent auditor"
|
|
542
|
+
}).chat({
|
|
536
543
|
jsonSchema: auditSchema,
|
|
537
544
|
messages: [{
|
|
538
|
-
role: "system",
|
|
539
|
-
content: opts.auditorInstruction ?? "You audit whether an AI agent is on the RIGHT ROUTE — not whether it works hard, but whether its actions serve the stated intents. Infer the REVEALED intent from the action pattern (what the trajectory is actually optimizing). Compare against the declared task intent, the user intent when given, and the meta-intent when given. Flawless execution down the wrong route is DIVERGED. Busy-work that neither advances nor harms is DRIFTING. Judge only from the trajectory — be specific about which actions ground your verdict. Recommend abort only when continuing cannot serve the intent."
|
|
540
|
-
}, {
|
|
541
545
|
role: "user",
|
|
542
546
|
content: `DECLARED INTENT (the task):\n${input.declaredIntent}\n\n` + (input.userIntent ? `USER INTENT (the principal's actual goal):\n${input.userIntent}\n\n` : "") + (input.metaIntent ? `META-INTENT (what the whole run is for):\n${input.metaIntent}\n\n` : "") + `TRAJECTORY (in order):\n${summarize(input.trace, opts.maxTraceLines ?? 80)}\n\nAudit the route: revealed intent, verdict, evidence, one recommendation.`
|
|
543
547
|
}]
|
|
@@ -1036,7 +1040,10 @@ function loopCostReceipt(result, model) {
|
|
|
1036
1040
|
model,
|
|
1037
1041
|
inputTokens: result.tokenUsage.input,
|
|
1038
1042
|
outputTokens: result.tokenUsage.output,
|
|
1039
|
-
...result.
|
|
1043
|
+
...result.tokenUsage.tokensKnown === false ? { usageUnknown: true } : {},
|
|
1044
|
+
...result.costUsdKnown !== false ? { actualCostUsd: result.costUsd } : {},
|
|
1045
|
+
...result.costUsdKnown === false ? { costUnknown: true } : {},
|
|
1046
|
+
...result.estimatedCostUsd !== void 0 ? { estimatedCostUsd: result.estimatedCostUsd } : {}
|
|
1040
1047
|
};
|
|
1041
1048
|
}
|
|
1042
1049
|
function modelFromLoopOptions(options) {
|
|
@@ -1071,6 +1078,20 @@ function loopDispatch(opts) {
|
|
|
1071
1078
|
}
|
|
1072
1079
|
//#endregion
|
|
1073
1080
|
//#region src/runtime/inline-sandbox-client.ts
|
|
1081
|
+
/**
|
|
1082
|
+
* The ONE pseudo-box adapter: present any one-shot `Executor` (router / bridge /
|
|
1083
|
+
* BYO) as a `SandboxClient` so the round-synchronous `runAgentRounds` can drive it
|
|
1084
|
+
* without each call site re-faking a box. This is the single shell that
|
|
1085
|
+
* `bench/src/router-executor.ts`, generate-eval's old `bridgeSandboxClient`, and
|
|
1086
|
+
* the search-bench bridge transport were each re-implementing.
|
|
1087
|
+
*
|
|
1088
|
+
* It is deliberately for NON-box executors only — a real sandbox harness already
|
|
1089
|
+
* IS a `SandboxClient` (boxes, sessions, fs, fork are real there). Here each
|
|
1090
|
+
* `streamPrompt` runs the executor once and emits the terminal
|
|
1091
|
+
* `{type:'result', data:{finalText, tokenUsage, costUsd}}` event that
|
|
1092
|
+
* `answerOutput`/the kernel's cost ledger already parse — no sessions, no fs,
|
|
1093
|
+
* no fork (those degrade gracefully via the optional `SandboxClient` methods).
|
|
1094
|
+
*/
|
|
1074
1095
|
function isAsyncIterable$1(v) {
|
|
1075
1096
|
return typeof v === "object" && v !== null && Symbol.asyncIterator in v;
|
|
1076
1097
|
}
|
|
@@ -1088,7 +1109,8 @@ async function settle(exec, task, signal) {
|
|
|
1088
1109
|
* instantiated fresh per `streamPrompt` (mirrors the per-spawn executor lifecycle):
|
|
1089
1110
|
* run once on the prompt, emit the terminal result event, tear down.
|
|
1090
1111
|
*/
|
|
1091
|
-
function inlineSandboxClient(factory) {
|
|
1112
|
+
function inlineSandboxClient(factory, defaults = {}) {
|
|
1113
|
+
const capturedDefaultProfile = defaults.profile === void 0 ? void 0 : agentProfileSchema.parse(structuredClone(defaults.profile));
|
|
1092
1114
|
let seq = 0;
|
|
1093
1115
|
return { async create(options) {
|
|
1094
1116
|
const id = `inline-${seq++}`;
|
|
@@ -1101,8 +1123,11 @@ function inlineSandboxClient(factory) {
|
|
|
1101
1123
|
const onAbort = () => controller.abort(callerSignal?.reason ?? /* @__PURE__ */ new Error("prompt aborted"));
|
|
1102
1124
|
if (callerSignal) if (callerSignal.aborted) onAbort();
|
|
1103
1125
|
else callerSignal.addEventListener("abort", onAbort, { once: true });
|
|
1126
|
+
const requestedProfile = (options?.backend && typeof options.backend === "object" ? options.backend.profile : void 0) ?? capturedDefaultProfile;
|
|
1127
|
+
const parsedProfile = agentProfileSchema.safeParse(requestedProfile);
|
|
1128
|
+
if (!parsedProfile.success) throw new Error("inlineSandboxClient: an exact AgentProfile is required; pass defaults.profile or create({ backend: { profile } })");
|
|
1104
1129
|
const exec = factory({
|
|
1105
|
-
profile:
|
|
1130
|
+
profile: parsedProfile.data,
|
|
1106
1131
|
harness: null
|
|
1107
1132
|
}, {
|
|
1108
1133
|
signal: controller.signal,
|
|
@@ -1114,23 +1139,43 @@ function inlineSandboxClient(factory) {
|
|
|
1114
1139
|
const tokensIn = artifact.spent.tokens.input;
|
|
1115
1140
|
const tokensOut = artifact.spent.tokens.output;
|
|
1116
1141
|
const costUsd = artifact.spent.usd;
|
|
1117
|
-
|
|
1142
|
+
const estimatedCostUsd = out?.estimatedCostUsd;
|
|
1143
|
+
if (artifact.spent.iterations > 0 || artifact.spent.tokensKnown === false || artifact.spent.usdKnown === false || tokensIn > 0 || tokensOut > 0 || costUsd > 0 || estimatedCostUsd !== void 0) yield {
|
|
1118
1144
|
type: "llm_call",
|
|
1119
1145
|
data: {
|
|
1120
|
-
|
|
1121
|
-
|
|
1122
|
-
|
|
1146
|
+
...artifact.spent.tokensKnown === false ? {} : {
|
|
1147
|
+
tokensIn,
|
|
1148
|
+
tokensOut
|
|
1149
|
+
},
|
|
1150
|
+
...artifact.spent.usdKnown !== false ? { costUsd } : {},
|
|
1151
|
+
...artifact.spent.tokensKnown === false ? { tokensKnown: false } : {},
|
|
1152
|
+
...artifact.spent.usdKnown === false ? { costKnown: false } : {},
|
|
1153
|
+
...artifact.spent.usdKnown !== false ? {
|
|
1154
|
+
costKnown: true,
|
|
1155
|
+
costProvenance: "provider-receipt"
|
|
1156
|
+
} : {},
|
|
1157
|
+
...estimatedCostUsd !== void 0 ? { estimatedCostUsd } : {},
|
|
1158
|
+
...out?.promptCache ? { promptCache: out.promptCache } : {}
|
|
1123
1159
|
}
|
|
1124
1160
|
};
|
|
1125
1161
|
yield {
|
|
1126
1162
|
type: "result",
|
|
1127
1163
|
data: {
|
|
1128
1164
|
finalText: out?.content ?? "",
|
|
1129
|
-
tokenUsage: {
|
|
1165
|
+
...artifact.spent.tokensKnown === false ? { tokensKnown: false } : { tokenUsage: {
|
|
1130
1166
|
inputTokens: tokensIn,
|
|
1131
1167
|
outputTokens: tokensOut
|
|
1168
|
+
} },
|
|
1169
|
+
...artifact.spent.usdKnown === false ? {
|
|
1170
|
+
costKnown: false,
|
|
1171
|
+
...costUsd > 0 ? { costUsd } : {}
|
|
1172
|
+
} : {
|
|
1173
|
+
costUsd,
|
|
1174
|
+
costKnown: true,
|
|
1175
|
+
costProvenance: "provider-receipt"
|
|
1132
1176
|
},
|
|
1133
|
-
|
|
1177
|
+
...estimatedCostUsd !== void 0 ? { estimatedCostUsd } : {},
|
|
1178
|
+
...out?.promptCache ? { promptCache: out.promptCache } : {}
|
|
1134
1179
|
}
|
|
1135
1180
|
};
|
|
1136
1181
|
} finally {
|
|
@@ -1159,36 +1204,52 @@ function inlineSandboxClient(factory) {
|
|
|
1159
1204
|
* local MCP process applies only when its full canonical bytes match the fixed
|
|
1160
1205
|
* constructor profile; a different generated profile is refused.
|
|
1161
1206
|
*
|
|
1162
|
-
* Event protocol matches `inlineSandboxClient`: one `llm_call
|
|
1163
|
-
*
|
|
1207
|
+
* Event protocol matches `inlineSandboxClient`: known token usage is emitted as one `llm_call`;
|
|
1208
|
+
* Router catalog cost remains a separately-labelled estimate, never billed spend.
|
|
1164
1209
|
*/
|
|
1165
1210
|
/** A same-host `SandboxClient` adapter with no process isolation. Local MCP is
|
|
1166
1211
|
* refused unless the caller explicitly supplies a policy that allows it. */
|
|
1167
1212
|
function localSandboxClient(opts) {
|
|
1168
|
-
|
|
1169
|
-
const
|
|
1170
|
-
|
|
1213
|
+
const defaultProfile = opts.profile === void 0 ? void 0 : executableAgentProfileSnapshot(opts.profile, "localSandboxClient default profile");
|
|
1214
|
+
const router = Object.freeze({ ...opts.router });
|
|
1215
|
+
if (opts.profileSecurityPolicy?.allowLocalMcp && defaultProfile === void 0) throw new ValidationError("localSandboxClient: allowLocalMcp requires a fixed author-controlled profile; dynamic profiles need a real sandbox");
|
|
1216
|
+
const trustedProfileDigest = opts.profileSecurityPolicy?.allowLocalMcp && defaultProfile !== void 0 ? canonicalAgentProfileDigest(defaultProfile) : void 0;
|
|
1171
1217
|
let seq = 0;
|
|
1172
1218
|
return { async create(options) {
|
|
1173
|
-
const profile = (options?.backend)?.profile ??
|
|
1174
|
-
const
|
|
1219
|
+
const profile = executableAgentProfileSnapshot((options?.backend)?.profile ?? defaultProfile, "localSandboxClient");
|
|
1220
|
+
const model = profileProviderModel(profile);
|
|
1221
|
+
const settings = profileModelExecutionSettings(profile, "localSandboxClient");
|
|
1222
|
+
const policyApplies = opts.profileSecurityPolicy !== void 0 && (!opts.profileSecurityPolicy.allowLocalMcp || trustedProfileDigest !== void 0 && canonicalAgentProfileDigest(profile) === trustedProfileDigest);
|
|
1175
1223
|
const mcp = await materializeLocalMcp(profile, {
|
|
1176
1224
|
...opts.keys ? { keys: opts.keys } : {},
|
|
1177
1225
|
...policyApplies ? { profileSecurityPolicy: opts.profileSecurityPolicy } : {}
|
|
1178
1226
|
});
|
|
1179
1227
|
const brain = routerBrain({
|
|
1180
|
-
routerBaseUrl:
|
|
1181
|
-
routerKey:
|
|
1182
|
-
model
|
|
1183
|
-
|
|
1228
|
+
routerBaseUrl: router.baseUrl,
|
|
1229
|
+
routerKey: router.key,
|
|
1230
|
+
model,
|
|
1231
|
+
...settings.retry !== void 0 ? { retry: settings.retry } : {},
|
|
1232
|
+
...settings.maxTokens !== void 0 ? { maxTokens: settings.maxTokens } : {},
|
|
1233
|
+
...settings.stream !== void 0 ? { stream: settings.stream } : {}
|
|
1234
|
+
}, {
|
|
1235
|
+
...settings.temperature !== void 0 ? { temperature: settings.temperature } : {},
|
|
1236
|
+
...settings.seed !== void 0 ? { seed: settings.seed } : {},
|
|
1237
|
+
...settings.toolChoice !== void 0 ? { toolChoice: settings.toolChoice } : {},
|
|
1238
|
+
...settings.extraBody !== void 0 ? { extraBody: settings.extraBody } : {},
|
|
1239
|
+
...profile.model?.reasoningEffort ? { reasoningEffort: profile.model.reasoningEffort } : {}
|
|
1240
|
+
});
|
|
1184
1241
|
const system = [profile.prompt?.systemPrompt, ...profile.prompt?.instructions ?? []].filter((s) => typeof s === "string" && s.trim().length > 0).join("\n\n");
|
|
1185
1242
|
return {
|
|
1186
1243
|
id: `local-${seq++}`,
|
|
1187
1244
|
async *streamPrompt(message, popts) {
|
|
1188
|
-
let
|
|
1245
|
+
let estimatedCostUsd = 0;
|
|
1246
|
+
let sawEstimatedCost = false;
|
|
1189
1247
|
const chat = async (messages, tools) => {
|
|
1190
1248
|
const r = await brain(messages, tools);
|
|
1191
|
-
if (r.
|
|
1249
|
+
if (r.costProvenance === "catalog-estimate" && r.costUsd !== void 0) {
|
|
1250
|
+
estimatedCostUsd += r.costUsd;
|
|
1251
|
+
sawEstimatedCost = true;
|
|
1252
|
+
}
|
|
1192
1253
|
return r;
|
|
1193
1254
|
};
|
|
1194
1255
|
const r = await runBrainLoop({
|
|
@@ -1202,26 +1263,31 @@ function localSandboxClient(opts) {
|
|
|
1202
1263
|
role: "user",
|
|
1203
1264
|
content: message
|
|
1204
1265
|
}],
|
|
1205
|
-
maxTurns,
|
|
1266
|
+
maxTurns: settings.maxTurns ?? 0,
|
|
1206
1267
|
hooks: { stopBefore: () => popts?.signal?.aborted === true }
|
|
1207
1268
|
});
|
|
1208
|
-
if (r.
|
|
1269
|
+
if (r.turns > 0) yield {
|
|
1209
1270
|
type: "llm_call",
|
|
1210
1271
|
data: {
|
|
1211
|
-
|
|
1212
|
-
|
|
1213
|
-
|
|
1272
|
+
model,
|
|
1273
|
+
...r.tokensKnown === false ? { tokensKnown: false } : {
|
|
1274
|
+
tokensIn: r.usage.input,
|
|
1275
|
+
tokensOut: r.usage.output
|
|
1276
|
+
},
|
|
1277
|
+
costKnown: false,
|
|
1278
|
+
...sawEstimatedCost ? { estimatedCostUsd } : {}
|
|
1214
1279
|
}
|
|
1215
1280
|
};
|
|
1216
1281
|
yield {
|
|
1217
1282
|
type: "result",
|
|
1218
1283
|
data: {
|
|
1219
1284
|
finalText: r.final,
|
|
1220
|
-
tokenUsage: {
|
|
1285
|
+
...r.tokensKnown === false ? { tokensKnown: false } : { tokenUsage: {
|
|
1221
1286
|
inputTokens: r.usage.input,
|
|
1222
1287
|
outputTokens: r.usage.output
|
|
1223
|
-
},
|
|
1224
|
-
|
|
1288
|
+
} },
|
|
1289
|
+
costKnown: false,
|
|
1290
|
+
...sawEstimatedCost ? { estimatedCostUsd } : {}
|
|
1225
1291
|
}
|
|
1226
1292
|
};
|
|
1227
1293
|
},
|
|
@@ -1269,28 +1335,26 @@ function resolveSandboxClient(opts) {
|
|
|
1269
1335
|
return opts.sandboxClient;
|
|
1270
1336
|
case "bridge": {
|
|
1271
1337
|
const bridge = opts.bridge;
|
|
1272
|
-
if (!bridge?.bearer
|
|
1338
|
+
if (!bridge?.bearer) throw new Error("resolveSandboxClient: backend 'bridge' requires bridge.bearer");
|
|
1273
1339
|
return inlineSandboxClient(createExecutor({
|
|
1274
1340
|
backend: "bridge",
|
|
1275
1341
|
bridgeUrl: bridge.url ?? "http://127.0.0.1:3355",
|
|
1276
1342
|
bridgeBearer: bridge.bearer,
|
|
1277
|
-
model: bridge.model,
|
|
1278
1343
|
timeoutMs: bridge.timeoutMs
|
|
1279
1344
|
}));
|
|
1280
1345
|
}
|
|
1281
1346
|
case "router": {
|
|
1282
1347
|
const router = opts.router;
|
|
1283
|
-
if (!router?.baseUrl || !router.key
|
|
1348
|
+
if (!router?.baseUrl || !router.key) throw new Error("resolveSandboxClient: backend 'router' requires router.baseUrl and router.key");
|
|
1284
1349
|
return inlineSandboxClient(createExecutor({
|
|
1285
1350
|
backend: "router",
|
|
1286
1351
|
routerBaseUrl: router.baseUrl,
|
|
1287
|
-
routerKey: router.key
|
|
1288
|
-
model: router.model
|
|
1352
|
+
routerKey: router.key
|
|
1289
1353
|
}));
|
|
1290
1354
|
}
|
|
1291
1355
|
case "local": {
|
|
1292
1356
|
const local = opts.local;
|
|
1293
|
-
if (!local?.router?.baseUrl || !local.router.key
|
|
1357
|
+
if (!local?.router?.baseUrl || !local.router.key) throw new Error("resolveSandboxClient: backend 'local' requires local.router.baseUrl and local.router.key");
|
|
1294
1358
|
return localSandboxClient(local);
|
|
1295
1359
|
}
|
|
1296
1360
|
}
|
|
@@ -1314,7 +1378,7 @@ function resolveSandboxClient(opts) {
|
|
|
1314
1378
|
*
|
|
1315
1379
|
* - LEVEL 0 (declarative): `cases` / `prompt` / `score` / `axis`.
|
|
1316
1380
|
* - LEVEL 1 (seams): `backends`, `flags`, `parseOutput`, `onCellEvents`,
|
|
1317
|
-
* `resolveModel`, `setup`/`teardown`, `export`, `
|
|
1381
|
+
* `resolveModel`, `setup`/`teardown`, `export`, `matrix`
|
|
1318
1382
|
* passthrough.
|
|
1319
1383
|
* - LEVEL 2 (replacement): `dispatch` and `judges` swap out the whole
|
|
1320
1384
|
* loop wiring or scoring; `runProfileMatrix` itself stays public as the
|
|
@@ -1373,10 +1437,6 @@ function splitList(v) {
|
|
|
1373
1437
|
function withSnapshot(model, snapshot) {
|
|
1374
1438
|
return model.includes("@") ? model : `${model}@${snapshot}`;
|
|
1375
1439
|
}
|
|
1376
|
-
/** The bare model id the backend actually serves (identity snapshot stripped). */
|
|
1377
|
-
function bareModel(model) {
|
|
1378
|
-
return model.split("@")[0] ?? model;
|
|
1379
|
-
}
|
|
1380
1440
|
function gitSha() {
|
|
1381
1441
|
try {
|
|
1382
1442
|
return execFileSync("git", ["rev-parse", "HEAD"], { encoding: "utf8" }).trim();
|
|
@@ -1458,13 +1518,11 @@ function defineLeaderboard(spec) {
|
|
|
1458
1518
|
if (!rawModels || rawModels.length === 0) throw new Error(`defineLeaderboard(${spec.name}): no models — pass --models, set spec.axis.models, or give spec.baseProfile a model.default`);
|
|
1459
1519
|
const models = rawModels.map((m) => withSnapshot(m, snapshot));
|
|
1460
1520
|
const profiles = expandProfileAxes({
|
|
1461
|
-
base: spec.baseProfile
|
|
1462
|
-
name: spec.name,
|
|
1463
|
-
model: { default: bareModel(models[0] ?? "") }
|
|
1464
|
-
},
|
|
1521
|
+
base: spec.baseProfile,
|
|
1465
1522
|
harnesses,
|
|
1466
1523
|
models
|
|
1467
1524
|
});
|
|
1525
|
+
for (const profile of profiles) assertExecutableAgentProfile(profile, `defineLeaderboard(${spec.name})`);
|
|
1468
1526
|
const ctx = {
|
|
1469
1527
|
name: spec.name,
|
|
1470
1528
|
backend: backendName,
|
|
@@ -1489,7 +1547,6 @@ function defineLeaderboard(spec) {
|
|
|
1489
1547
|
bridge: {
|
|
1490
1548
|
url: process.env.CLI_BRIDGE_URL,
|
|
1491
1549
|
bearer,
|
|
1492
|
-
model: bareModel(models[0] ?? ""),
|
|
1493
1550
|
timeoutMs: 9e5
|
|
1494
1551
|
}
|
|
1495
1552
|
});
|
|
@@ -1519,21 +1576,11 @@ function defineLeaderboard(spec) {
|
|
|
1519
1576
|
return [...served][0] ?? cellProfile.model?.default;
|
|
1520
1577
|
},
|
|
1521
1578
|
toLoopOptions: (cellScenario, cellProfile) => {
|
|
1522
|
-
const axis = harnessAxisOf(cellProfile);
|
|
1523
|
-
const modelId = bareModel(axis?.model ?? models[0] ?? "");
|
|
1524
|
-
const backendModel = {
|
|
1525
|
-
...spec.modelBackend,
|
|
1526
|
-
...!isHarnessNativeModel(modelId) || backendName === "cli-bridge" ? { model: modelId } : {}
|
|
1527
|
-
};
|
|
1528
1579
|
return {
|
|
1529
1580
|
driver: naiveRetryDriver(shots),
|
|
1530
1581
|
agentRun: {
|
|
1531
1582
|
profile: cellProfile,
|
|
1532
|
-
taskToPrompt: (s) => `${promptOf(s)}\n\n<!-- independent-attempt:${shotNonce++}
|
|
1533
|
-
...axis ? { sandboxOverrides: { backend: {
|
|
1534
|
-
type: axis.harness,
|
|
1535
|
-
...Object.keys(backendModel).length > 0 ? { model: backendModel } : {}
|
|
1536
|
-
} } } : {}
|
|
1583
|
+
taskToPrompt: (s) => `${promptOf(s)}\n\n<!-- independent-attempt:${shotNonce++} -->`
|
|
1537
1584
|
},
|
|
1538
1585
|
output: { parse: (events) => spec.parseOutput ? spec.parseOutput(events, cellScenario.case) : collectAgentResponseText(events) ?? "" },
|
|
1539
1586
|
validator: { validate: async (output) => {
|
|
@@ -1654,11 +1701,10 @@ async function harvestCorpus(opts) {
|
|
|
1654
1701
|
if (opts.signal?.aborted) return;
|
|
1655
1702
|
try {
|
|
1656
1703
|
const obs = await observe(input, {
|
|
1657
|
-
|
|
1658
|
-
|
|
1704
|
+
profile: opts.profile,
|
|
1705
|
+
executor: opts.executor,
|
|
1659
1706
|
corpus: opts.corpus,
|
|
1660
1707
|
tags: opts.tags ?? [],
|
|
1661
|
-
...opts.analystInstruction ? { analystInstruction: opts.analystInstruction } : {},
|
|
1662
1708
|
...opts.signal ? { signal: opts.signal } : {}
|
|
1663
1709
|
});
|
|
1664
1710
|
report.runsObserved += 1;
|
|
@@ -1942,7 +1988,7 @@ function observedBestScore(settledSoFar) {
|
|
|
1942
1988
|
* CONCRETE blocker (never an eager over-fan, never a silent drop), and a `blocked` outcome always
|
|
1943
1989
|
* names at least one blocker (a shape that cannot finish MUST say why — `blocked([])` throws).
|
|
1944
1990
|
*
|
|
1945
|
-
* @
|
|
1991
|
+
* @stable
|
|
1946
1992
|
*/
|
|
1947
1993
|
/**
|
|
1948
1994
|
* The single content-free valid-only winner selector. Among the gated-VALID children only
|
|
@@ -1977,6 +2023,8 @@ function selectValidWinner(opts) {
|
|
|
1977
2023
|
* pool would not admit, or a stage whose `collect` chose to block) short-circuits — its blockers
|
|
1978
2024
|
* ARE the pipeline's blockers, never coerced past a failed stage. The terminal stage's `done`
|
|
1979
2025
|
* deliverable is the pipeline's deliverable.
|
|
2026
|
+
*
|
|
2027
|
+
* @stable
|
|
1980
2028
|
*/
|
|
1981
2029
|
function pipeline(stages) {
|
|
1982
2030
|
if (stages.length === 0) throw new ValidationError("pipeline: at least one stage is required");
|
|
@@ -2013,6 +2061,8 @@ function pipeline(stages) {
|
|
|
2013
2061
|
* `opts.width` swaps the single round for `rollingDispatch`: at most `width` items live at once,
|
|
2014
2062
|
* refilled the instant one settles. Selection, blockers, and the conserved pool are unchanged —
|
|
2015
2063
|
* the refill behavior lives in the existing combinator rather than in a rival primitive.
|
|
2064
|
+
*
|
|
2065
|
+
* @stable
|
|
2016
2066
|
*/
|
|
2017
2067
|
function fanout(items, opts) {
|
|
2018
2068
|
if (opts.synthesize && opts.selectWinner) throw new ValidationError("fanout: pass at most one of `synthesize` or `selectWinner`");
|
|
@@ -2096,6 +2146,8 @@ function fanout(items, opts) {
|
|
|
2096
2146
|
* `until` on the resulting trace-derived findings (the analyst spawns into THIS scope, so its
|
|
2097
2147
|
* compute is conserved-pooled — equal-k holds by construction). Absent an analyst the findings
|
|
2098
2148
|
* argument is the empty array — never a fabricated finding (fail-loud honesty over a silent default).
|
|
2149
|
+
*
|
|
2150
|
+
* @stable
|
|
2099
2151
|
*/
|
|
2100
2152
|
function loopUntil(seed, spec) {
|
|
2101
2153
|
return (ctx) => ({
|
|
@@ -2144,6 +2196,8 @@ function loopUntil(seed, spec) {
|
|
|
2144
2196
|
* reaches another judge's task; the merge never spawns or re-ranks). A `down` judge carries no
|
|
2145
2197
|
* verdict and is excluded from the merge denominator. A panel that admitted no judge is a
|
|
2146
2198
|
* concrete blocker before `merge` is consulted.
|
|
2199
|
+
*
|
|
2200
|
+
* @stable
|
|
2147
2201
|
*/
|
|
2148
2202
|
function panel(spec) {
|
|
2149
2203
|
if (spec.judges.length === 0) throw new ValidationError("panel: at least one judge is required");
|
|
@@ -2193,6 +2247,8 @@ function panel(spec) {
|
|
|
2193
2247
|
* it; only a `valid` verifier verdict ships. Any other outcome (implement down, verifier down,
|
|
2194
2248
|
* verifier verdict absent or not `valid`) is a concrete blocker carrying the failure verbatim —
|
|
2195
2249
|
* never a coerced "done". The implement child does not grade itself.
|
|
2250
|
+
*
|
|
2251
|
+
* @stable
|
|
2196
2252
|
*/
|
|
2197
2253
|
function verify(spec) {
|
|
2198
2254
|
return (ctx) => ({
|
|
@@ -2237,6 +2293,8 @@ function verify(spec) {
|
|
|
2237
2293
|
* the widen loop sees it. The shipped default (`flatWidenGate`) never widens, so no widen child is
|
|
2238
2294
|
* ever live when the analyst runs and the wire is exact; a non-flat gate must drive the analyst on
|
|
2239
2295
|
* a scope whose siblings are quiesced, or read findings without the shared-cursor drain.
|
|
2296
|
+
*
|
|
2297
|
+
* @stable
|
|
2240
2298
|
*/
|
|
2241
2299
|
function widen(spec) {
|
|
2242
2300
|
return (ctx) => ({
|
|
@@ -2686,19 +2744,22 @@ function registerShape(name, factory) {
|
|
|
2686
2744
|
* receive a ctx with the persona seams merged in — so a persona never has to pre-close its
|
|
2687
2745
|
* factories by hand. A persona may instead supply a fully-built `registry` and skip the wrap.
|
|
2688
2746
|
*
|
|
2689
|
-
* @
|
|
2747
|
+
* @stable
|
|
2690
2748
|
*/
|
|
2691
2749
|
/**
|
|
2692
2750
|
* Build a frozen `Persona`. Fails loud on the executors-supplied invariant: a persona with
|
|
2693
2751
|
* neither a pre-built registry nor a seam bag cannot resolve its built-in runtimes, so it is
|
|
2694
2752
|
* unrunnable — refuse it at definition time, not at the first spawn. Pure; no I/O.
|
|
2753
|
+
*
|
|
2754
|
+
* @stable
|
|
2695
2755
|
*/
|
|
2696
2756
|
function definePersona(input) {
|
|
2697
2757
|
if (!input.executors.registry && !input.executors.seams) throw new ValidationError(`definePersona("${input.name}"): executors must supply a registry or a seams bag (built-in runtimes read their seams off ExecutorContext; neither was provided)`);
|
|
2698
2758
|
if (!input.root || typeof input.root !== "object" || !("harness" in input.root)) throw new ValidationError(`definePersona("${input.name}"): root must be an AgentSpec`);
|
|
2759
|
+
const root = executableAgentSpecSnapshot(input.root, `definePersona("${input.name}")`);
|
|
2699
2760
|
return Object.freeze({
|
|
2700
2761
|
name: input.name,
|
|
2701
|
-
root
|
|
2762
|
+
root,
|
|
2702
2763
|
directive: input.directive,
|
|
2703
2764
|
context: input.context,
|
|
2704
2765
|
executors: input.executors,
|
|
@@ -2741,6 +2802,8 @@ function createShapeContext(persona, budget, analyst) {
|
|
|
2741
2802
|
* `ShapeContext`, and runs the resulting root `Agent` to a typed `SupervisedResult<Outcome>`.
|
|
2742
2803
|
* Fail loud on an unknown shape name or an unresolvable persona registry — never a silent
|
|
2743
2804
|
* default-shape fallback.
|
|
2805
|
+
*
|
|
2806
|
+
* @stable
|
|
2744
2807
|
*/
|
|
2745
2808
|
async function runPersonified(options) {
|
|
2746
2809
|
const { persona } = options;
|
|
@@ -3291,22 +3354,34 @@ async function pool(items, limit, fn) {
|
|
|
3291
3354
|
async function preflightModels(cfg) {
|
|
3292
3355
|
if (cfg.modelPreflight === false) return;
|
|
3293
3356
|
if (cfg.worker.complete && !cfg.modelPreflight) return;
|
|
3294
|
-
const
|
|
3357
|
+
const profiles = [cfg.worker.workerProfile, cfg.worker.analystProfile ?? cfg.worker.workerProfile];
|
|
3358
|
+
const profilesByModel = /* @__PURE__ */ new Map();
|
|
3359
|
+
for (const [index, profile] of profiles.entries()) {
|
|
3360
|
+
const model = concreteModelId(profile.model?.default);
|
|
3361
|
+
if (!model) throw new Error(`Benchmark ${index === 0 ? "worker" : "analyst"} AgentProfile.model.default must name an exact model`);
|
|
3362
|
+
if (!profilesByModel.has(model)) profilesByModel.set(model, profile);
|
|
3363
|
+
}
|
|
3364
|
+
const models = [...profilesByModel.keys()];
|
|
3295
3365
|
const timeoutMs = cfg.modelPreflightTimeoutMs ?? 3e4;
|
|
3296
3366
|
if (!Number.isFinite(timeoutMs) || timeoutMs <= 0) throw new Error("modelPreflightTimeoutMs must be a positive finite number");
|
|
3297
3367
|
const check = cfg.modelPreflight ?? (async (model, worker, signal) => {
|
|
3298
|
-
|
|
3299
|
-
|
|
3300
|
-
|
|
3301
|
-
|
|
3302
|
-
|
|
3303
|
-
|
|
3304
|
-
|
|
3305
|
-
|
|
3306
|
-
|
|
3307
|
-
|
|
3308
|
-
|
|
3309
|
-
|
|
3368
|
+
const profile = profilesByModel.get(model);
|
|
3369
|
+
if (!profile) throw new Error(`Benchmark preflight has no AgentProfile for model ${model}`);
|
|
3370
|
+
await profileChatClient({
|
|
3371
|
+
profile,
|
|
3372
|
+
context: `runBenchmark model preflight (${model})`,
|
|
3373
|
+
executor: {
|
|
3374
|
+
backend: "router",
|
|
3375
|
+
routerBaseUrl: worker.routerBaseUrl,
|
|
3376
|
+
routerKey: worker.routerKey
|
|
3377
|
+
}
|
|
3378
|
+
}).chat({
|
|
3379
|
+
model,
|
|
3380
|
+
messages: [{
|
|
3381
|
+
role: "user",
|
|
3382
|
+
content: "Reply OK."
|
|
3383
|
+
}]
|
|
3384
|
+
}, { signal });
|
|
3310
3385
|
});
|
|
3311
3386
|
const failures = (await Promise.allSettled(models.map(async (model) => {
|
|
3312
3387
|
const controller = new AbortController();
|
|
@@ -3362,8 +3437,10 @@ async function runBenchmark(cfg) {
|
|
|
3362
3437
|
resolved: r.resolved,
|
|
3363
3438
|
progression: r.progression,
|
|
3364
3439
|
usd: r.usd,
|
|
3440
|
+
usdKnown: r.usdKnown,
|
|
3365
3441
|
ms: r.ms,
|
|
3366
|
-
tokens: r.tokens
|
|
3442
|
+
tokens: r.tokens,
|
|
3443
|
+
tokensKnown: r.tokensKnown
|
|
3367
3444
|
};
|
|
3368
3445
|
} catch (e) {
|
|
3369
3446
|
errors[s.name] = e instanceof Error ? e.message.slice(0, 300) : String(e);
|
|
@@ -3372,11 +3449,13 @@ async function runBenchmark(cfg) {
|
|
|
3372
3449
|
resolved: false,
|
|
3373
3450
|
progression: [],
|
|
3374
3451
|
usd: 0,
|
|
3452
|
+
usdKnown: true,
|
|
3375
3453
|
ms: 0,
|
|
3376
3454
|
tokens: {
|
|
3377
3455
|
input: 0,
|
|
3378
3456
|
output: 0
|
|
3379
|
-
}
|
|
3457
|
+
},
|
|
3458
|
+
tokensKnown: true
|
|
3380
3459
|
};
|
|
3381
3460
|
}
|
|
3382
3461
|
row = {
|
|
@@ -3403,6 +3482,7 @@ async function runBenchmark(cfg) {
|
|
|
3403
3482
|
score: mean(cells.map((c) => c.score)),
|
|
3404
3483
|
resolved: mean(cells.map((c) => c.resolved ? 1 : 0)),
|
|
3405
3484
|
usd: mean(cells.map((c) => c.usd)),
|
|
3485
|
+
usdKnownRate: mean(cells.map((c) => c.usdKnown ? 1 : 0)),
|
|
3406
3486
|
ms: mean(cells.map((c) => c.ms))
|
|
3407
3487
|
};
|
|
3408
3488
|
}
|
|
@@ -3794,6 +3874,8 @@ export default defineStrategy('your-strategy-name', async ({ surface, task, budg
|
|
|
3794
3874
|
// your composition (listTools comes from the destructured context — it is NOT a global)
|
|
3795
3875
|
})
|
|
3796
3876
|
`;
|
|
3877
|
+
/** Standing behavior callers put in the strategy-author AgentProfile. */
|
|
3878
|
+
const strategyAuthorSystemPrompt = "You are a senior researcher authoring optimization strategies for agent loops: you read per-task losses like experimental data, form a mechanism-level hypothesis, and author the one composition that tests it. Output exactly one fenced ```ts code block and nothing else.";
|
|
3797
3879
|
/** Static CONTRACT lint over an authored strategy module — the module-boundary
|
|
3798
3880
|
* enforcement of the harness's two measurement invariants:
|
|
3799
3881
|
* - author blindness: the only import allowed is the kernel surface. A body that could
|
|
@@ -3820,21 +3902,17 @@ function assertStrategyContract(code) {
|
|
|
3820
3902
|
}
|
|
3821
3903
|
/** One authoring attempt: chat with the given model, extract the fenced module. Throws
|
|
3822
3904
|
* when the reply carries no code block. */
|
|
3823
|
-
async function requestAuthoredCode(opts,
|
|
3824
|
-
const res = await
|
|
3825
|
-
|
|
3826
|
-
|
|
3827
|
-
|
|
3828
|
-
|
|
3829
|
-
|
|
3830
|
-
|
|
3831
|
-
|
|
3832
|
-
role: "user",
|
|
3833
|
-
content: `${opts.contract ?? strategyAuthorContract}\n\nBASELINE RESULTS on the "${opts.environmentName}" environment (budget=${opts.budget}) — the per-task losses are your gradient:\n${opts.lossesJson}\n\nAuthor ONE new strategy that you expect to beat the baselines on THIS environment at the same budget.\n${strategyAuthorMethod}\n\nOutput only the module code block.`
|
|
3834
|
-
}]
|
|
3835
|
-
}, { ...opts.signal ? { signal: opts.signal } : {} });
|
|
3905
|
+
async function requestAuthoredCode(opts, profile) {
|
|
3906
|
+
const res = await profileChatClient({
|
|
3907
|
+
profile,
|
|
3908
|
+
executor: opts.executor,
|
|
3909
|
+
context: "strategy author"
|
|
3910
|
+
}).chat({ messages: [{
|
|
3911
|
+
role: "user",
|
|
3912
|
+
content: `${opts.contract ?? strategyAuthorContract}\n\nBASELINE RESULTS on the "${opts.environmentName}" environment (budget=${opts.budget}) — the per-task losses are your gradient:\n${opts.lossesJson}\n\nAuthor ONE new strategy that you expect to beat the baselines on THIS environment at the same budget.\n${strategyAuthorMethod}\n\nOutput only the module code block.`
|
|
3913
|
+
}] }, { ...opts.signal ? { signal: opts.signal } : {} });
|
|
3836
3914
|
const match = res.content.match(/```(?:ts|typescript)?\s*\n([\s\S]*?)```/);
|
|
3837
|
-
if (!match?.[1]) throw new Error(`authorStrategy: no code block in the author's reply
|
|
3915
|
+
if (!match?.[1]) throw new Error(`authorStrategy: no code block in the author's reply: ${res.content.slice(0, 300)}`);
|
|
3838
3916
|
return match[1];
|
|
3839
3917
|
}
|
|
3840
3918
|
/** Author + load a strategy from losses. Throws when the author emits no loadable module;
|
|
@@ -3842,10 +3920,10 @@ async function requestAuthoredCode(opts, model) {
|
|
|
3842
3920
|
async function authorStrategy(opts) {
|
|
3843
3921
|
let code;
|
|
3844
3922
|
try {
|
|
3845
|
-
code = await requestAuthoredCode(opts, opts.
|
|
3923
|
+
code = await requestAuthoredCode(opts, opts.profile);
|
|
3846
3924
|
} catch (primaryError) {
|
|
3847
|
-
if (!opts.
|
|
3848
|
-
code = await requestAuthoredCode(opts, opts.
|
|
3925
|
+
if (!opts.fallbackProfile) throw primaryError;
|
|
3926
|
+
code = await requestAuthoredCode(opts, opts.fallbackProfile);
|
|
3849
3927
|
}
|
|
3850
3928
|
assertStrategyContract(code);
|
|
3851
3929
|
mkdirSync(opts.outDir, { recursive: true });
|
|
@@ -3883,6 +3961,8 @@ async function authorStrategy(opts) {
|
|
|
3883
3961
|
* Lineage fields (`parent`, `generation`) are recorded on every archive node so a
|
|
3884
3962
|
* descendant-productivity parent-selection policy can be added without changing the
|
|
3885
3963
|
* report schema; the v1 search authors from the latest tournament's losses.
|
|
3964
|
+
*
|
|
3965
|
+
* @experimental
|
|
3886
3966
|
*/
|
|
3887
3967
|
/** Strategy means recomputed over the DISCRIMINATING tasks only — tasks where the field
|
|
3888
3968
|
* strategies did not all score identically. Zero-spread tasks (everyone 1.0, everyone
|
|
@@ -4084,11 +4164,9 @@ async function runStrategyEvolution(cfg) {
|
|
|
4084
4164
|
const contract = `${strategyAuthorContract}${cfg.objective === "cost" ? `\n\nYOUR OBJECTIVE: match or exceed the incumbent's SCORE while spending LESS (the losses include usd per task). Promotion requires proven score non-inferiority PLUS significant cost savings — a strategy that ties the score at half the cost WINS; a cheaper strategy that loses score by more than ${((cfg.scoreTolerance ?? .05) * 100).toFixed(0)}pp LOSES.` : ""}\n\nEXAMPLE TOOLS FROM ONE TASK (tool sets VARY per task on this domain — a strategy MUST select tool names from await listTools(handle) at runtime; hardcoding these example names will zero your score on most tasks):\n${toolCatalog}\n\nSTRATEGIES ALREADY IN THE TOURNAMENT (author something MEANINGFULLY different — a new composition, not a rename):\n${fieldSummary(archive)}\n\nYou are authoring candidate ${i + 1} of ${populationSize} this generation; explore a distinct region of the strategy space from your siblings.`;
|
|
4085
4165
|
try {
|
|
4086
4166
|
const authored = await authorStrategy({
|
|
4087
|
-
|
|
4088
|
-
|
|
4089
|
-
...cfg.author.
|
|
4090
|
-
...cfg.author.temperature !== void 0 ? { temperature: cfg.author.temperature } : {},
|
|
4091
|
-
...cfg.author.maxTokens !== void 0 ? { maxTokens: cfg.author.maxTokens } : {},
|
|
4167
|
+
profile: cfg.author.profile,
|
|
4168
|
+
executor: cfg.author.executor,
|
|
4169
|
+
...cfg.author.fallbackProfile ? { fallbackProfile: cfg.author.fallbackProfile } : {},
|
|
4092
4170
|
contract,
|
|
4093
4171
|
environmentName: cfg.environment.name,
|
|
4094
4172
|
lossesJson,
|
|
@@ -4225,24 +4303,24 @@ async function runStrategyEvolution(cfg) {
|
|
|
4225
4303
|
const tolerance = cfg.reproducerCheck.tolerance ?? .05;
|
|
4226
4304
|
const championHoldoutScore = holdout.perStrategy[incumbent.name]?.score ?? 0;
|
|
4227
4305
|
try {
|
|
4228
|
-
const summary = (await
|
|
4229
|
-
|
|
4230
|
-
|
|
4231
|
-
|
|
4232
|
-
|
|
4233
|
-
|
|
4234
|
-
|
|
4235
|
-
},
|
|
4236
|
-
|
|
4237
|
-
|
|
4238
|
-
|
|
4239
|
-
|
|
4306
|
+
const summary = (await profileChatClient({
|
|
4307
|
+
profile: {
|
|
4308
|
+
...cfg.author.profile,
|
|
4309
|
+
prompt: {
|
|
4310
|
+
...cfg.author.profile.prompt,
|
|
4311
|
+
systemPrompt: `Summarize the optimization strategy implemented by this code in at most ${words} words. Describe the COMPOSITION (shots, critique, artifact handling, restarts, stopping) — not the code. Output only the summary.`
|
|
4312
|
+
}
|
|
4313
|
+
},
|
|
4314
|
+
executor: cfg.author.executor,
|
|
4315
|
+
context: "strategy reproducer summary"
|
|
4316
|
+
}).chat({ messages: [{
|
|
4317
|
+
role: "user",
|
|
4318
|
+
content: championCode
|
|
4319
|
+
}] })).content.trim();
|
|
4240
4320
|
const reproduced = await authorStrategy({
|
|
4241
|
-
|
|
4242
|
-
|
|
4243
|
-
...cfg.author.
|
|
4244
|
-
...cfg.author.maxTokens !== void 0 ? { maxTokens: cfg.author.maxTokens } : {},
|
|
4245
|
-
temperature: .2,
|
|
4321
|
+
profile: cfg.author.profile,
|
|
4322
|
+
executor: cfg.author.executor,
|
|
4323
|
+
...cfg.author.fallbackProfile ? { fallbackProfile: cfg.author.fallbackProfile } : {},
|
|
4246
4324
|
contract: `${strategyAuthorContract}\n\nIMPLEMENT EXACTLY THIS STRATEGY (a colleague's description — do not invent a different approach):\n${summary}`,
|
|
4247
4325
|
environmentName: cfg.environment.name,
|
|
4248
4326
|
lossesJson: "[]",
|
|
@@ -4289,631 +4367,121 @@ async function runStrategyEvolution(cfg) {
|
|
|
4289
4367
|
};
|
|
4290
4368
|
}
|
|
4291
4369
|
//#endregion
|
|
4292
|
-
//#region src/runtime/
|
|
4293
|
-
/**
|
|
4294
|
-
* `streamAgentTurn` — the ONE run-a-turn event-stream contract over every
|
|
4295
|
-
* execution substrate: a sandbox box (`SandboxInstance.streamPrompt`), a
|
|
4296
|
-
* one-shot `Executor` (cli-bridge / router / BYO, via `ExecutorFactory`), and
|
|
4297
|
-
* an in-process `AgentExecutionBackend` (the `resolveAgentBackend` output).
|
|
4298
|
-
*
|
|
4299
|
-
* One function, one vocabulary: every backend kind yields the existing
|
|
4300
|
-
* `RuntimeStreamEvent` union incrementally and ALWAYS terminates with a
|
|
4301
|
-
* `final` event whose `text` is the turn's final text and whose
|
|
4302
|
-
* `metadata.tokenUsage` / `metadata.costUsd` / `metadata.model` carry the
|
|
4303
|
-
* turn's metered usage. `collectAgentTurn` drains a stream into that terminal
|
|
4304
|
-
* summary plus the full event list.
|
|
4305
|
-
*
|
|
4306
|
-
* This is a UNIFICATION seam, not a new stream parser — each kind is a thin
|
|
4307
|
-
* adapter over code that already exists and is already hardened:
|
|
4308
|
-
* - `box` — `mapSandboxEvent` + `extractLlmCallEvent` (sandbox-events.ts)
|
|
4309
|
-
* project the sandbox event stream; nothing is re-mapped here.
|
|
4310
|
-
* - `executor` — `inlineSandboxClient` (the ONE executor→box adapter) turns
|
|
4311
|
-
* the factory into a box, then the box path drives it. The
|
|
4312
|
-
* executor's settle/teardown lifecycle stays in that adapter.
|
|
4313
|
-
* - `chat` — the backend's own `stream()` surface, normalized by
|
|
4314
|
-
* `normalizeBackendStreamEvent` (the same projection
|
|
4315
|
-
* `runAgentTaskStream` applies).
|
|
4316
|
-
*
|
|
4317
|
-
* Distinct from `openSandboxRun` (box-only, session resume over one persistent
|
|
4318
|
-
* artifact, raw `SandboxEvent` deliverables) and from `runAgentTaskStream`
|
|
4319
|
-
* (full task lifecycle: knowledge preflight, session store, resume). This is
|
|
4320
|
-
* the minimal turn primitive underneath both worlds: prompt in, one normalized
|
|
4321
|
-
* event stream out, terminal result+usage guaranteed on every non-thrown path.
|
|
4322
|
-
*
|
|
4323
|
-
* Stream envelope: `backend_start` → incremental events → (`backend_error` on
|
|
4324
|
-
* failure) → `final`. A caller-initiated abort terminates with
|
|
4325
|
-
* `final.status: 'aborted'`; an expired `timeoutMs` deadline with
|
|
4326
|
-
* `final.status: 'failed'` — so cancellation stays distinguishable from a
|
|
4327
|
-
* blown deadline.
|
|
4328
|
-
*
|
|
4329
|
-
* Mid-stream lifecycle work needs NO extra API: the generator is pull-based,
|
|
4330
|
-
* so the producer is suspended between yields and resumes only when the caller
|
|
4331
|
-
* pulls again. A consumer can therefore run arbitrary async work between
|
|
4332
|
-
* events — sync state on each `tool_result`, decide a no-op retry after
|
|
4333
|
-
* draining, run a pre-`done` flush when it receives `final` and BEFORE it
|
|
4334
|
-
* forwards its own terminal event downstream. The interleaving is guaranteed
|
|
4335
|
-
* (and locked by test): nothing is produced past the event the caller is
|
|
4336
|
-
* holding.
|
|
4337
|
-
*
|
|
4338
|
-
* @experimental
|
|
4339
|
-
*/
|
|
4370
|
+
//#region src/runtime/supervise/chat-transport-executor.ts
|
|
4340
4371
|
/**
|
|
4341
|
-
*
|
|
4342
|
-
* `RuntimeStreamEvent` vocabulary incrementally and always ends with a `final`
|
|
4343
|
-
* event carrying the turn's text and usage (`metadata.tokenUsage`,
|
|
4344
|
-
* `metadata.costUsd?`, `metadata.model?`) — on success, failure, abort, and
|
|
4345
|
-
* timeout alike. The generator never throws; failures surface in-band as
|
|
4346
|
-
* `backend_error` + `final` with a typed `error` detail.
|
|
4372
|
+
* A session-owning composition over Runtime's canonical Router tool-loop executor.
|
|
4347
4373
|
*
|
|
4348
|
-
*
|
|
4349
|
-
|
|
4350
|
-
|
|
4351
|
-
const label = backend.kind === "chat" ? backend.backend.kind : backend.kind;
|
|
4352
|
-
const task = {
|
|
4353
|
-
id: `turn-${crypto.randomUUID()}`,
|
|
4354
|
-
intent: prompt
|
|
4355
|
-
};
|
|
4356
|
-
const acc = {
|
|
4357
|
-
deltaText: "",
|
|
4358
|
-
input: 0,
|
|
4359
|
-
output: 0,
|
|
4360
|
-
costUsd: 0
|
|
4361
|
-
};
|
|
4362
|
-
const deadline = deriveTurnSignal(opts.signal, opts.timeoutMs ?? 0);
|
|
4363
|
-
let session;
|
|
4364
|
-
try {
|
|
4365
|
-
session = await startTurnSession(backend, task, prompt, deadline.signal, label);
|
|
4366
|
-
yield {
|
|
4367
|
-
type: "backend_start",
|
|
4368
|
-
task,
|
|
4369
|
-
session,
|
|
4370
|
-
backend: label,
|
|
4371
|
-
timestamp: nowIso()
|
|
4372
|
-
};
|
|
4373
|
-
const inner = backend.kind === "chat" ? driveChatTurn(backend.backend, task, session, prompt, deadline.signal, acc) : driveBoxTurn(backend.kind === "executor" ? await inlineSandboxClient(backend.factory).create() : backend.box, prompt, deadline.signal, backend.agentRunName ?? "agent", acc, {
|
|
4374
|
-
...backend.kind !== "executor" && backend.options ? { options: backend.options } : {},
|
|
4375
|
-
preserveToolParts: opts.preserveToolParts === true,
|
|
4376
|
-
...opts.onRawEvent ? { onRawEvent: opts.onRawEvent } : {}
|
|
4377
|
-
});
|
|
4378
|
-
for await (const event of inner) {
|
|
4379
|
-
yield event;
|
|
4380
|
-
throwIfAborted(deadline.signal);
|
|
4381
|
-
}
|
|
4382
|
-
yield buildFinalEvent(task, session, acc, {
|
|
4383
|
-
status: "completed",
|
|
4384
|
-
reason: "turn completed"
|
|
4385
|
-
});
|
|
4386
|
-
} catch (err) {
|
|
4387
|
-
const callerAborted = opts.signal?.aborted === true;
|
|
4388
|
-
const status = callerAborted ? "aborted" : "failed";
|
|
4389
|
-
const message = err instanceof Error ? err.message : String(err);
|
|
4390
|
-
const error = err instanceof BackendTransportError ? {
|
|
4391
|
-
kind: "transport",
|
|
4392
|
-
message,
|
|
4393
|
-
status: err.status,
|
|
4394
|
-
body: err.body
|
|
4395
|
-
} : {
|
|
4396
|
-
kind: "backend",
|
|
4397
|
-
message
|
|
4398
|
-
};
|
|
4399
|
-
yield {
|
|
4400
|
-
type: "backend_error",
|
|
4401
|
-
task,
|
|
4402
|
-
...session ? { session } : {},
|
|
4403
|
-
backend: label,
|
|
4404
|
-
message,
|
|
4405
|
-
recoverable: !callerAborted,
|
|
4406
|
-
error,
|
|
4407
|
-
timestamp: nowIso()
|
|
4408
|
-
};
|
|
4409
|
-
yield buildFinalEvent(task, session, acc, {
|
|
4410
|
-
status,
|
|
4411
|
-
reason: message,
|
|
4412
|
-
error
|
|
4413
|
-
});
|
|
4414
|
-
} finally {
|
|
4415
|
-
deadline.dispose();
|
|
4416
|
-
}
|
|
4417
|
-
}
|
|
4418
|
-
/**
|
|
4419
|
-
* Drain a `streamAgentTurn` stream (or any `RuntimeStreamEvent` stream that
|
|
4420
|
-
* honors its terminal contract) into the turn summary plus the full event
|
|
4421
|
-
* list. Fail-loud: throws when the stream ends without a terminal `final`
|
|
4422
|
-
* event — a stream that violates the contract must not read as an empty turn.
|
|
4374
|
+
* This module adds conversation persistence for graph-edge `resume` continuity. It does not own
|
|
4375
|
+
* model selection, prompts, generation controls, retries, tool policy, or provider accounting:
|
|
4376
|
+
* those are lowered from one exact `AgentProfile` by `createExecutor({ backend: 'router-tools' })`.
|
|
4423
4377
|
*
|
|
4424
4378
|
* @experimental
|
|
4425
4379
|
*/
|
|
4426
|
-
|
|
4427
|
-
|
|
4428
|
-
|
|
4429
|
-
const final = events.at(-1);
|
|
4430
|
-
if (final?.type !== "final") throw new Error(`collectAgentTurn: stream ended without a terminal 'final' event (last: ${final ? final.type : "none"})`);
|
|
4431
|
-
const metadata = final.metadata ?? {};
|
|
4432
|
-
const tokenUsage = metadata.tokenUsage && typeof metadata.tokenUsage === "object" ? metadata.tokenUsage : {};
|
|
4433
|
-
const usage = {
|
|
4434
|
-
input: finiteNumber(tokenUsage.input) ?? 0,
|
|
4435
|
-
output: finiteNumber(tokenUsage.output) ?? 0
|
|
4436
|
-
};
|
|
4437
|
-
const costUsd = finiteNumber(metadata.costUsd);
|
|
4438
|
-
if (costUsd !== void 0) usage.costUsd = costUsd;
|
|
4439
|
-
if (typeof metadata.model === "string" && metadata.model.length > 0) usage.model = metadata.model;
|
|
4440
|
-
return {
|
|
4441
|
-
finalText: final.text ?? "",
|
|
4442
|
-
usage,
|
|
4443
|
-
events,
|
|
4444
|
-
status: final.status,
|
|
4445
|
-
...final.error ? { error: final.error } : {}
|
|
4446
|
-
};
|
|
4447
|
-
}
|
|
4448
|
-
/** Start the backend's session when it owns one (`chat` kind); mint a local
|
|
4449
|
-
* correlation session otherwise. Box/executor turns carry no server session
|
|
4450
|
-
* here — resume lives in `openSandboxRun`/`SandboxLineage`, not this primitive. */
|
|
4451
|
-
async function startTurnSession(backend, task, prompt, signal, label) {
|
|
4452
|
-
if (backend.kind === "chat" && backend.backend.start) return backend.backend.start({
|
|
4453
|
-
task,
|
|
4454
|
-
message: prompt
|
|
4455
|
-
}, {
|
|
4456
|
-
task,
|
|
4457
|
-
knowledge: emptyReadiness(task),
|
|
4458
|
-
signal
|
|
4459
|
-
});
|
|
4460
|
-
return newRuntimeSession(label);
|
|
4461
|
-
}
|
|
4462
|
-
/**
|
|
4463
|
-
* One turn over a box: `box.streamPrompt` projected through the existing
|
|
4464
|
-
* `mapSandboxEvent` (text/reasoning deltas +
|
|
4465
|
-
* cost-bearing `llm_call`s), plus the opt-in `mapSandboxToolEvent` tool-part
|
|
4466
|
-
* projection. Usage accumulates off the mapped `llm_call` events — the same
|
|
4467
|
-
* fold `sumSandboxUsage` applies. Final text prefers the terminal
|
|
4468
|
-
* `result`/`done`/`final` payload over concatenated deltas, because the
|
|
4469
|
-
* sandbox `message.part.updated` fallback may carry running accumulations.
|
|
4470
|
-
*/
|
|
4471
|
-
async function* driveBoxTurn(box, prompt, signal, agentRunName, acc, cfg) {
|
|
4472
|
-
const callOptions = {
|
|
4473
|
-
...cfg.options ?? {},
|
|
4474
|
-
signal
|
|
4475
|
-
};
|
|
4476
|
-
const stream = box.streamPrompt(prompt, callOptions);
|
|
4477
|
-
const toolParts = cfg.preserveToolParts ? createSandboxToolPartState() : void 0;
|
|
4478
|
-
for await (const event of stream) {
|
|
4479
|
-
if (cfg.onRawEvent) await cfg.onRawEvent(event);
|
|
4480
|
-
const terminalText = terminalTextFromSandboxEvent(event);
|
|
4481
|
-
if (terminalText !== void 0) acc.terminalText = terminalText;
|
|
4482
|
-
if (toolParts) for (const toolEvent of mapSandboxToolEvent(event, toolParts)) yield toolEvent;
|
|
4483
|
-
const mapped = mapSandboxEvent(event, { agentRunName });
|
|
4484
|
-
if (!mapped) continue;
|
|
4485
|
-
foldEvent(mapped, acc, agentRunName);
|
|
4486
|
-
yield mapped;
|
|
4487
|
-
}
|
|
4488
|
-
}
|
|
4489
|
-
/** One turn over an in-process backend: its own `stream()` surface, projected
|
|
4490
|
-
* through the same `normalizeBackendStreamEvent` the task lifecycle applies. */
|
|
4491
|
-
async function* driveChatTurn(backend, task, session, prompt, signal, acc) {
|
|
4492
|
-
const input = {
|
|
4493
|
-
task,
|
|
4494
|
-
message: prompt
|
|
4495
|
-
};
|
|
4496
|
-
const context = {
|
|
4497
|
-
task,
|
|
4498
|
-
knowledge: emptyReadiness(task),
|
|
4499
|
-
session,
|
|
4500
|
-
signal
|
|
4501
|
-
};
|
|
4502
|
-
for await (const raw of backend.stream(input, context)) {
|
|
4503
|
-
const event = normalizeBackendStreamEvent(raw, task, session);
|
|
4504
|
-
foldEvent(event, acc);
|
|
4505
|
-
yield event;
|
|
4506
|
-
}
|
|
4507
|
-
}
|
|
4508
|
-
/** Fold one normalized event into the turn accumulator (text + usage).
|
|
4509
|
-
* `fallbackModelLabel` — a mapper-stamped run label to exclude from
|
|
4510
|
-
* `usage.model` (it is not a backend-reported model). */
|
|
4511
|
-
function foldEvent(event, acc, fallbackModelLabel) {
|
|
4512
|
-
if (event.type === "text_delta") {
|
|
4513
|
-
acc.deltaText += event.text;
|
|
4514
|
-
return;
|
|
4515
|
-
}
|
|
4516
|
-
if (event.type === "llm_call") {
|
|
4517
|
-
acc.input += event.tokensIn ?? 0;
|
|
4518
|
-
acc.output += event.tokensOut ?? 0;
|
|
4519
|
-
acc.costUsd += event.costUsd ?? 0;
|
|
4520
|
-
if (event.model && event.model !== fallbackModelLabel) acc.model = event.model;
|
|
4521
|
-
}
|
|
4522
|
-
}
|
|
4523
|
-
/** Read the final text off a terminal sandbox event, when present. */
|
|
4524
|
-
function terminalTextFromSandboxEvent(event) {
|
|
4525
|
-
if (!event || typeof event !== "object") return void 0;
|
|
4526
|
-
const type = String(event.type ?? "");
|
|
4527
|
-
if (type !== "result" && type !== "done" && type !== "final") return void 0;
|
|
4528
|
-
const data = event.data && typeof event.data === "object" ? event.data : {};
|
|
4529
|
-
for (const key of [
|
|
4530
|
-
"finalText",
|
|
4531
|
-
"text",
|
|
4532
|
-
"response",
|
|
4533
|
-
"content"
|
|
4534
|
-
]) {
|
|
4535
|
-
const value = data[key];
|
|
4536
|
-
if (typeof value === "string") return value;
|
|
4537
|
-
}
|
|
4538
|
-
}
|
|
4539
|
-
function buildFinalEvent(task, session, acc, outcome) {
|
|
4540
|
-
const finalText = acc.terminalText ?? acc.deltaText;
|
|
4380
|
+
/** In-memory, process-local conversation store with detached reads and writes. */
|
|
4381
|
+
function createChatSessionStore() {
|
|
4382
|
+
const sessions = /* @__PURE__ */ new Map();
|
|
4541
4383
|
return {
|
|
4542
|
-
|
|
4543
|
-
|
|
4544
|
-
|
|
4545
|
-
status: outcome.status,
|
|
4546
|
-
reason: outcome.reason,
|
|
4547
|
-
...finalText ? { text: finalText } : {},
|
|
4548
|
-
metadata: {
|
|
4549
|
-
tokenUsage: {
|
|
4550
|
-
input: acc.input,
|
|
4551
|
-
output: acc.output
|
|
4552
|
-
},
|
|
4553
|
-
...acc.costUsd > 0 ? { costUsd: acc.costUsd } : {},
|
|
4554
|
-
...acc.model ? { model: acc.model } : {}
|
|
4384
|
+
load(workerId) {
|
|
4385
|
+
const messages = sessions.get(workerId);
|
|
4386
|
+
return messages === void 0 ? void 0 : structuredClone(messages);
|
|
4555
4387
|
},
|
|
4556
|
-
|
|
4557
|
-
|
|
4558
|
-
};
|
|
4559
|
-
}
|
|
4560
|
-
/** Minimal ready-by-construction readiness report for a requirement-free turn. */
|
|
4561
|
-
function emptyReadiness(task) {
|
|
4562
|
-
return scoreKnowledgeReadiness({
|
|
4563
|
-
taskId: task.id,
|
|
4564
|
-
requirements: []
|
|
4565
|
-
});
|
|
4566
|
-
}
|
|
4567
|
-
function finiteNumber(value) {
|
|
4568
|
-
return typeof value === "number" && Number.isFinite(value) ? value : void 0;
|
|
4569
|
-
}
|
|
4570
|
-
function throwIfAborted(signal) {
|
|
4571
|
-
if (!signal.aborted) return;
|
|
4572
|
-
throw signal.reason instanceof Error ? signal.reason : new Error(String(signal.reason));
|
|
4573
|
-
}
|
|
4574
|
-
/**
|
|
4575
|
-
* Derive the turn's effective abort signal: fires when EITHER the caller's
|
|
4576
|
-
* signal aborts OR the `timeoutMs` deadline elapses. `dispose()` clears the
|
|
4577
|
-
* timer so a finished turn never leaks a pending timeout. `timeoutMs <= 0`
|
|
4578
|
-
* disables the deadline. Node-portable (no `AbortSignal.any`, which needs
|
|
4579
|
-
* >=20.3 — the package floor is >=20).
|
|
4580
|
-
*/
|
|
4581
|
-
function deriveTurnSignal(callerSignal, timeoutMs) {
|
|
4582
|
-
const controller = new AbortController();
|
|
4583
|
-
const timer = timeoutMs > 0 ? setTimeout(() => controller.abort(/* @__PURE__ */ new Error(`agent turn timed out after ${timeoutMs}ms`)), timeoutMs) : void 0;
|
|
4584
|
-
if (timer && typeof timer.unref === "function") timer.unref();
|
|
4585
|
-
const onCallerAbort = () => controller.abort(callerSignal?.reason ?? /* @__PURE__ */ new Error("agent turn aborted"));
|
|
4586
|
-
if (callerSignal) if (callerSignal.aborted) onCallerAbort();
|
|
4587
|
-
else callerSignal.addEventListener("abort", onCallerAbort, { once: true });
|
|
4588
|
-
return {
|
|
4589
|
-
signal: controller.signal,
|
|
4590
|
-
dispose: () => {
|
|
4591
|
-
if (timer) clearTimeout(timer);
|
|
4592
|
-
callerSignal?.removeEventListener("abort", onCallerAbort);
|
|
4388
|
+
save(workerId, messages) {
|
|
4389
|
+
sessions.set(workerId, structuredClone(messages));
|
|
4593
4390
|
}
|
|
4594
4391
|
};
|
|
4595
4392
|
}
|
|
4596
|
-
|
|
4597
|
-
|
|
4598
|
-
|
|
4599
|
-
|
|
4600
|
-
|
|
4601
|
-
|
|
4602
|
-
|
|
4603
|
-
|
|
4604
|
-
|
|
4605
|
-
|
|
4606
|
-
|
|
4607
|
-
|
|
4608
|
-
|
|
4609
|
-
|
|
4610
|
-
|
|
4611
|
-
|
|
4612
|
-
|
|
4613
|
-
|
|
4614
|
-
|
|
4615
|
-
|
|
4616
|
-
|
|
4617
|
-
|
|
4618
|
-
|
|
4619
|
-
|
|
4620
|
-
|
|
4621
|
-
|
|
4622
|
-
|
|
4623
|
-
|
|
4624
|
-
|
|
4625
|
-
|
|
4626
|
-
*/
|
|
4627
|
-
/** The default transport: POST `${url}/chat/completions` with an optional bearer. Fail-loud on
|
|
4628
|
-
* any non-2xx — the status and body head become the settle reason. */
|
|
4629
|
-
function chatCompletionsTransport(opts) {
|
|
4630
|
-
if (typeof opts.url !== "string" || opts.url.length === 0) throw new ValidationError("chatCompletionsTransport: url required");
|
|
4631
|
-
const endpoint = `${opts.url.replace(/\/$/, "")}/chat/completions`;
|
|
4632
|
-
return async (body, signal) => {
|
|
4633
|
-
const res = await fetch(endpoint, {
|
|
4634
|
-
method: "POST",
|
|
4635
|
-
headers: {
|
|
4636
|
-
"content-type": "application/json",
|
|
4637
|
-
...opts.bearer ? { authorization: `Bearer ${opts.bearer}` } : {}
|
|
4393
|
+
function exactProfile(profile, context) {
|
|
4394
|
+
const parsed = agentProfileSchema.safeParse(profile);
|
|
4395
|
+
if (!parsed.success) throw new ValidationError(`${context}: invalid AgentProfile: ${parsed.error.message}`);
|
|
4396
|
+
assertExecutableAgentProfile(parsed.data, context);
|
|
4397
|
+
return parsed.data;
|
|
4398
|
+
}
|
|
4399
|
+
function initialMessages(opts) {
|
|
4400
|
+
if (!opts.resume) return void 0;
|
|
4401
|
+
if (!opts.sessions) throw new ValidationError("chat transport: a 'resume' spawn needs the session store holding the prior conversation");
|
|
4402
|
+
const prior = opts.sessions.load(opts.resume.ofWorker);
|
|
4403
|
+
if (prior === void 0) throw new ValidationError(`chat transport: no recorded conversation for worker '${opts.resume.ofWorker}'`);
|
|
4404
|
+
return prior;
|
|
4405
|
+
}
|
|
4406
|
+
function executorConfig(opts) {
|
|
4407
|
+
const profile = exactProfile(opts.profile, "chat transport");
|
|
4408
|
+
if (!opts.complete && !opts.url) throw new ValidationError("chat transport: url required unless complete is injected");
|
|
4409
|
+
const tools = opts.tools ?? [];
|
|
4410
|
+
for (const tool of tools) if (!tool.spec.function.name || typeof tool.execute !== "function") throw new ValidationError("chat transport: every tool needs spec.function.name and execute");
|
|
4411
|
+
const resumed = initialMessages(opts);
|
|
4412
|
+
return {
|
|
4413
|
+
profile,
|
|
4414
|
+
config: {
|
|
4415
|
+
backend: "router-tools",
|
|
4416
|
+
routerBaseUrl: opts.url ?? "http://injected.invalid",
|
|
4417
|
+
routerKey: opts.bearer ?? (opts.complete ? "injected-transport" : ""),
|
|
4418
|
+
tools: tools.map((tool) => tool.spec),
|
|
4419
|
+
executeToolCall: async (name, args, task) => {
|
|
4420
|
+
const tool = tools.find((candidate) => candidate.spec.function.name === name);
|
|
4421
|
+
if (!tool) throw new ValidationError(`chat transport: unknown tool ${JSON.stringify(name)}`);
|
|
4422
|
+
return tool.execute(args, task);
|
|
4638
4423
|
},
|
|
4639
|
-
|
|
4640
|
-
...
|
|
4641
|
-
|
|
4642
|
-
|
|
4643
|
-
|
|
4424
|
+
...opts.complete ? { complete: opts.complete } : {},
|
|
4425
|
+
...resumed ? { initialMessages: resumed } : {},
|
|
4426
|
+
...opts.sessions && opts.sessionKey ? { onMessages: (messages) => {
|
|
4427
|
+
opts.sessions?.save(opts.sessionKey, messages);
|
|
4428
|
+
} } : {}
|
|
4429
|
+
}
|
|
4644
4430
|
};
|
|
4645
4431
|
}
|
|
4646
|
-
|
|
4647
|
-
|
|
4648
|
-
|
|
4649
|
-
|
|
4650
|
-
|
|
4651
|
-
load: (workerId) => sessions.get(workerId),
|
|
4652
|
-
save: (workerId, messages) => {
|
|
4653
|
-
sessions.set(workerId, structuredClone(messages));
|
|
4654
|
-
}
|
|
4432
|
+
function buildChatTransportExecutor(opts, context) {
|
|
4433
|
+
const { config, profile } = executorConfig(opts);
|
|
4434
|
+
const spec = {
|
|
4435
|
+
profile,
|
|
4436
|
+
harness: null
|
|
4655
4437
|
};
|
|
4438
|
+
return mapExecutorResult(createExecutor(config)(spec, context), (result) => {
|
|
4439
|
+
const raw = result.out;
|
|
4440
|
+
const content = typeof raw?.content === "string" ? raw.content : "";
|
|
4441
|
+
return {
|
|
4442
|
+
outRef: contentAddress({
|
|
4443
|
+
kind: "chat-transport",
|
|
4444
|
+
profile,
|
|
4445
|
+
content
|
|
4446
|
+
}),
|
|
4447
|
+
out: content,
|
|
4448
|
+
...result.verdict ? { verdict: result.verdict } : {}
|
|
4449
|
+
};
|
|
4450
|
+
});
|
|
4656
4451
|
}
|
|
4657
|
-
const CHAT_TRANSPORT_RUNTIME = "chat-transport";
|
|
4658
4452
|
/**
|
|
4659
|
-
* Build
|
|
4660
|
-
*
|
|
4661
|
-
* loop completion → host tool calls → tool messages until the model answers without a tool call
|
|
4662
|
-
* (or the turn cap). Settles with the final assistant text as `out`.
|
|
4663
|
-
*
|
|
4664
|
-
* Fail-loud contract: transport failures (non-2xx, network faults, malformed completions) throw
|
|
4665
|
-
* `ValidationError`, which the scope settles as an INFRA failure (`Settled.down.infra`) — never a
|
|
4666
|
-
* fake success. The accumulated conversation is still recorded before the throw when a store is
|
|
4667
|
-
* configured, because the inference HAPPENED and a resume may continue a failed session (the
|
|
4668
|
-
* kernel deliberately allows resume-after-failure; the seam decides).
|
|
4453
|
+
* Build one exact profile-driven chat executor through `createExecutor`.
|
|
4454
|
+
* Prefer `chatWorkerSeam` for supervised work because it supplies trusted node identity.
|
|
4669
4455
|
*/
|
|
4670
4456
|
function chatTransportExecutor(opts) {
|
|
4671
|
-
|
|
4672
|
-
|
|
4673
|
-
|
|
4674
|
-
for (const tool of opts.tools ?? []) if (typeof tool.spec?.function?.name !== "string" || typeof tool.execute !== "function") throw new ValidationError("chatTransportExecutor: every tools entry needs spec.function.name + execute");
|
|
4675
|
-
const maxTurns = opts.maxTurnsPerShot ?? 200;
|
|
4676
|
-
if (!Number.isInteger(maxTurns) || maxTurns < 1) throw new ValidationError("chatTransportExecutor: maxTurnsPerShot must be a positive integer");
|
|
4677
|
-
if (opts.maxTokens !== void 0 && (!Number.isInteger(opts.maxTokens) || opts.maxTokens < 1)) throw new ValidationError("chatTransportExecutor: maxTokens must be a positive integer");
|
|
4678
|
-
let seed;
|
|
4679
|
-
if (opts.resume) {
|
|
4680
|
-
if (!opts.sessions) throw new ValidationError("chatTransportExecutor: a 'resume' spawn needs `sessions` — the store holding the conversation this shot continues");
|
|
4681
|
-
const prior = opts.sessions.load(opts.resume.ofWorker);
|
|
4682
|
-
if (prior === void 0) throw new ValidationError(`chatTransportExecutor: no recorded conversation for worker '${opts.resume.ofWorker}' — the session store holds only conversations recorded by this process (the kernel’s process-local resume boundary)`);
|
|
4683
|
-
seed = structuredClone(prior);
|
|
4684
|
-
} else seed = opts.system !== void 0 && opts.system.length > 0 ? [{
|
|
4685
|
-
role: "system",
|
|
4686
|
-
content: opts.system
|
|
4687
|
-
}] : [];
|
|
4688
|
-
const transport = opts.complete ?? chatCompletionsTransport({
|
|
4689
|
-
url: opts.url,
|
|
4690
|
-
...opts.bearer ? { bearer: opts.bearer } : {}
|
|
4691
|
-
});
|
|
4692
|
-
const toolSpecs = (opts.tools ?? []).map((tool) => tool.spec);
|
|
4693
|
-
const toolByName = new Map((opts.tools ?? []).map((tool) => [tool.spec.function.name, tool]));
|
|
4694
|
-
const controller = new AbortController();
|
|
4695
|
-
let artifact;
|
|
4696
|
-
let executed = false;
|
|
4697
|
-
const executionId = opts.sessionKey ?? `chat-session-${randomUUID()}`;
|
|
4698
|
-
const attemptId = opts.attemptId ?? newExecutionAttemptId(executionId);
|
|
4699
|
-
const executor = {
|
|
4700
|
-
runtime: CHAT_TRANSPORT_RUNTIME,
|
|
4701
|
-
async execute(task, signal) {
|
|
4702
|
-
if (executed) throw new ValidationError("chatTransportExecutor: execute() called twice on one instance");
|
|
4703
|
-
executed = true;
|
|
4704
|
-
const started = Date.now();
|
|
4705
|
-
const messages = seed;
|
|
4706
|
-
messages.push({
|
|
4707
|
-
role: "user",
|
|
4708
|
-
content: taskToPrompt(task)
|
|
4709
|
-
});
|
|
4710
|
-
const linked = mergeAbortSignals(signal, controller.signal);
|
|
4711
|
-
const tokens = zeroTokenUsage();
|
|
4712
|
-
let tokensKnown = true;
|
|
4713
|
-
let usd = 0;
|
|
4714
|
-
let usdKnown = true;
|
|
4715
|
-
let turns = 0;
|
|
4716
|
-
let lastText = "";
|
|
4717
|
-
try {
|
|
4718
|
-
for (let t = 0; t < maxTurns; t += 1) {
|
|
4719
|
-
const body = {
|
|
4720
|
-
model,
|
|
4721
|
-
messages,
|
|
4722
|
-
...toolSpecs.length > 0 ? {
|
|
4723
|
-
tools: toolSpecs,
|
|
4724
|
-
tool_choice: "auto"
|
|
4725
|
-
} : {},
|
|
4726
|
-
...opts.temperature !== void 0 ? { temperature: opts.temperature } : {},
|
|
4727
|
-
...opts.maxTokens !== void 0 ? { max_tokens: opts.maxTokens } : {}
|
|
4728
|
-
};
|
|
4729
|
-
let raw;
|
|
4730
|
-
try {
|
|
4731
|
-
raw = await transport(body, linked);
|
|
4732
|
-
} catch (cause) {
|
|
4733
|
-
if (cause instanceof Error && cause.name === "AbortError") throw cause;
|
|
4734
|
-
if (cause instanceof ValidationError) throw cause;
|
|
4735
|
-
throw new ValidationError(`chatTransportExecutor: transport failed: ${cause instanceof Error ? cause.message : String(cause)}`);
|
|
4736
|
-
}
|
|
4737
|
-
turns += 1;
|
|
4738
|
-
const data = raw;
|
|
4739
|
-
const usage = data?.usage;
|
|
4740
|
-
if (usage && typeof usage.prompt_tokens === "number" && typeof usage.completion_tokens === "number") {
|
|
4741
|
-
tokens.input += usage.prompt_tokens;
|
|
4742
|
-
tokens.output += usage.completion_tokens;
|
|
4743
|
-
} else tokensKnown = false;
|
|
4744
|
-
const turnCost = typeof usage?.cost === "number" ? usage.cost : typeof usage?.cost_usd === "number" ? usage.cost_usd : void 0;
|
|
4745
|
-
if (turnCost !== void 0) usd += turnCost;
|
|
4746
|
-
else usdKnown = false;
|
|
4747
|
-
const msg = data?.choices?.[0]?.message;
|
|
4748
|
-
if (msg === void 0) throw new ValidationError("chatTransportExecutor: transport returned no choices[0].message");
|
|
4749
|
-
if (typeof msg.content === "string" && msg.content.length > 0) lastText = msg.content;
|
|
4750
|
-
const toolCalls = msg.tool_calls ?? [];
|
|
4751
|
-
if (toolCalls.length === 0 || toolSpecs.length === 0) {
|
|
4752
|
-
messages.push({
|
|
4753
|
-
role: "assistant",
|
|
4754
|
-
content: msg.content ?? ""
|
|
4755
|
-
});
|
|
4756
|
-
break;
|
|
4757
|
-
}
|
|
4758
|
-
messages.push({
|
|
4759
|
-
role: "assistant",
|
|
4760
|
-
content: msg.content ?? "",
|
|
4761
|
-
tool_calls: toolCalls.map((tc, i) => ({
|
|
4762
|
-
id: tc.id ?? `call_${i}`,
|
|
4763
|
-
type: "function",
|
|
4764
|
-
function: {
|
|
4765
|
-
name: tc.function?.name ?? "",
|
|
4766
|
-
arguments: tc.function?.arguments ?? "{}"
|
|
4767
|
-
}
|
|
4768
|
-
}))
|
|
4769
|
-
});
|
|
4770
|
-
for (let i = 0; i < toolCalls.length; i += 1) {
|
|
4771
|
-
const tc = toolCalls[i];
|
|
4772
|
-
const id = tc?.id ?? `call_${i}`;
|
|
4773
|
-
const name = tc?.function?.name ?? "";
|
|
4774
|
-
const tool = toolByName.get(name);
|
|
4775
|
-
if (!tool) {
|
|
4776
|
-
messages.push({
|
|
4777
|
-
role: "tool",
|
|
4778
|
-
tool_call_id: id,
|
|
4779
|
-
content: `error: unknown tool '${name}'`
|
|
4780
|
-
});
|
|
4781
|
-
continue;
|
|
4782
|
-
}
|
|
4783
|
-
let args;
|
|
4784
|
-
try {
|
|
4785
|
-
args = JSON.parse(tc?.function?.arguments ?? "{}");
|
|
4786
|
-
} catch {
|
|
4787
|
-
messages.push({
|
|
4788
|
-
role: "tool",
|
|
4789
|
-
tool_call_id: id,
|
|
4790
|
-
content: "error: tool arguments were not valid JSON"
|
|
4791
|
-
});
|
|
4792
|
-
continue;
|
|
4793
|
-
}
|
|
4794
|
-
let result;
|
|
4795
|
-
try {
|
|
4796
|
-
result = await tool.execute(args, task);
|
|
4797
|
-
} catch (cause) {
|
|
4798
|
-
result = `error: ${cause instanceof Error ? cause.message : String(cause)}`;
|
|
4799
|
-
}
|
|
4800
|
-
messages.push({
|
|
4801
|
-
role: "tool",
|
|
4802
|
-
tool_call_id: id,
|
|
4803
|
-
content: result
|
|
4804
|
-
});
|
|
4805
|
-
}
|
|
4806
|
-
}
|
|
4807
|
-
} finally {
|
|
4808
|
-
if (opts.sessions && opts.sessionKey !== void 0) opts.sessions.save(opts.sessionKey, messages);
|
|
4809
|
-
}
|
|
4810
|
-
const spent = {
|
|
4811
|
-
iterations: turns,
|
|
4812
|
-
tokens,
|
|
4813
|
-
...tokensKnown ? {} : { tokensKnown: false },
|
|
4814
|
-
usd,
|
|
4815
|
-
...usdKnown ? {} : { usdKnown: false },
|
|
4816
|
-
ms: Date.now() - started
|
|
4817
|
-
};
|
|
4818
|
-
artifact = {
|
|
4819
|
-
outRef: contentAddress({
|
|
4820
|
-
kind: "chat-transport",
|
|
4821
|
-
model,
|
|
4822
|
-
content: lastText,
|
|
4823
|
-
turns
|
|
4824
|
-
}),
|
|
4825
|
-
out: lastText,
|
|
4826
|
-
spent
|
|
4827
|
-
};
|
|
4828
|
-
return artifact;
|
|
4829
|
-
},
|
|
4830
|
-
teardown(_grace) {
|
|
4831
|
-
controller.abort();
|
|
4832
|
-
return Promise.resolve({ destroyed: true });
|
|
4833
|
-
},
|
|
4834
|
-
resultArtifact() {
|
|
4835
|
-
if (!artifact) throw new ValidationError("chatTransportExecutor: resultArtifact() read before execute()");
|
|
4836
|
-
return {
|
|
4837
|
-
...artifact,
|
|
4838
|
-
spent: artifact.spent
|
|
4839
|
-
};
|
|
4840
|
-
}
|
|
4841
|
-
};
|
|
4842
|
-
if (opts.profile === void 0) return executor;
|
|
4843
|
-
return attestRuntimeOwnedExecutor(executor, {
|
|
4844
|
-
effectiveProfile: opts.profile,
|
|
4845
|
-
backend: "chat-transport",
|
|
4846
|
-
model: {
|
|
4847
|
-
status: "known",
|
|
4848
|
-
id: model
|
|
4849
|
-
},
|
|
4850
|
-
execution: {
|
|
4851
|
-
kind: "session",
|
|
4852
|
-
id: executionId
|
|
4853
|
-
},
|
|
4854
|
-
materializer: "chat-transport-conversation",
|
|
4855
|
-
plan: {
|
|
4856
|
-
kind: "openai-chat-conversation",
|
|
4857
|
-
model,
|
|
4858
|
-
maxTurnsPerShot: maxTurns,
|
|
4859
|
-
tools: toolSpecs,
|
|
4860
|
-
resumeOf: opts.resume?.ofWorker ?? null
|
|
4861
|
-
}
|
|
4862
|
-
}, {
|
|
4863
|
-
attemptId,
|
|
4864
|
-
binding: {
|
|
4865
|
-
endpoint: opts.complete ? "injected-transport" : opts.url,
|
|
4866
|
-
model,
|
|
4867
|
-
sessionKey: opts.sessionKey ?? null
|
|
4868
|
-
},
|
|
4869
|
-
descriptor: {
|
|
4870
|
-
kind: "chat-transport-session",
|
|
4871
|
-
transport: opts.complete ? "injected" : "http",
|
|
4872
|
-
backend: "chat-transport"
|
|
4873
|
-
}
|
|
4457
|
+
return buildChatTransportExecutor(opts, {
|
|
4458
|
+
signal: new AbortController().signal,
|
|
4459
|
+
seams: {}
|
|
4874
4460
|
});
|
|
4875
4461
|
}
|
|
4876
|
-
/**
|
|
4877
|
-
* The `makeWorkerAgent` seam over {@link chatTransportExecutor} — the continuity consumer
|
|
4878
|
-
* `workerFromBackend` refuses to be. Every spawn becomes one conversation shot: the spawned
|
|
4879
|
-
* profile's system prompt + instructions (which is where a graph's delegates directive lands)
|
|
4880
|
-
* seed a fresh session, and a `'resume'` spawn re-attaches by loading `resume.ofWorker`'s
|
|
4881
|
-
* recorded message list from the seam's session store. Conversations are recorded under the
|
|
4882
|
-
* kernel node id, which is exactly what a later `resume.ofWorker` names.
|
|
4883
|
-
*/
|
|
4462
|
+
/** Session-owning worker factory for graph continuity. */
|
|
4884
4463
|
function chatWorkerSeam(opts) {
|
|
4885
|
-
if (!opts.complete &&
|
|
4464
|
+
if (!opts.complete && !opts.url) throw new ValidationError("chatWorkerSeam: url required unless complete is injected");
|
|
4886
4465
|
const sessions = opts.sessions ?? createChatSessionStore();
|
|
4887
4466
|
return (rawProfile, spawnContext) => {
|
|
4888
|
-
const
|
|
4889
|
-
if (!parsed.success) throw new ValidationError(`chatWorkerSeam: invalid AgentProfile: ${parsed.error.message}`);
|
|
4890
|
-
const profile = parsed.data;
|
|
4891
|
-
const model = concreteProfileModel(profile) ?? concreteModelId(opts.model);
|
|
4892
|
-
if (!model) throw new ValidationError("chatWorkerSeam: no model — set ChatWorkerSeamOptions.model or AgentProfile.model.default");
|
|
4893
|
-
const system = [profile.prompt?.systemPrompt, ...profile.prompt?.instructions ?? []].filter((line) => typeof line === "string" && line.trim().length > 0).join("\n");
|
|
4467
|
+
const profile = exactProfile(rawProfile, "chatWorkerSeam");
|
|
4894
4468
|
return {
|
|
4895
4469
|
name: profile.name ?? "chat-worker",
|
|
4896
4470
|
act: async () => void 0,
|
|
4897
4471
|
executorSpec: {
|
|
4898
4472
|
profile,
|
|
4899
4473
|
harness: null,
|
|
4900
|
-
executorFactory: (executorSpec,
|
|
4901
|
-
const executor =
|
|
4902
|
-
url: opts.url,
|
|
4903
|
-
...opts.bearer !== void 0 ? { bearer: opts.bearer } : {},
|
|
4904
|
-
model,
|
|
4905
|
-
...system.length > 0 ? { system } : {},
|
|
4906
|
-
...opts.tools !== void 0 ? { tools: opts.tools } : {},
|
|
4907
|
-
...opts.temperature !== void 0 ? { temperature: opts.temperature } : {},
|
|
4908
|
-
...opts.maxTokens !== void 0 ? { maxTokens: opts.maxTokens } : {},
|
|
4909
|
-
...opts.maxTurnsPerShot !== void 0 ? { maxTurnsPerShot: opts.maxTurnsPerShot } : {},
|
|
4910
|
-
...opts.complete !== void 0 ? { complete: opts.complete } : {},
|
|
4911
|
-
sessions,
|
|
4912
|
-
...ctx.node?.nodeId !== void 0 ? { sessionKey: ctx.node.nodeId } : {},
|
|
4913
|
-
...spawnContext?.resume !== void 0 ? { resume: spawnContext.resume } : {},
|
|
4474
|
+
executorFactory: (executorSpec, context) => {
|
|
4475
|
+
const executor = buildChatTransportExecutor({
|
|
4914
4476
|
profile: executorSpec.profile,
|
|
4915
|
-
...
|
|
4916
|
-
|
|
4477
|
+
...opts.url ? { url: opts.url } : {},
|
|
4478
|
+
...opts.bearer ? { bearer: opts.bearer } : {},
|
|
4479
|
+
...opts.tools ? { tools: opts.tools } : {},
|
|
4480
|
+
...opts.complete ? { complete: opts.complete } : {},
|
|
4481
|
+
sessions,
|
|
4482
|
+
...context.node?.nodeId ? { sessionKey: context.node.nodeId } : {},
|
|
4483
|
+
...spawnContext?.resume ? { resume: spawnContext.resume } : {}
|
|
4484
|
+
}, context);
|
|
4917
4485
|
return opts.deliverable ? gateOnDeliverable(executor, opts.deliverable) : executor;
|
|
4918
4486
|
}
|
|
4919
4487
|
}
|
|
@@ -4921,457 +4489,6 @@ function chatWorkerSeam(opts) {
|
|
|
4921
4489
|
};
|
|
4922
4490
|
}
|
|
4923
4491
|
//#endregion
|
|
4924
|
-
//#region src/runtime/supervise/graph.ts
|
|
4925
|
-
/**
|
|
4926
|
-
*
|
|
4927
|
-
* `runGraph` — agent graphs: profiles as nodes, registry-backed prompt directives as edges.
|
|
4928
|
-
*
|
|
4929
|
-
* A topology is PLAIN DATA an agent can author in a few lines: nodes are canonical
|
|
4930
|
-
* `AgentProfile`s (the ONLY way a node is described — no role-builder functions), edges are typed
|
|
4931
|
-
* values carrying versioned {@link PromptHandle} directives, `deliverable` (termination) and
|
|
4932
|
-
* `budget` (one conserved pool) are mandatory. Driver↔worker is the two-node cyclic instance;
|
|
4933
|
-
* "agent 3 analyzes 1 and 2 and reports to 1" is ONE edge, not a framework.
|
|
4934
|
-
*
|
|
4935
|
-
* NOT A SECOND SCHEDULER. `runGraph` is an interpretation layer over what already runs:
|
|
4936
|
-
* `supervise()` is the execution core — the same `supervisorAgent`/`driverAgent` machinery,
|
|
4937
|
-
* `makeWorkerAgent` seam, conserved-pool budget, and deliverable-gated settlement every
|
|
4938
|
-
* supervised run uses. (`runAgentRounds` is deliberately NOT the substrate here.) What the graph
|
|
4939
|
-
* layer ADDS is exactly what a bespoke driver loop never
|
|
4940
|
-
* had:
|
|
4941
|
-
*
|
|
4942
|
-
* 1. **Node pinning** — a spawn names a node (`profile.name` = node id) and the node's canonical
|
|
4943
|
-
* profile is what runs; a driver cannot smuggle capabilities into a worker it did not define.
|
|
4944
|
-
* 2. **Observable edges** — every delegates/analyzes traversal lands in an EDGE LEDGER
|
|
4945
|
-
* (`delivered | stripped | empty | unpropagated`, with byte counts), in memory on
|
|
4946
|
-
* the result AND as `edge` events in the run journal. The motivating incident: a filter
|
|
4947
|
-
* silently replaced 1,700-char steering with 241 chars of boilerplate for three rounds and
|
|
4948
|
-
* NO artifact said so — an unobservable edge cannot be trusted and its directive cannot be
|
|
4949
|
-
* optimized.
|
|
4950
|
-
* 3. **Directives as data** — edge text lives in the prompt registry (`<surface>/v<n>`), so every
|
|
4951
|
-
* edge is a versioned optimization target, never prose hardcoded in a builder function.
|
|
4952
|
-
* 4. **Per-edge traversal caps** — the cyclic-graph backstop. A delegates edge whose cap is
|
|
4953
|
-
* exhausted REFUSES further traversals (fail loud), so a cycle cannot spin the pool dry.
|
|
4954
|
-
* 5. **Continuity as data** — a delegates edge may declare `continuity: 'resume'`, so each spawn
|
|
4955
|
-
* after the node's first re-attaches to its latest SETTLED session (the spawn context hands
|
|
4956
|
-
* the executor seam `resume: { ofWorker, sequence }`; the kernel keeps identity, ordering,
|
|
4957
|
-
* ledger truth, and the one conserved pool). Every ledger row states how its hop continued:
|
|
4958
|
-
* `'fresh' | 'resume'` for spawns, `'steer'` for mid-run deliveries — fresh respawns, session
|
|
4959
|
-
* resumes, and live steers are all plain data, each a ledgered fact.
|
|
4960
|
-
*
|
|
4961
|
-
* ORACLES ARE ENVIRONMENT, NEVER WORKERS. Graders/verifiers must not be spawnable in the graph —
|
|
4962
|
-
* a delegates edge to them leaks the rubric. An `analyzes` edge names its analyst in one of two
|
|
4963
|
-
* forms: a LENS id from the environment's registry (a pure function over trace evidence), or the
|
|
4964
|
-
* id of a graph NODE — a tool-equipped analyst AGENT spawned on each matching settle with the
|
|
4965
|
-
* node's pinned profile, whose settle output IS the findings. Either way the oracle doctrine
|
|
4966
|
-
* holds: an analyst node can never be a delegates target (refused loudly), so no driver can hand
|
|
4967
|
-
* it work, and an id living in both the registry and the nodes is refused as ambiguous.
|
|
4968
|
-
*
|
|
4969
|
-
* @experimental
|
|
4970
|
-
*/
|
|
4971
|
-
/** Default per-edge traversal cap — the cyclic-graph backstop when an edge names none. */
|
|
4972
|
-
const defaultEdgeTraversalCap = 32;
|
|
4973
|
-
/** A delegates edge exhausted its traversal cap and the run produced no winner: the cap, not the
|
|
4974
|
-
* task, ended it. Carries the full evidence so failing loud loses nothing. */
|
|
4975
|
-
var GraphEdgeCapError = class extends Error {
|
|
4976
|
-
exhaustedEdges;
|
|
4977
|
-
ledger;
|
|
4978
|
-
result;
|
|
4979
|
-
constructor(exhaustedEdges, ledger, result) {
|
|
4980
|
-
super(`runGraph: edge traversal cap exhausted on ${exhaustedEdges.join(", ")} and the run delivered no winner — the cap (the cyclic-graph backstop), not the task, ended this run. Raise maxTraversals on the edge or fix the cycle; the full edge ledger and the supervised result ride on this error.`);
|
|
4981
|
-
this.name = "GraphEdgeCapError";
|
|
4982
|
-
this.exhaustedEdges = exhaustedEdges;
|
|
4983
|
-
this.ledger = ledger;
|
|
4984
|
-
this.result = result;
|
|
4985
|
-
}
|
|
4986
|
-
};
|
|
4987
|
-
function edgeId(edge) {
|
|
4988
|
-
return edge.kind === "delegates" ? `delegates:${edge.from}->${edge.to}` : `analyzes:${edge.analyst}:${edge.over.join("+")}->${edge.to}`;
|
|
4989
|
-
}
|
|
4990
|
-
/** Validate the graph and resolve every directive BEFORE any compute is spent — an invalid
|
|
4991
|
-
* topology or an unknown directive is a configuration fault, never a mid-run surprise. */
|
|
4992
|
-
function validateGraph(graph, registry, analysts) {
|
|
4993
|
-
if (!Array.isArray(graph.nodes) || graph.nodes.length === 0) throw new ValidationError("runGraph: graph.nodes must be a non-empty array");
|
|
4994
|
-
if (!Array.isArray(graph.edges) || graph.edges.length === 0) throw new ValidationError("runGraph: graph.edges must be a non-empty array");
|
|
4995
|
-
if (typeof graph.deliverable?.check !== "function") throw new ValidationError("runGraph: graph.deliverable is mandatory (termination oracle)");
|
|
4996
|
-
if (typeof graph.budget !== "object" || graph.budget === null) throw new ValidationError("runGraph: graph.budget is mandatory (the conserved pool)");
|
|
4997
|
-
const byId = /* @__PURE__ */ new Map();
|
|
4998
|
-
for (const node of graph.nodes) {
|
|
4999
|
-
if (typeof node.id !== "string" || node.id.length === 0) throw new ValidationError("runGraph: every node needs a non-empty string id");
|
|
5000
|
-
if (byId.has(node.id)) throw new ValidationError(`runGraph: duplicate node id '${node.id}'`);
|
|
5001
|
-
const parsed = agentProfileSchema.safeParse(node.profile);
|
|
5002
|
-
if (!parsed.success) throw new ValidationError(`runGraph: node '${node.id}' has an invalid AgentProfile: ${parsed.error.message}`);
|
|
5003
|
-
if (node.profile.name !== node.id) throw new ValidationError(`runGraph: node '${node.id}' has profile.name ${JSON.stringify(node.profile.name)} — profile.name IS the node identity (node pinning and analyst routing match on it) and must equal the node id`);
|
|
5004
|
-
byId.set(node.id, node);
|
|
5005
|
-
}
|
|
5006
|
-
const requireNode = (id, where) => {
|
|
5007
|
-
const node = byId.get(id);
|
|
5008
|
-
if (!node) throw new ValidationError(`runGraph: ${where} references unknown node '${id}'`);
|
|
5009
|
-
return node;
|
|
5010
|
-
};
|
|
5011
|
-
const delegates = graph.edges.filter((edge) => edge.kind === "delegates");
|
|
5012
|
-
const analyzes = graph.edges.filter((edge) => edge.kind === "analyzes");
|
|
5013
|
-
if (delegates.length === 0) throw new ValidationError("runGraph: at least one delegates edge is required (who spawns whom)");
|
|
5014
|
-
for (const edge of graph.edges) registry.resolve(edge.directive);
|
|
5015
|
-
for (const edge of delegates) {
|
|
5016
|
-
requireNode(edge.from, edgeId(edge));
|
|
5017
|
-
requireNode(edge.to, edgeId(edge));
|
|
5018
|
-
if (edge.from === edge.to) throw new ValidationError(`runGraph: ${edgeId(edge)} delegates to itself — the driver↔worker cycle is the settle-return loop, not a self-edge`);
|
|
5019
|
-
if (edge.continuity !== void 0 && edge.continuity !== "fresh" && edge.continuity !== "resume") throw new ValidationError(`runGraph: ${edgeId(edge)} has invalid continuity ${JSON.stringify(edge.continuity)} — a delegates edge's continuity is 'fresh' or 'resume'`);
|
|
5020
|
-
}
|
|
5021
|
-
const delegatedTo = new Set(delegates.map((edge) => edge.to));
|
|
5022
|
-
const roots = [...new Set(delegates.map((edge) => edge.from))].filter((id) => !delegatedTo.has(id));
|
|
5023
|
-
if (roots.length !== 1) throw new ValidationError(`runGraph: expected exactly ONE root (a node that delegates and is never delegated to), found ${roots.length === 0 ? "none — delegates edges form a cycle with no entry" : roots.join(", ")}. P0 executes driver↔worker(s); nested driver graphs are P3.`);
|
|
5024
|
-
const root = requireNode(roots[0], "root resolution");
|
|
5025
|
-
for (const edge of delegates) if (edge.from !== root.id) throw new ValidationError(`runGraph: ${edgeId(edge)} delegates from a non-root node — P0 executes one driver over its workers (the 2-node cyclic case, star-generalized); deeper delegation is P3`);
|
|
5026
|
-
const analystIds = /* @__PURE__ */ new Set();
|
|
5027
|
-
const analystNodes = /* @__PURE__ */ new Map();
|
|
5028
|
-
for (const edge of analyzes) {
|
|
5029
|
-
if (edge.continuity !== void 0) throw new ValidationError(`runGraph: ${edgeId(edge)} carries continuity — analysts are spawned by the analyst machinery (every analyst run is a fresh session over settled evidence), so continuity is a delegates-edge axis only`);
|
|
5030
|
-
if (analystIds.has(edge.analyst)) throw new ValidationError(`runGraph: two analyzes edges share analyst '${edge.analyst}' — one analyzes edge per analyst lens (traversals are ledgered by analyst id; a second edge would silently absorb the first's). Register the lens under a second id for a second edge.`);
|
|
5031
|
-
analystIds.add(edge.analyst);
|
|
5032
|
-
const analystNode = byId.get(edge.analyst);
|
|
5033
|
-
const inRegistry = analysts?.kinds.some((kind) => kind.id === edge.analyst) === true;
|
|
5034
|
-
if (analystNode !== void 0 && inRegistry) throw new ValidationError(`runGraph: ${edgeId(edge)} analyst '${edge.analyst}' is BOTH a graph node and a lens in the analysts registry — the id alone distinguishes the two analyst forms, so this is ambiguous; rename the node or register the lens under another id`);
|
|
5035
|
-
if (analystNode !== void 0) {
|
|
5036
|
-
if (analystNode.id === root.id) throw new ValidationError(`runGraph: ${edgeId(edge)} names the ROOT as its analyst — the root is the driver; give the analyst its own node with no delegates edge pointing at it`);
|
|
5037
|
-
if (delegatedTo.has(analystNode.id)) throw new ValidationError(`runGraph: ${edgeId(edge)} names node '${edge.analyst}' as its analyst, but that node is a delegates target — oracle doctrine: an analyst is never delegated to. An analyst NODE is legal only with NO delegates edge pointing at it; give the analyst its own delegates-free node or pass a lens id from RunGraphOptions.analysts.`);
|
|
5038
|
-
analystNodes.set(analystNode.id, analystNode);
|
|
5039
|
-
} else if (!analysts) throw new ValidationError(`runGraph: ${edgeId(edge)} analyst '${edge.analyst}' is not a graph node, and no RunGraphOptions.analysts registry was provided to resolve it as a lens`);
|
|
5040
|
-
else if (!inRegistry) throw new ValidationError(`runGraph: ${edgeId(edge)} analyst '${edge.analyst}' is neither a graph node nor in the analysts registry (known lenses: ${analysts.kinds.map((kind) => kind.id).join(", ") || "none"})`);
|
|
5041
|
-
if (edge.over.length === 0) throw new ValidationError(`runGraph: ${edgeId(edge)} must analyze at least one node`);
|
|
5042
|
-
for (const over of edge.over) {
|
|
5043
|
-
requireNode(over, edgeId(edge));
|
|
5044
|
-
if (over === root.id) throw new ValidationError(`runGraph: ${edgeId(edge)} analyzes the ROOT — analysts observe settled workers, and the root never settles as one, so this edge would silently never fire; list delegates-target nodes only`);
|
|
5045
|
-
}
|
|
5046
|
-
requireNode(edge.to, edgeId(edge));
|
|
5047
|
-
}
|
|
5048
|
-
for (const edge of analyzes) for (const over of edge.over) if (analystNodes.has(over)) throw new ValidationError(`runGraph: ${edgeId(edge)} analyzes '${over}', which is an analyst node — an analyst run settles as a finding, never as a worker, so this edge would silently never fire; analyst nodes are not analyzable`);
|
|
5049
|
-
const workers = /* @__PURE__ */ new Map();
|
|
5050
|
-
const delegatesByWorker = /* @__PURE__ */ new Map();
|
|
5051
|
-
for (const edge of delegates) {
|
|
5052
|
-
if (delegatesByWorker.has(edge.to)) throw new ValidationError(`runGraph: node '${edge.to}' is the target of two delegates edges — one delegation directive per worker node (version the directive instead of forking the edge)`);
|
|
5053
|
-
delegatesByWorker.set(edge.to, edge);
|
|
5054
|
-
workers.set(edge.to, requireNode(edge.to, edgeId(edge)));
|
|
5055
|
-
}
|
|
5056
|
-
for (const node of graph.nodes) if (node.id !== root.id && !workers.has(node.id) && !analystNodes.has(node.id)) throw new ValidationError(`runGraph: node '${node.id}' has no delegates edge to it — an unreachable node never runs`);
|
|
5057
|
-
return {
|
|
5058
|
-
root,
|
|
5059
|
-
workers,
|
|
5060
|
-
delegatesByWorker,
|
|
5061
|
-
analyzes,
|
|
5062
|
-
analystNodes
|
|
5063
|
-
};
|
|
5064
|
-
}
|
|
5065
|
-
const byteLength = (text) => Buffer.byteLength(text, "utf8");
|
|
5066
|
-
function stringifyPayload(payload) {
|
|
5067
|
-
if (typeof payload === "string") return payload;
|
|
5068
|
-
try {
|
|
5069
|
-
return JSON.stringify(payload) ?? String(payload);
|
|
5070
|
-
} catch {
|
|
5071
|
-
return String(payload);
|
|
5072
|
-
}
|
|
5073
|
-
}
|
|
5074
|
-
/**
|
|
5075
|
-
* Execute an {@link AgentGraph}. The root node becomes the supervisor (`supervise()` — the
|
|
5076
|
-
* execution core), each worker node is spawnable BY NODE ID (`spawn_agent` with
|
|
5077
|
-
* `profile: { name: '<node id>' }`; the node's canonical profile is pinned by the graph), each
|
|
5078
|
-
* delegates directive is appended to the worker profile's `prompt.instructions` per traversal,
|
|
5079
|
-
* and each analyzes edge becomes an analyst-on-settle route with a real DESTINATION. Every
|
|
5080
|
-
* traversal is ledgered and journaled.
|
|
5081
|
-
*/
|
|
5082
|
-
function runGraph(graph, opts) {
|
|
5083
|
-
const registry = opts.registry ?? kernelPromptRegistry();
|
|
5084
|
-
const { root, workers, delegatesByWorker, analyzes, analystNodes } = validateGraph(graph, registry, opts.analysts);
|
|
5085
|
-
if (!opts.backend && !opts.makeWorkerAgent) throw new ValidationError("runGraph: provide opts.backend (where nodes run) or opts.makeWorkerAgent");
|
|
5086
|
-
const journal = opts.journal ?? new InMemorySpawnJournal();
|
|
5087
|
-
const blobs = opts.blobs ?? new InMemoryResultBlobStore();
|
|
5088
|
-
const runId = opts.runId ?? `graph-${canonicalCandidateDigest(graph.nodes.map((n) => n.id)).slice(7, 19)}`;
|
|
5089
|
-
const now = opts.now ?? Date.now;
|
|
5090
|
-
const ledger = [];
|
|
5091
|
-
const journaled = /* @__PURE__ */ new Set();
|
|
5092
|
-
const traversalCounts = /* @__PURE__ */ new Map();
|
|
5093
|
-
const exhausted = /* @__PURE__ */ new Set();
|
|
5094
|
-
const exhaustedDelegates = /* @__PURE__ */ new Set();
|
|
5095
|
-
const journalWrites = [];
|
|
5096
|
-
let ledgerSeq = 0;
|
|
5097
|
-
const appendJournal = (entry, nodeIdForEvent) => {
|
|
5098
|
-
if (journaled.has(entry)) return Promise.resolve();
|
|
5099
|
-
journaled.add(entry);
|
|
5100
|
-
const write = journal.appendEvent(runId, {
|
|
5101
|
-
kind: "edge",
|
|
5102
|
-
id: nodeIdForEvent,
|
|
5103
|
-
edge: {
|
|
5104
|
-
kind: entry.kind,
|
|
5105
|
-
from: entry.from,
|
|
5106
|
-
to: entry.to,
|
|
5107
|
-
directive: entry.directive
|
|
5108
|
-
},
|
|
5109
|
-
traversal: entry.traversal,
|
|
5110
|
-
outcome: entry.outcome,
|
|
5111
|
-
continuity: entry.continuity,
|
|
5112
|
-
bytes: entry.bytes,
|
|
5113
|
-
...entry.reason !== void 0 ? { reason: entry.reason } : {},
|
|
5114
|
-
seq: ledgerSeq++,
|
|
5115
|
-
at: new Date(now()).toISOString()
|
|
5116
|
-
});
|
|
5117
|
-
journalWrites.push(write);
|
|
5118
|
-
return write;
|
|
5119
|
-
};
|
|
5120
|
-
const record = (entry, journalNow) => {
|
|
5121
|
-
const count = (traversalCounts.get(entry.edge) ?? 0) + 1;
|
|
5122
|
-
traversalCounts.set(entry.edge, count);
|
|
5123
|
-
const row = {
|
|
5124
|
-
...entry,
|
|
5125
|
-
traversal: count
|
|
5126
|
-
};
|
|
5127
|
-
ledger.push(row);
|
|
5128
|
-
if (journalNow) appendJournal(row, row.workerId ?? `graph:${row.to}`);
|
|
5129
|
-
return row;
|
|
5130
|
-
};
|
|
5131
|
-
const makeLeaf = opts.makeWorkerAgent ?? workerFromBackend(opts.backend, graph.deliverable);
|
|
5132
|
-
const nodeByWorkerId = /* @__PURE__ */ new Map();
|
|
5133
|
-
const pendingByAssignment = /* @__PURE__ */ new Map();
|
|
5134
|
-
const graphWorker = (authoredProfile, spawnContext) => {
|
|
5135
|
-
const requested = typeof authoredProfile?.name === "string" ? authoredProfile.name : void 0;
|
|
5136
|
-
if (spawnContext?.analyst !== void 0) {
|
|
5137
|
-
const analystNode = analystNodes.get(spawnContext.analyst);
|
|
5138
|
-
if (!analystNode || requested !== analystNode.id) throw new ValidationError(`runGraph: analyst run for ${JSON.stringify(spawnContext.analyst)} does not name an analyst node of this graph (analyst nodes: ${[...analystNodes.keys()].join(", ") || "none"})`);
|
|
5139
|
-
return makeLeaf(analystNode.profile, spawnContext);
|
|
5140
|
-
}
|
|
5141
|
-
const node = requested !== void 0 ? workers.get(requested) : void 0;
|
|
5142
|
-
if (!node) throw new ValidationError(`runGraph: spawn_agent named profile ${JSON.stringify(requested)} which is not a worker node of this graph (nodes: ${[...workers.keys()].join(", ")}). Spawn by node id: profile.name selects the node; the node profile itself is pinned by the graph.`);
|
|
5143
|
-
const edge = delegatesByWorker.get(node.id);
|
|
5144
|
-
const id = edgeId(edge);
|
|
5145
|
-
const cap = edge.maxTraversals ?? 32;
|
|
5146
|
-
const used = traversalCounts.get(id) ?? 0;
|
|
5147
|
-
const spawnContinuity = spawnContext?.continuity ?? "fresh";
|
|
5148
|
-
if (used >= cap) {
|
|
5149
|
-
exhausted.add(id);
|
|
5150
|
-
exhaustedDelegates.add(id);
|
|
5151
|
-
record({
|
|
5152
|
-
edge: id,
|
|
5153
|
-
kind: "delegates",
|
|
5154
|
-
from: edge.from,
|
|
5155
|
-
to: edge.to,
|
|
5156
|
-
directive: formatPromptHandle(edge.directive),
|
|
5157
|
-
outcome: "unpropagated",
|
|
5158
|
-
continuity: spawnContinuity,
|
|
5159
|
-
bytes: 0,
|
|
5160
|
-
reason: `traversal-cap-exhausted (max ${cap})`
|
|
5161
|
-
}, true);
|
|
5162
|
-
throw new ValidationError(`runGraph: delegates edge ${id} exhausted its traversal cap (${cap}) — the cyclic-graph backstop refused this spawn`);
|
|
5163
|
-
}
|
|
5164
|
-
const directiveText = registry.resolve(edge.directive).text;
|
|
5165
|
-
const taskText = stringifyPayload(spawnContext?.task);
|
|
5166
|
-
const bytes = byteLength(directiveText) + byteLength(taskText);
|
|
5167
|
-
const row = record({
|
|
5168
|
-
edge: id,
|
|
5169
|
-
kind: "delegates",
|
|
5170
|
-
from: edge.from,
|
|
5171
|
-
to: edge.to,
|
|
5172
|
-
directive: formatPromptHandle(edge.directive),
|
|
5173
|
-
outcome: bytes === 0 ? "empty" : "delivered",
|
|
5174
|
-
continuity: spawnContinuity,
|
|
5175
|
-
bytes,
|
|
5176
|
-
...bytes === 0 ? { reason: "no directive text and no task payload" } : {}
|
|
5177
|
-
}, false);
|
|
5178
|
-
if (spawnContext?.assignmentId !== void 0) pendingByAssignment.set(spawnContext.assignmentId, row);
|
|
5179
|
-
else appendJournal(row, `graph:${row.to}`);
|
|
5180
|
-
const pinned = directiveText.length === 0 ? node.profile : {
|
|
5181
|
-
...node.profile,
|
|
5182
|
-
prompt: {
|
|
5183
|
-
...node.profile.prompt ?? {},
|
|
5184
|
-
instructions: [...node.profile.prompt?.instructions ?? [], directiveText]
|
|
5185
|
-
}
|
|
5186
|
-
};
|
|
5187
|
-
return makeLeaf(pinned, spawnContext);
|
|
5188
|
-
};
|
|
5189
|
-
const routes = analyzes.map((edge) => {
|
|
5190
|
-
const analystNode = analystNodes.get(edge.analyst);
|
|
5191
|
-
if (analystNode) return {
|
|
5192
|
-
kind: edge.analyst,
|
|
5193
|
-
over: edge.over,
|
|
5194
|
-
agent: analystNode.profile,
|
|
5195
|
-
directive: registry.resolve(edge.directive).text,
|
|
5196
|
-
...edge.to === root.id ? {} : { to: edge.to }
|
|
5197
|
-
};
|
|
5198
|
-
return edge.to === root.id ? {
|
|
5199
|
-
kind: edge.analyst,
|
|
5200
|
-
over: edge.over
|
|
5201
|
-
} : {
|
|
5202
|
-
kind: edge.analyst,
|
|
5203
|
-
over: edge.over,
|
|
5204
|
-
to: edge.to,
|
|
5205
|
-
directive: registry.resolve(edge.directive).text
|
|
5206
|
-
};
|
|
5207
|
-
});
|
|
5208
|
-
const driverAnalyzesBriefs = analyzes.filter((edge) => edge.to === root.id).map((edge) => analystNodes.has(edge.analyst) ? `Findings from analyst '${edge.analyst}' (a tool-equipped analyst agent node, over: ${edge.over.join(", ")}) will arrive as finding events.` : `Findings from analyst '${edge.analyst}' (over: ${edge.over.join(", ")}) will arrive as finding events.\n${registry.resolve(edge.directive).text}`);
|
|
5209
|
-
const continuityByProfile = {};
|
|
5210
|
-
for (const [nodeId, edge] of delegatesByWorker) if (edge.continuity !== void 0) continuityByProfile[nodeId] = edge.continuity;
|
|
5211
|
-
const graphBrief = [
|
|
5212
|
-
"AGENT GRAPH: you are the driver node of a fixed topology. You may spawn ONLY these worker",
|
|
5213
|
-
"nodes, by EXACT name (spawn_agent with profile: { name: '<node id>' }; the node's full",
|
|
5214
|
-
"profile is pinned by the graph — any other profile fields you author are ignored):",
|
|
5215
|
-
...[...workers.values()].map((node) => {
|
|
5216
|
-
const edge = delegatesByWorker.get(node.id);
|
|
5217
|
-
const cap = edge.maxTraversals ?? 32;
|
|
5218
|
-
const description = typeof node.profile.description === "string" && node.profile.description.length > 0 ? ` — ${node.profile.description}` : "";
|
|
5219
|
-
const continuityNote = edge.continuity === "resume" ? "; continuity: resume — each spawn after the first re-attaches to this node's latest settled session (spawn again to continue it; steer while it is live)" : "";
|
|
5220
|
-
return `- '${node.id}'${description} (delegation cap: ${cap} traversals${continuityNote})`;
|
|
5221
|
-
}),
|
|
5222
|
-
...driverAnalyzesBriefs.length > 0 ? ["", ...driverAnalyzesBriefs] : []
|
|
5223
|
-
].join("\n");
|
|
5224
|
-
const rootProfile = {
|
|
5225
|
-
...root.profile,
|
|
5226
|
-
prompt: {
|
|
5227
|
-
...root.profile.prompt ?? {},
|
|
5228
|
-
instructions: [...root.profile.prompt?.instructions ?? [], graphBrief]
|
|
5229
|
-
}
|
|
5230
|
-
};
|
|
5231
|
-
const strippedByDigest = /* @__PURE__ */ new Map();
|
|
5232
|
-
const authorizeMessage = opts.authorizeMessage ? (input) => {
|
|
5233
|
-
const decision = opts.authorizeMessage(input);
|
|
5234
|
-
if (decision.instruction !== input.instruction) strippedByDigest.set(canonicalCandidateDigest(decision.instruction), { composedBytes: byteLength(input.instruction) });
|
|
5235
|
-
return decision;
|
|
5236
|
-
} : void 0;
|
|
5237
|
-
const routedAnalyzesByAnalyst = /* @__PURE__ */ new Map();
|
|
5238
|
-
const driverAnalyzesByAnalyst = /* @__PURE__ */ new Map();
|
|
5239
|
-
for (const edge of analyzes) (edge.to === root.id ? driverAnalyzesByAnalyst : routedAnalyzesByAnalyst).set(edge.analyst, edge);
|
|
5240
|
-
const analyzesCapReached = (edge) => {
|
|
5241
|
-
const cap = edge.maxTraversals ?? 32;
|
|
5242
|
-
if ((traversalCounts.get(edgeId(edge)) ?? 0) < cap) return false;
|
|
5243
|
-
exhausted.add(edgeId(edge));
|
|
5244
|
-
return true;
|
|
5245
|
-
};
|
|
5246
|
-
const ledgerAnalyzes = (edge, outcome, bytes, reason, workerId) => {
|
|
5247
|
-
const capped = analyzesCapReached(edge);
|
|
5248
|
-
record({
|
|
5249
|
-
edge: edgeId(edge),
|
|
5250
|
-
kind: "analyzes",
|
|
5251
|
-
from: edge.over.join("+"),
|
|
5252
|
-
to: edge.to,
|
|
5253
|
-
directive: formatPromptHandle(edge.directive),
|
|
5254
|
-
outcome: capped ? "unpropagated" : outcome,
|
|
5255
|
-
continuity: "steer",
|
|
5256
|
-
bytes,
|
|
5257
|
-
...capped ? { reason: `traversal-cap-exhausted (max ${edge.maxTraversals ?? 32})` } : reason !== void 0 ? { reason } : {},
|
|
5258
|
-
...workerId !== void 0 ? { workerId } : {}
|
|
5259
|
-
}, true);
|
|
5260
|
-
};
|
|
5261
|
-
const onCoordinationEvent = async (_context, _eventId, recordEnvelope) => {
|
|
5262
|
-
const event = recordEnvelope.event;
|
|
5263
|
-
if (event.type === "finding") {
|
|
5264
|
-
const edge = driverAnalyzesByAnalyst.get(event.finding.analyst);
|
|
5265
|
-
if (!edge) return;
|
|
5266
|
-
const sourceNode = nodeByWorkerId.get(event.finding.fromWorker);
|
|
5267
|
-
if (sourceNode === void 0 || !edge.over.includes(sourceNode)) return;
|
|
5268
|
-
const findingsText = event.finding.findings === void 0 ? "" : stringifyPayload(event.finding.findings);
|
|
5269
|
-
const directiveBytes = byteLength(registry.resolve(edge.directive).text);
|
|
5270
|
-
const empty = findingsText.length === 0;
|
|
5271
|
-
ledgerAnalyzes(edge, empty ? "empty" : "delivered", directiveBytes + byteLength(findingsText), empty ? "analyst returned no findings" : void 0, event.finding.fromWorker);
|
|
5272
|
-
return;
|
|
5273
|
-
}
|
|
5274
|
-
if (event.type === "steer") {
|
|
5275
|
-
const down = event.down;
|
|
5276
|
-
if (event.analyst !== void 0) {
|
|
5277
|
-
const edge = routedAnalyzesByAnalyst.get(event.analyst);
|
|
5278
|
-
if (!edge) return;
|
|
5279
|
-
ledgerAnalyzes(edge, down.delivered ? "delivered" : "unpropagated", byteLength(down.instruction), down.delivered ? void 0 : down.outcome, down.toWorker);
|
|
5280
|
-
return;
|
|
5281
|
-
}
|
|
5282
|
-
const nodeId = nodeByWorkerId.get(down.toWorker);
|
|
5283
|
-
if (nodeId === void 0) return;
|
|
5284
|
-
const edge = delegatesByWorker.get(nodeId);
|
|
5285
|
-
if (!edge) return;
|
|
5286
|
-
const stripped = strippedByDigest.get(down.instructionDigest);
|
|
5287
|
-
record({
|
|
5288
|
-
edge: edgeId(edge),
|
|
5289
|
-
kind: "delegates",
|
|
5290
|
-
from: edge.from,
|
|
5291
|
-
to: edge.to,
|
|
5292
|
-
directive: formatPromptHandle(edge.directive),
|
|
5293
|
-
outcome: !down.delivered ? "unpropagated" : stripped ? "stripped" : "delivered",
|
|
5294
|
-
continuity: "steer",
|
|
5295
|
-
bytes: byteLength(down.instruction),
|
|
5296
|
-
...!down.delivered ? { reason: down.outcome } : stripped ? { reason: `authorization narrowed ${stripped.composedBytes} composed bytes` } : {},
|
|
5297
|
-
workerId: down.toWorker
|
|
5298
|
-
}, true);
|
|
5299
|
-
}
|
|
5300
|
-
};
|
|
5301
|
-
const hooks = composeRuntimeHooks({ onEvent: (event) => {
|
|
5302
|
-
if (event.target !== "agent.spawn" || event.phase !== "after") return;
|
|
5303
|
-
const payload = event.payload;
|
|
5304
|
-
if (typeof payload?.childId !== "string" || typeof payload.assignmentId !== "string") return;
|
|
5305
|
-
const pending = pendingByAssignment.get(payload.assignmentId);
|
|
5306
|
-
if (!pending) return;
|
|
5307
|
-
pendingByAssignment.delete(payload.assignmentId);
|
|
5308
|
-
const bound = {
|
|
5309
|
-
...pending,
|
|
5310
|
-
workerId: payload.childId
|
|
5311
|
-
};
|
|
5312
|
-
ledger[ledger.indexOf(pending)] = bound;
|
|
5313
|
-
nodeByWorkerId.set(payload.childId, bound.to);
|
|
5314
|
-
return appendJournal(bound, payload.childId);
|
|
5315
|
-
} }, opts.hooks);
|
|
5316
|
-
const start = async () => {
|
|
5317
|
-
const result = await supervise(rootProfile, graphTask(graph, root), {
|
|
5318
|
-
budget: graph.budget,
|
|
5319
|
-
deliverable: graph.deliverable,
|
|
5320
|
-
makeWorkerAgent: graphWorker,
|
|
5321
|
-
journal,
|
|
5322
|
-
blobs,
|
|
5323
|
-
runId,
|
|
5324
|
-
hooks,
|
|
5325
|
-
onCoordinationEvent,
|
|
5326
|
-
...routes.length > 0 ? {
|
|
5327
|
-
analyzeOnSettle: routes,
|
|
5328
|
-
...opts.analysts ? { analysts: opts.analysts } : {}
|
|
5329
|
-
} : {},
|
|
5330
|
-
...Object.keys(continuityByProfile).length > 0 ? { continuityByProfile } : {},
|
|
5331
|
-
...opts.watchWorkers ? { watchWorkers: opts.watchWorkers } : {},
|
|
5332
|
-
...opts.router ? { router: opts.router } : {},
|
|
5333
|
-
...opts.brain ? { brain: opts.brain } : {},
|
|
5334
|
-
...authorizeMessage ? { authorizeMessage } : {},
|
|
5335
|
-
...opts.perWorker ? { perWorker: opts.perWorker } : {},
|
|
5336
|
-
...opts.maxTurns !== void 0 ? { maxTurns: opts.maxTurns } : {},
|
|
5337
|
-
...opts.maxLiveWorkers !== void 0 ? { maxLiveWorkers: opts.maxLiveWorkers } : {},
|
|
5338
|
-
...opts.signal ? { signal: opts.signal } : {},
|
|
5339
|
-
...opts.now ? { now: opts.now } : {},
|
|
5340
|
-
...opts.otel ? { otel: opts.otel } : {},
|
|
5341
|
-
...opts.stallAfterMs !== void 0 ? { stallAfterMs: opts.stallAfterMs } : {},
|
|
5342
|
-
...opts.allowedModels ? { allowedModels: opts.allowedModels } : {}
|
|
5343
|
-
});
|
|
5344
|
-
for (const pending of pendingByAssignment.values()) {
|
|
5345
|
-
const refused = {
|
|
5346
|
-
...pending,
|
|
5347
|
-
outcome: "unpropagated",
|
|
5348
|
-
bytes: 0,
|
|
5349
|
-
reason: `no-live-worker-bound (spawn refused after the factory, or a keyed re-spawn deduplicated to a completed result; ${pending.bytes} composed bytes never crossed)`
|
|
5350
|
-
};
|
|
5351
|
-
ledger[ledger.indexOf(pending)] = refused;
|
|
5352
|
-
await appendJournal(refused, `graph:${refused.to}`);
|
|
5353
|
-
}
|
|
5354
|
-
pendingByAssignment.clear();
|
|
5355
|
-
await Promise.all(journalWrites);
|
|
5356
|
-
const exhaustedEdges = Object.freeze([...exhausted]);
|
|
5357
|
-
const frozenLedger = Object.freeze(ledger.map((row) => Object.freeze({ ...row })));
|
|
5358
|
-
const lifecycleEnded = result.kind === "no-winner" && (result.reason === "aborted" || result.reason === "budget-exhausted");
|
|
5359
|
-
if (result.kind !== "winner" && !lifecycleEnded && exhaustedDelegates.size > 0) throw new GraphEdgeCapError(Object.freeze([...exhaustedDelegates]), frozenLedger, result);
|
|
5360
|
-
return {
|
|
5361
|
-
result,
|
|
5362
|
-
ledger: frozenLedger,
|
|
5363
|
-
exhaustedEdges,
|
|
5364
|
-
runId
|
|
5365
|
-
};
|
|
5366
|
-
};
|
|
5367
|
-
return start();
|
|
5368
|
-
}
|
|
5369
|
-
/** The root task: the graph's own framing. The deliverable (mandatory) is the termination; the
|
|
5370
|
-
* task names what the topology exists to produce. */
|
|
5371
|
-
function graphTask(graph, root) {
|
|
5372
|
-
return graph.deliverable.describe ?? `Deliver the graph's deliverable by driving your worker nodes (root: '${root.id}').`;
|
|
5373
|
-
}
|
|
5374
|
-
//#endregion
|
|
5375
4492
|
//#region src/runtime/supervise/patch-checks.ts
|
|
5376
4493
|
const DEFAULT_MAX_DIFF_LINES = 400;
|
|
5377
4494
|
/**
|
|
@@ -5810,7 +4927,6 @@ function worktreeFanout(options) {
|
|
|
5810
4927
|
return gateOnDeliverable(createWorktreeCliExecutor({
|
|
5811
4928
|
repoRoot: options.repoRoot,
|
|
5812
4929
|
profile: item.profile,
|
|
5813
|
-
harness: item.harness,
|
|
5814
4930
|
taskPrompt: options.taskPrompt,
|
|
5815
4931
|
executionAttemptId: ctx.node.attemptId,
|
|
5816
4932
|
...item.budgetExempt !== void 0 ? { budgetExempt: item.budgetExempt } : {},
|
|
@@ -5983,7 +5099,7 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
|
|
|
5983
5099
|
const guidance = typeof brief === "string" ? brief.trim() : brief ? JSON.stringify(brief) : "";
|
|
5984
5100
|
const attemptTask = guidance ? {
|
|
5985
5101
|
...task,
|
|
5986
|
-
|
|
5102
|
+
userPrompt: `${task.userPrompt}\n\n— Supervisor guidance for THIS attempt (incorporate it; do not just repeat a prior approach) —\n${guidance}`
|
|
5987
5103
|
} : task;
|
|
5988
5104
|
const r = await runAgentic({
|
|
5989
5105
|
surface: traced.surface,
|
|
@@ -5992,8 +5108,8 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
|
|
|
5992
5108
|
budget: worker.budget ?? 1,
|
|
5993
5109
|
routerBaseUrl: worker.routerBaseUrl,
|
|
5994
5110
|
routerKey: worker.routerKey,
|
|
5995
|
-
|
|
5996
|
-
...worker.
|
|
5111
|
+
workerProfile: worker.profile,
|
|
5112
|
+
...worker.analystProfile ? { analystProfile: worker.analystProfile } : {},
|
|
5997
5113
|
...worker.innerTurns !== void 0 ? { innerTurns: worker.innerTurns } : {}
|
|
5998
5114
|
});
|
|
5999
5115
|
const out = {
|
|
@@ -6006,7 +5122,9 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
|
|
|
6006
5122
|
const spent = {
|
|
6007
5123
|
iterations: r.completions,
|
|
6008
5124
|
tokens: r.tokens,
|
|
5125
|
+
...r.tokensKnown ? {} : { tokensKnown: false },
|
|
6009
5126
|
usd: r.usd,
|
|
5127
|
+
...r.usdKnown ? {} : { usdKnown: false },
|
|
6010
5128
|
ms: r.ms
|
|
6011
5129
|
};
|
|
6012
5130
|
artifact = {
|
|
@@ -6036,13 +5154,13 @@ async function superviseSurface(profile, task, opts) {
|
|
|
6036
5154
|
const innerTurns = opts.worker.innerTurns ?? 6;
|
|
6037
5155
|
const router = opts.router ?? {
|
|
6038
5156
|
routerBaseUrl: opts.worker.routerBaseUrl,
|
|
6039
|
-
routerKey: opts.worker.routerKey
|
|
6040
|
-
model: opts.worker.model
|
|
5157
|
+
routerKey: opts.worker.routerKey
|
|
6041
5158
|
};
|
|
6042
5159
|
const budget = opts.budget ?? {
|
|
6043
5160
|
maxIterations: (innerTurns + 2) * 5 + 16,
|
|
6044
|
-
maxTokens:
|
|
5161
|
+
maxTokens: 1e9
|
|
6045
5162
|
};
|
|
5163
|
+
const workerMaxTokens = profileMaxTokens(opts.worker.profile) ?? Math.max(1, Math.floor(budget.maxTokens / 8));
|
|
6046
5164
|
const makeWorkerAgent = (rawProfile) => {
|
|
6047
5165
|
const p = rawProfile ?? {};
|
|
6048
5166
|
return {
|
|
@@ -6067,7 +5185,7 @@ async function superviseSurface(profile, task, opts) {
|
|
|
6067
5185
|
maxLiveWorkers: opts.maxLiveWorkers ?? 1,
|
|
6068
5186
|
perWorker: {
|
|
6069
5187
|
maxIterations: innerTurns + 2,
|
|
6070
|
-
maxTokens:
|
|
5188
|
+
maxTokens: workerMaxTokens
|
|
6071
5189
|
},
|
|
6072
5190
|
router,
|
|
6073
5191
|
...analysts ? {
|
|
@@ -6087,6 +5205,12 @@ async function superviseSurface(profile, task, opts) {
|
|
|
6087
5205
|
completions: sp.iterations
|
|
6088
5206
|
};
|
|
6089
5207
|
}
|
|
5208
|
+
function profileMaxTokens(profile) {
|
|
5209
|
+
const value = profile.model?.metadata?.maxTokens;
|
|
5210
|
+
if (value === void 0) return void 0;
|
|
5211
|
+
if (!Number.isSafeInteger(value) || value < 1) throw new Error("superviseSurface: AgentProfile.model.metadata.maxTokens must be a positive safe integer");
|
|
5212
|
+
return value;
|
|
5213
|
+
}
|
|
6090
5214
|
//#endregion
|
|
6091
5215
|
//#region src/runtime/verifier-environment.ts
|
|
6092
5216
|
const submitTool = {
|
|
@@ -6501,6 +5625,6 @@ function tail(s) {
|
|
|
6501
5625
|
return s.slice(-400);
|
|
6502
5626
|
}
|
|
6503
5627
|
//#endregion
|
|
6504
|
-
export {
|
|
5628
|
+
export { pipeline as $, assertStrategyContract as A, createMcpEnvironment as At, trajectoryReport as B, chatTransportExecutor as C, renderLeaderboardSvg as Ct, pickChampion as D, McpSpawnFault as Dt, discriminatingMeans as E, defaultAuditorInstruction as Et, openSandboxRun as F, resolveSecretEnv as Ft, registerShape as G, runPersonified as H, printBenchmarkReport as I, secretEnvOfMcpServer as It, renderCorpusToInstructions as J, FileCorpus as K, runBenchmark as L, strategyAuthorContract as M, envKeyProvider as Mt, strategyAuthorSystemPrompt as N, mcpSecretEnvMetadataKey as Nt, runStrategyEvolution as O, connectStdioMcp as Ot, SandboxRunAbortError as P, resolveMcpServerLaunch as Pt, panel as Q, promotionGate as R, runCoderChecks as S, renderLeaderboardMarkdown as St, createChatSessionStore as T, auditIntent as Tt, builtinShapes as U, definePersona as V, createShapeRegistry as W, flatWidenGate as X, fanout as Y, loopUntil as Z, settledWorkerOut as _, sentinelCompletion as _t, localShell as a, createScopeAnalyst as at, analyzeTrace as b, pairwiseSignificance as bt, createVerifierEnvironment as c, harvestCorpus as ct, worktreeFanout as d, localSandboxClient as dt, selectValidWinner as et, EVIDENCE_MAX_CHARS as f, inlineSandboxClient as ft, composeWorkerEvidence as g, deterministicCompletion as gt, closingWorkerNote as h, completionAuthorizes as ht, jjWorkspace as i, buildSteerContext as it, authorStrategy as j, sanitizeMcpToolSchema as jt, selectChampion as k, materializeLocalMcp as kt, failuresAnalyst as l, defineLeaderboard as lt, VERIFY_TAIL_CHARS as m, loopDispatch as mt, makeFinding$1 as n, widen as nt, runInWorkspace as o, registryScopeAnalyst as ot, NOTE_MAX_CHARS as p, loopCampaignDispatch as pt, InMemoryCorpus as q, gitWorkspace as r, assertTraceDerivedFindings as rt, createWaterfallCollector as s, inProcessSandboxClient as st, computeFindingId$1 as t, verify as tt, superviseSurface as u, resolveSandboxClient as ut, copyUntrackedIntoClone as v, stopSentinel as vt, chatWorkerSeam as w, renderPairwiseMarkdown as wt, patchDelivered as x, renderLeaderboardHtml as xt, withUntrackedArtifacts as y, leaderboard as yt, equalKOnCost as z };
|
|
6505
5629
|
|
|
6506
|
-
//# sourceMappingURL=runtime-
|
|
5630
|
+
//# sourceMappingURL=runtime-hiAABiTk.js.map
|