@tangle-network/agent-runtime 0.128.0 → 0.131.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (136) hide show
  1. package/README.md +70 -20
  2. package/dist/{activation-DhWJ3p8N.js → activation-BCzMOTaV.js} +3 -3
  3. package/dist/{activation-DhWJ3p8N.js.map → activation-BCzMOTaV.js.map} +1 -1
  4. package/dist/agent.d.ts +2 -3
  5. package/dist/agent.js +4 -5
  6. package/dist/agent.js.map +1 -1
  7. package/dist/{analyst-loop-DvSciOfB.js → analyst-loop-BE8cDs5Q.js} +2 -2
  8. package/dist/{analyst-loop-DvSciOfB.js.map → analyst-loop-BE8cDs5Q.js.map} +1 -1
  9. package/dist/analyst-loop.d.ts +1 -1
  10. package/dist/analyst-loop.js +1 -1
  11. package/dist/authoring-CvHwo1oW.js +163 -0
  12. package/dist/authoring-CvHwo1oW.js.map +1 -0
  13. package/dist/candidate-execution/index.js +4 -4
  14. package/dist/{candidate-execution-BFpq-Xi6.js → candidate-execution-BNxKr-Bu.js} +5 -5
  15. package/dist/{candidate-execution-BFpq-Xi6.js.map → candidate-execution-BNxKr-Bu.js.map} +1 -1
  16. package/dist/{conversation-BpLQZGPH.js → conversation-DNtxaJ1Z.js} +113 -31
  17. package/dist/conversation-DNtxaJ1Z.js.map +1 -0
  18. package/dist/conversation.d.ts +2 -2
  19. package/dist/conversation.js +2 -2
  20. package/dist/environment-provider-CxvSd1W6.d.ts +86 -0
  21. package/dist/{environment-provider-DqFS6FSZ.js → environment-provider-Dyg8DtLK.js} +8 -4
  22. package/dist/environment-provider-Dyg8DtLK.js.map +1 -0
  23. package/dist/environment-provider.d.ts +1 -1
  24. package/dist/environment-provider.js +1 -1
  25. package/dist/graph-BJTxGOFB.js +471 -0
  26. package/dist/graph-BJTxGOFB.js.map +1 -0
  27. package/dist/{improvement-cycle-IJgCbKWQ.js → improvement-cycle-Csp38cWg.js} +133 -224
  28. package/dist/improvement-cycle-Csp38cWg.js.map +1 -0
  29. package/dist/{index-BhZhQw77.d.ts → index-CoO7atyo.d.ts} +556 -1278
  30. package/dist/{index-BhuzfG2r.d.ts → index-DwGtu9nc.d.ts} +7 -9
  31. package/dist/{index-Efjb3nrQ.d.ts → index-qYHpsmG2.d.ts} +36 -18
  32. package/dist/index.d.ts +353 -11
  33. package/dist/index.js +111 -354
  34. package/dist/index.js.map +1 -1
  35. package/dist/intelligence.d.ts +6 -6
  36. package/dist/intelligence.js +9 -8
  37. package/dist/intelligence.js.map +1 -1
  38. package/dist/kernel.d.ts +7 -5
  39. package/dist/kernel.js +13 -9
  40. package/dist/{knowledge-DF63xPr4.js → knowledge-ce0_uKCl.js} +19 -17
  41. package/dist/knowledge-ce0_uKCl.js.map +1 -0
  42. package/dist/knowledge.d.ts +1 -1
  43. package/dist/knowledge.js +1 -1
  44. package/dist/{loop-runner-bin-Ckp_9tmD.d.ts → loop-runner-bin-BwgZ8m_7.d.ts} +6 -3
  45. package/dist/{loop-runner-bin-CWqOpCEw.js → loop-runner-bin-DSbuDDqM.js} +5 -27
  46. package/dist/loop-runner-bin-DSbuDDqM.js.map +1 -0
  47. package/dist/loop-runner-bin.d.ts +1 -1
  48. package/dist/loop-runner-bin.js +1 -1
  49. package/dist/materialization-COJ1UYQ-.js +272 -0
  50. package/dist/materialization-COJ1UYQ-.js.map +1 -0
  51. package/dist/mcp/bin.js +39 -47
  52. package/dist/mcp/bin.js.map +1 -1
  53. package/dist/mcp/index.d.ts +24 -26
  54. package/dist/mcp/index.js +66 -83
  55. package/dist/mcp/index.js.map +1 -1
  56. package/dist/mcp/memory-bin.js +1 -1
  57. package/dist/{memory-server-DL6cE2Ag.js → memory-server-5HEJH672.js} +2 -2
  58. package/dist/{memory-server-DL6cE2Ag.js.map → memory-server-5HEJH672.js.map} +1 -1
  59. package/dist/model-policy-CqziaqS1.js +232 -0
  60. package/dist/model-policy-CqziaqS1.js.map +1 -0
  61. package/dist/{openai-tools-B68JaOCx.d.ts → openai-tools-D3oMyVGl.d.ts} +2 -2
  62. package/dist/{openai-tools-D3XfrrQ6.js → openai-tools-ru75mLjq.js} +2 -2
  63. package/dist/openai-tools-ru75mLjq.js.map +1 -0
  64. package/dist/{prepare--8EvLqCr.js → prepare-DYWjVcPx.js} +169 -153
  65. package/dist/prepare-DYWjVcPx.js.map +1 -0
  66. package/dist/primeintellect/index.d.ts +7 -6
  67. package/dist/primeintellect/index.js +9 -11
  68. package/dist/primeintellect/index.js.map +1 -1
  69. package/dist/profiles.d.ts +21 -174
  70. package/dist/profiles.js +67 -276
  71. package/dist/profiles.js.map +1 -1
  72. package/dist/{protected-model-port-CXVfOUu_.js → protected-model-port-48ALLGxT.js} +2 -2
  73. package/dist/{protected-model-port-CXVfOUu_.js.map → protected-model-port-48ALLGxT.js.map} +1 -1
  74. package/dist/{redact-PQzmE1Jn.d.ts → redact-9_Gf-8m3.d.ts} +58 -69
  75. package/dist/{researcher-CoVqNhfI.js → researcher-Skz5-Uc8.js} +50 -26
  76. package/dist/researcher-Skz5-Uc8.js.map +1 -0
  77. package/dist/{run-layout-C2FGmZ3v.js → run-layout-CeJsEAom.js} +2 -2
  78. package/dist/{run-layout-C2FGmZ3v.js.map → run-layout-CeJsEAom.js.map} +1 -1
  79. package/dist/runtime-D-QfLbSd.d.ts +893 -0
  80. package/dist/{runtime-5uDVVfER.js → runtime-hiAABiTk.js} +315 -1191
  81. package/dist/runtime-hiAABiTk.js.map +1 -0
  82. package/dist/{sandbox-events-Yhd1GYWl.js → sandbox-events-CRDwc5WN.js} +42 -5
  83. package/dist/sandbox-events-CRDwc5WN.js.map +1 -0
  84. package/dist/snapshot-CXiiuHhL.js +21 -0
  85. package/dist/snapshot-CXiiuHhL.js.map +1 -0
  86. package/dist/{spawn-journal-DsZKDqeh.js → spawn-journal-saHQzqYi.js} +15 -21
  87. package/dist/spawn-journal-saHQzqYi.js.map +1 -0
  88. package/dist/stream-agent-turn-C852AgMT.d.ts +128 -0
  89. package/dist/stream-agent-turn-rYgaOLO0.js +910 -0
  90. package/dist/stream-agent-turn-rYgaOLO0.js.map +1 -0
  91. package/dist/{structural-rollout-zY0oqQzO.js → structural-rollout-3uxVGcg2.js} +459 -235
  92. package/dist/structural-rollout-3uxVGcg2.js.map +1 -0
  93. package/dist/{supervise-CsTKbH9R.js → supervise-iPN27pO0.js} +864 -4785
  94. package/dist/supervise-iPN27pO0.js.map +1 -0
  95. package/dist/{supervisor-DpjO0Gmy.js → supervisor-CV6Jh28D.js} +7439 -2897
  96. package/dist/supervisor-CV6Jh28D.js.map +1 -0
  97. package/dist/testing.d.ts +3 -1
  98. package/dist/testing.js +271 -221
  99. package/dist/testing.js.map +1 -1
  100. package/dist/{top-app-5unxqovu.js → top-app-G4M18b6i.js} +3 -3
  101. package/dist/{top-app-5unxqovu.js.map → top-app-G4M18b6i.js.map} +1 -1
  102. package/dist/tui/bin.js +1 -1
  103. package/dist/tui/index.js +1 -1
  104. package/dist/{environment-provider-CUFsyymu.d.ts → types-C6Q-J0Dt.d.ts} +51 -114
  105. package/dist/{types-DnNGJ5Gz.d.ts → types-ebIY0dMG.d.ts} +556 -30
  106. package/dist/{util-MVgdwuIS.js → util-Bw6srryQ.js} +3 -2
  107. package/dist/{util-MVgdwuIS.js.map → util-Bw6srryQ.js.map} +1 -1
  108. package/dist/{workspace-archive-BQxvkypI.js → workspace-archive-aOfJ47ms.js} +3 -3
  109. package/dist/{workspace-archive-BQxvkypI.js.map → workspace-archive-aOfJ47ms.js.map} +1 -1
  110. package/package.json +12 -15
  111. package/skills/agent-graphs/IMPROVE.md +3 -3
  112. package/skills/agent-graphs/SKILL.md +4 -5
  113. package/skills/agent-graphs/cases/review-pipeline.json +1 -2
  114. package/skills/agent-graphs/cases/unmeasured-harness.json +2 -4
  115. package/dist/backends-CiOCyRHb.js +0 -743
  116. package/dist/backends-CiOCyRHb.js.map +0 -1
  117. package/dist/conversation-BpLQZGPH.js.map +0 -1
  118. package/dist/environment-provider-DqFS6FSZ.js.map +0 -1
  119. package/dist/improvement-cycle-IJgCbKWQ.js.map +0 -1
  120. package/dist/index-DLM0W1h1.d.ts +0 -545
  121. package/dist/knowledge-DF63xPr4.js.map +0 -1
  122. package/dist/local-harness-BIajef4A.d.ts +0 -465
  123. package/dist/loop-runner-bin-CWqOpCEw.js.map +0 -1
  124. package/dist/model-resolution-Btd9iIKV.js +0 -98
  125. package/dist/model-resolution-Btd9iIKV.js.map +0 -1
  126. package/dist/openai-tools-D3XfrrQ6.js.map +0 -1
  127. package/dist/prepare--8EvLqCr.js.map +0 -1
  128. package/dist/researcher-CoVqNhfI.js.map +0 -1
  129. package/dist/runtime-5uDVVfER.js.map +0 -1
  130. package/dist/sandbox-events-Yhd1GYWl.js.map +0 -1
  131. package/dist/spawn-journal-DsZKDqeh.js.map +0 -1
  132. package/dist/structural-rollout-zY0oqQzO.js.map +0 -1
  133. package/dist/supervise-CsTKbH9R.js.map +0 -1
  134. package/dist/supervisor-DpjO0Gmy.js.map +0 -1
  135. package/dist/types-C9j4qg6l.d.ts +0 -500
  136. package/skills/agent-graphs/cases/floor-trap-pi.json +0 -11
@@ -1,20 +1,24 @@
1
- import { n as AnalystError, r as BackendTransportError, s as PlannerError, u as ValidationError } from "./errors-DEAvWQPy.js";
2
- import { i as normalizeBackendStreamEvent, o as newRuntimeSession, s as nowIso } from "./backends-CiOCyRHb.js";
3
- import { i as InMemorySpawnJournal, r as InMemoryResultBlobStore, v as isTraceAnalysisStore, x as contentAddress } from "./spawn-journal-DsZKDqeh.js";
4
- import { a as randomSuffix, c as stringifySafe, f as zeroTokenUsage, r as isAbortError, s as sleep, t as addTokenUsage } from "./util-MVgdwuIS.js";
1
+ import { n as AnalystError, s as PlannerError, u as ValidationError } from "./errors-DEAvWQPy.js";
2
+ import "./stream-agent-turn-rYgaOLO0.js";
3
+ import { i as InMemorySpawnJournal, r as InMemoryResultBlobStore, v as isTraceAnalysisStore, x as contentAddress } from "./spawn-journal-saHQzqYi.js";
4
+ import { a as randomSuffix, c as stringifySafe, f as zeroTokenUsage, r as isAbortError, s as sleep, t as addTokenUsage } from "./util-Bw6srryQ.js";
5
5
  import { i as redactProtectedValue, r as redactProtectedReason } from "./protected-redaction--F3v1oo8.js";
6
- import { $t as concreteProfileModel, Qt as concreteModelId, _t as routerChatWithUsage, bt as runBrainLoop, ct as attestRuntimeOwnedExecutor, dt as newExecutionAttemptId, en as isHarnessNativeModel, et as rollingDispatch, ht as routerBrain, l as withDriverExecutor, m as settledToIteration, n as createSupervisor } from "./supervisor-DpjO0Gmy.js";
7
- import { C as observe, O as strategyAuthorMethod, b as sample, v as refine, x as sampleThenRefine, y as runAgentic } from "./structural-rollout-zY0oqQzO.js";
8
- import { i as notifyRuntimeHookEvent, t as composeRuntimeHooks } from "./runtime-hooks-C7iJOWm3.js";
9
- import { a as notifySandboxEventObserver, i as mapSandboxToolEvent, r as mapSandboxEvent, t as createSandboxToolPartState } from "./sandbox-events-Yhd1GYWl.js";
10
- import { $t as kernelPromptRegistry, Ct as mergeAbortSignals, Ft as createSandboxLineage, It as probeSandboxCapabilities, Nt as defaultSelectWinner, Ot as createPushTraceSource, Pt as runAgentRounds, Qt as formatPromptHandle, St as createExecutorRegistry, Tt as createWorktreeCliExecutor, Ut as canonicalizeAuthoredProfile, n as supervise, r as workerFromBackend, rn as gateOnDeliverable, wt as taskToPrompt, xt as createExecutor } from "./supervise-CsTKbH9R.js";
11
- import { CODING_HARNESSES, InMemoryTraceStore, OUTPUT_VALUE, benjaminiHochberg, buildTrajectory, computeFindingId as computeFindingId$1, confidenceInterval, expandProfileAxes, harnessAxisOf, makeFinding as makeFinding$1, pairedBootstrap, paretoFrontier, scoreKnowledgeReadiness, wilcoxonSignedRank, wilson } from "@tangle-network/agent-eval";
6
+ import { a as notifySandboxEventObserver } from "./sandbox-events-CRDwc5WN.js";
7
+ import { c as profileProviderModel, i as concreteModelId, s as profileModelExecutionSettings, t as assertExecutableAgentProfile } from "./model-policy-CqziaqS1.js";
8
+ import { v as executableAgentProfileSnapshot, y as executableAgentSpecSnapshot } from "./materialization-COJ1UYQ-.js";
9
+ import { $ as createSandboxLineage, I as createWorktreeCliExecutor, N as createExecutor, P as createExecutorRegistry, Q as runAgentRounds, U as createPushTraceSource, Z as defaultSelectWinner, et as probeSandboxCapabilities, l as withDriverExecutor, m as settledToIteration, n as createSupervisor, nt as routerBrain, rt as runBrainLoop, w as rollingDispatch } from "./supervisor-CV6Jh28D.js";
10
+ import "./environment-provider-Dyg8DtLK.js";
11
+ import { i as notifyRuntimeHookEvent } from "./runtime-hooks-C7iJOWm3.js";
12
+ import { C as observe, O as strategyAuthorMethod, T as profileChatClient, b as sample, v as refine, x as sampleThenRefine, y as runAgentic } from "./structural-rollout-3uxVGcg2.js";
13
+ import { It as gateOnDeliverable, Lt as mapExecutorResult, n as supervise } from "./supervise-iPN27pO0.js";
14
+ import "./authoring-CvHwo1oW.js";
15
+ import "./graph-BJTxGOFB.js";
16
+ import { CODING_HARNESSES, InMemoryTraceStore, OUTPUT_VALUE, benjaminiHochberg, buildTrajectory, computeFindingId as computeFindingId$1, confidenceInterval, expandProfileAxes, makeFinding as makeFinding$1, pairedBootstrap, paretoFrontier, wilcoxonSignedRank, wilson } from "@tangle-network/agent-eval";
12
17
  import { heldoutSignificance, runProfileMatrix } from "@tangle-network/agent-eval/campaign";
13
- import { agentProfileSchema, canonicalCandidateDigest, validateAgentProfileSecurity } from "@tangle-network/agent-interface";
14
- import { randomUUID } from "node:crypto";
18
+ import { agentProfileSchema, canonicalAgentProfileDigest, validateAgentProfileSecurity } from "@tangle-network/agent-interface";
19
+ import { dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
15
20
  import { mkdir, readFile, writeFile } from "node:fs/promises";
16
21
  import { appendFileSync, chmodSync, constants, copyFileSync, existsSync, lstatSync, mkdirSync, mkdtempSync, readFileSync, readlinkSync, rmSync, symlinkSync, writeFileSync } from "node:fs";
17
- import { dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
18
22
  import { execFileSync, spawn, spawnSync } from "node:child_process";
19
23
  import { tmpdir } from "node:os";
20
24
  import { createInterface } from "node:readline";
@@ -531,13 +535,13 @@ const auditSchema = {
531
535
  };
532
536
  /** The route-rigor analyst: compare declared vs revealed vs user intent over a trajectory and return aligned / drifting / diverged with evidence and one recommended intervention. */
533
537
  async function auditIntent(input, opts) {
534
- const res = await opts.chat.chat({
535
- ...opts.model ? { model: opts.model } : {},
538
+ const res = await profileChatClient({
539
+ profile: opts.profile,
540
+ executor: opts.executor,
541
+ context: "intent auditor"
542
+ }).chat({
536
543
  jsonSchema: auditSchema,
537
544
  messages: [{
538
- role: "system",
539
- content: opts.auditorInstruction ?? "You audit whether an AI agent is on the RIGHT ROUTE — not whether it works hard, but whether its actions serve the stated intents. Infer the REVEALED intent from the action pattern (what the trajectory is actually optimizing). Compare against the declared task intent, the user intent when given, and the meta-intent when given. Flawless execution down the wrong route is DIVERGED. Busy-work that neither advances nor harms is DRIFTING. Judge only from the trajectory — be specific about which actions ground your verdict. Recommend abort only when continuing cannot serve the intent."
540
- }, {
541
545
  role: "user",
542
546
  content: `DECLARED INTENT (the task):\n${input.declaredIntent}\n\n` + (input.userIntent ? `USER INTENT (the principal's actual goal):\n${input.userIntent}\n\n` : "") + (input.metaIntent ? `META-INTENT (what the whole run is for):\n${input.metaIntent}\n\n` : "") + `TRAJECTORY (in order):\n${summarize(input.trace, opts.maxTraceLines ?? 80)}\n\nAudit the route: revealed intent, verdict, evidence, one recommendation.`
543
547
  }]
@@ -1036,7 +1040,10 @@ function loopCostReceipt(result, model) {
1036
1040
  model,
1037
1041
  inputTokens: result.tokenUsage.input,
1038
1042
  outputTokens: result.tokenUsage.output,
1039
- ...result.costUsd > 0 ? { actualCostUsd: result.costUsd } : {}
1043
+ ...result.tokenUsage.tokensKnown === false ? { usageUnknown: true } : {},
1044
+ ...result.costUsdKnown !== false ? { actualCostUsd: result.costUsd } : {},
1045
+ ...result.costUsdKnown === false ? { costUnknown: true } : {},
1046
+ ...result.estimatedCostUsd !== void 0 ? { estimatedCostUsd: result.estimatedCostUsd } : {}
1040
1047
  };
1041
1048
  }
1042
1049
  function modelFromLoopOptions(options) {
@@ -1071,6 +1078,20 @@ function loopDispatch(opts) {
1071
1078
  }
1072
1079
  //#endregion
1073
1080
  //#region src/runtime/inline-sandbox-client.ts
1081
+ /**
1082
+ * The ONE pseudo-box adapter: present any one-shot `Executor` (router / bridge /
1083
+ * BYO) as a `SandboxClient` so the round-synchronous `runAgentRounds` can drive it
1084
+ * without each call site re-faking a box. This is the single shell that
1085
+ * `bench/src/router-executor.ts`, generate-eval's old `bridgeSandboxClient`, and
1086
+ * the search-bench bridge transport were each re-implementing.
1087
+ *
1088
+ * It is deliberately for NON-box executors only — a real sandbox harness already
1089
+ * IS a `SandboxClient` (boxes, sessions, fs, fork are real there). Here each
1090
+ * `streamPrompt` runs the executor once and emits the terminal
1091
+ * `{type:'result', data:{finalText, tokenUsage, costUsd}}` event that
1092
+ * `answerOutput`/the kernel's cost ledger already parse — no sessions, no fs,
1093
+ * no fork (those degrade gracefully via the optional `SandboxClient` methods).
1094
+ */
1074
1095
  function isAsyncIterable$1(v) {
1075
1096
  return typeof v === "object" && v !== null && Symbol.asyncIterator in v;
1076
1097
  }
@@ -1088,7 +1109,8 @@ async function settle(exec, task, signal) {
1088
1109
  * instantiated fresh per `streamPrompt` (mirrors the per-spawn executor lifecycle):
1089
1110
  * run once on the prompt, emit the terminal result event, tear down.
1090
1111
  */
1091
- function inlineSandboxClient(factory) {
1112
+ function inlineSandboxClient(factory, defaults = {}) {
1113
+ const capturedDefaultProfile = defaults.profile === void 0 ? void 0 : agentProfileSchema.parse(structuredClone(defaults.profile));
1092
1114
  let seq = 0;
1093
1115
  return { async create(options) {
1094
1116
  const id = `inline-${seq++}`;
@@ -1101,8 +1123,11 @@ function inlineSandboxClient(factory) {
1101
1123
  const onAbort = () => controller.abort(callerSignal?.reason ?? /* @__PURE__ */ new Error("prompt aborted"));
1102
1124
  if (callerSignal) if (callerSignal.aborted) onAbort();
1103
1125
  else callerSignal.addEventListener("abort", onAbort, { once: true });
1126
+ const requestedProfile = (options?.backend && typeof options.backend === "object" ? options.backend.profile : void 0) ?? capturedDefaultProfile;
1127
+ const parsedProfile = agentProfileSchema.safeParse(requestedProfile);
1128
+ if (!parsedProfile.success) throw new Error("inlineSandboxClient: an exact AgentProfile is required; pass defaults.profile or create({ backend: { profile } })");
1104
1129
  const exec = factory({
1105
- profile: { name: id },
1130
+ profile: parsedProfile.data,
1106
1131
  harness: null
1107
1132
  }, {
1108
1133
  signal: controller.signal,
@@ -1114,23 +1139,43 @@ function inlineSandboxClient(factory) {
1114
1139
  const tokensIn = artifact.spent.tokens.input;
1115
1140
  const tokensOut = artifact.spent.tokens.output;
1116
1141
  const costUsd = artifact.spent.usd;
1117
- if (tokensIn || tokensOut || costUsd) yield {
1142
+ const estimatedCostUsd = out?.estimatedCostUsd;
1143
+ if (artifact.spent.iterations > 0 || artifact.spent.tokensKnown === false || artifact.spent.usdKnown === false || tokensIn > 0 || tokensOut > 0 || costUsd > 0 || estimatedCostUsd !== void 0) yield {
1118
1144
  type: "llm_call",
1119
1145
  data: {
1120
- tokensIn,
1121
- tokensOut,
1122
- costUsd
1146
+ ...artifact.spent.tokensKnown === false ? {} : {
1147
+ tokensIn,
1148
+ tokensOut
1149
+ },
1150
+ ...artifact.spent.usdKnown !== false ? { costUsd } : {},
1151
+ ...artifact.spent.tokensKnown === false ? { tokensKnown: false } : {},
1152
+ ...artifact.spent.usdKnown === false ? { costKnown: false } : {},
1153
+ ...artifact.spent.usdKnown !== false ? {
1154
+ costKnown: true,
1155
+ costProvenance: "provider-receipt"
1156
+ } : {},
1157
+ ...estimatedCostUsd !== void 0 ? { estimatedCostUsd } : {},
1158
+ ...out?.promptCache ? { promptCache: out.promptCache } : {}
1123
1159
  }
1124
1160
  };
1125
1161
  yield {
1126
1162
  type: "result",
1127
1163
  data: {
1128
1164
  finalText: out?.content ?? "",
1129
- tokenUsage: {
1165
+ ...artifact.spent.tokensKnown === false ? { tokensKnown: false } : { tokenUsage: {
1130
1166
  inputTokens: tokensIn,
1131
1167
  outputTokens: tokensOut
1168
+ } },
1169
+ ...artifact.spent.usdKnown === false ? {
1170
+ costKnown: false,
1171
+ ...costUsd > 0 ? { costUsd } : {}
1172
+ } : {
1173
+ costUsd,
1174
+ costKnown: true,
1175
+ costProvenance: "provider-receipt"
1132
1176
  },
1133
- costUsd
1177
+ ...estimatedCostUsd !== void 0 ? { estimatedCostUsd } : {},
1178
+ ...out?.promptCache ? { promptCache: out.promptCache } : {}
1134
1179
  }
1135
1180
  };
1136
1181
  } finally {
@@ -1159,36 +1204,52 @@ function inlineSandboxClient(factory) {
1159
1204
  * local MCP process applies only when its full canonical bytes match the fixed
1160
1205
  * constructor profile; a different generated profile is refused.
1161
1206
  *
1162
- * Event protocol matches `inlineSandboxClient`: one `llm_call` metering event
1163
- * + one terminal `result` event with finalText/tokenUsage/costUsd.
1207
+ * Event protocol matches `inlineSandboxClient`: known token usage is emitted as one `llm_call`;
1208
+ * Router catalog cost remains a separately-labelled estimate, never billed spend.
1164
1209
  */
1165
1210
  /** A same-host `SandboxClient` adapter with no process isolation. Local MCP is
1166
1211
  * refused unless the caller explicitly supplies a policy that allows it. */
1167
1212
  function localSandboxClient(opts) {
1168
- if (opts.profileSecurityPolicy?.allowLocalMcp && opts.profile === void 0) throw new ValidationError("localSandboxClient: allowLocalMcp requires a fixed author-controlled profile; dynamic profiles need a real sandbox");
1169
- const trustedProfileDigest = opts.profileSecurityPolicy?.allowLocalMcp && opts.profile !== void 0 ? canonicalCandidateDigest(opts.profile) : void 0;
1170
- const maxTurns = opts.maxTurns ?? 8;
1213
+ const defaultProfile = opts.profile === void 0 ? void 0 : executableAgentProfileSnapshot(opts.profile, "localSandboxClient default profile");
1214
+ const router = Object.freeze({ ...opts.router });
1215
+ if (opts.profileSecurityPolicy?.allowLocalMcp && defaultProfile === void 0) throw new ValidationError("localSandboxClient: allowLocalMcp requires a fixed author-controlled profile; dynamic profiles need a real sandbox");
1216
+ const trustedProfileDigest = opts.profileSecurityPolicy?.allowLocalMcp && defaultProfile !== void 0 ? canonicalAgentProfileDigest(defaultProfile) : void 0;
1171
1217
  let seq = 0;
1172
1218
  return { async create(options) {
1173
- const profile = (options?.backend)?.profile ?? opts.profile ?? {};
1174
- const policyApplies = opts.profileSecurityPolicy !== void 0 && (!opts.profileSecurityPolicy.allowLocalMcp || trustedProfileDigest !== void 0 && canonicalCandidateDigest(profile) === trustedProfileDigest);
1219
+ const profile = executableAgentProfileSnapshot((options?.backend)?.profile ?? defaultProfile, "localSandboxClient");
1220
+ const model = profileProviderModel(profile);
1221
+ const settings = profileModelExecutionSettings(profile, "localSandboxClient");
1222
+ const policyApplies = opts.profileSecurityPolicy !== void 0 && (!opts.profileSecurityPolicy.allowLocalMcp || trustedProfileDigest !== void 0 && canonicalAgentProfileDigest(profile) === trustedProfileDigest);
1175
1223
  const mcp = await materializeLocalMcp(profile, {
1176
1224
  ...opts.keys ? { keys: opts.keys } : {},
1177
1225
  ...policyApplies ? { profileSecurityPolicy: opts.profileSecurityPolicy } : {}
1178
1226
  });
1179
1227
  const brain = routerBrain({
1180
- routerBaseUrl: opts.router.baseUrl,
1181
- routerKey: opts.router.key,
1182
- model: opts.router.model
1183
- }, opts.temperature !== void 0 ? { temperature: opts.temperature } : {});
1228
+ routerBaseUrl: router.baseUrl,
1229
+ routerKey: router.key,
1230
+ model,
1231
+ ...settings.retry !== void 0 ? { retry: settings.retry } : {},
1232
+ ...settings.maxTokens !== void 0 ? { maxTokens: settings.maxTokens } : {},
1233
+ ...settings.stream !== void 0 ? { stream: settings.stream } : {}
1234
+ }, {
1235
+ ...settings.temperature !== void 0 ? { temperature: settings.temperature } : {},
1236
+ ...settings.seed !== void 0 ? { seed: settings.seed } : {},
1237
+ ...settings.toolChoice !== void 0 ? { toolChoice: settings.toolChoice } : {},
1238
+ ...settings.extraBody !== void 0 ? { extraBody: settings.extraBody } : {},
1239
+ ...profile.model?.reasoningEffort ? { reasoningEffort: profile.model.reasoningEffort } : {}
1240
+ });
1184
1241
  const system = [profile.prompt?.systemPrompt, ...profile.prompt?.instructions ?? []].filter((s) => typeof s === "string" && s.trim().length > 0).join("\n\n");
1185
1242
  return {
1186
1243
  id: `local-${seq++}`,
1187
1244
  async *streamPrompt(message, popts) {
1188
- let costUsd = 0;
1245
+ let estimatedCostUsd = 0;
1246
+ let sawEstimatedCost = false;
1189
1247
  const chat = async (messages, tools) => {
1190
1248
  const r = await brain(messages, tools);
1191
- if (r.costUsd) costUsd += r.costUsd;
1249
+ if (r.costProvenance === "catalog-estimate" && r.costUsd !== void 0) {
1250
+ estimatedCostUsd += r.costUsd;
1251
+ sawEstimatedCost = true;
1252
+ }
1192
1253
  return r;
1193
1254
  };
1194
1255
  const r = await runBrainLoop({
@@ -1202,26 +1263,31 @@ function localSandboxClient(opts) {
1202
1263
  role: "user",
1203
1264
  content: message
1204
1265
  }],
1205
- maxTurns,
1266
+ maxTurns: settings.maxTurns ?? 0,
1206
1267
  hooks: { stopBefore: () => popts?.signal?.aborted === true }
1207
1268
  });
1208
- if (r.usage.input || r.usage.output || costUsd) yield {
1269
+ if (r.turns > 0) yield {
1209
1270
  type: "llm_call",
1210
1271
  data: {
1211
- tokensIn: r.usage.input,
1212
- tokensOut: r.usage.output,
1213
- costUsd
1272
+ model,
1273
+ ...r.tokensKnown === false ? { tokensKnown: false } : {
1274
+ tokensIn: r.usage.input,
1275
+ tokensOut: r.usage.output
1276
+ },
1277
+ costKnown: false,
1278
+ ...sawEstimatedCost ? { estimatedCostUsd } : {}
1214
1279
  }
1215
1280
  };
1216
1281
  yield {
1217
1282
  type: "result",
1218
1283
  data: {
1219
1284
  finalText: r.final,
1220
- tokenUsage: {
1285
+ ...r.tokensKnown === false ? { tokensKnown: false } : { tokenUsage: {
1221
1286
  inputTokens: r.usage.input,
1222
1287
  outputTokens: r.usage.output
1223
- },
1224
- costUsd
1288
+ } },
1289
+ costKnown: false,
1290
+ ...sawEstimatedCost ? { estimatedCostUsd } : {}
1225
1291
  }
1226
1292
  };
1227
1293
  },
@@ -1269,28 +1335,26 @@ function resolveSandboxClient(opts) {
1269
1335
  return opts.sandboxClient;
1270
1336
  case "bridge": {
1271
1337
  const bridge = opts.bridge;
1272
- if (!bridge?.bearer || !bridge.model) throw new Error("resolveSandboxClient: backend 'bridge' requires bridge.bearer and bridge.model");
1338
+ if (!bridge?.bearer) throw new Error("resolveSandboxClient: backend 'bridge' requires bridge.bearer");
1273
1339
  return inlineSandboxClient(createExecutor({
1274
1340
  backend: "bridge",
1275
1341
  bridgeUrl: bridge.url ?? "http://127.0.0.1:3355",
1276
1342
  bridgeBearer: bridge.bearer,
1277
- model: bridge.model,
1278
1343
  timeoutMs: bridge.timeoutMs
1279
1344
  }));
1280
1345
  }
1281
1346
  case "router": {
1282
1347
  const router = opts.router;
1283
- if (!router?.baseUrl || !router.key || !router.model) throw new Error("resolveSandboxClient: backend 'router' requires router.baseUrl, router.key and router.model");
1348
+ if (!router?.baseUrl || !router.key) throw new Error("resolveSandboxClient: backend 'router' requires router.baseUrl and router.key");
1284
1349
  return inlineSandboxClient(createExecutor({
1285
1350
  backend: "router",
1286
1351
  routerBaseUrl: router.baseUrl,
1287
- routerKey: router.key,
1288
- model: router.model
1352
+ routerKey: router.key
1289
1353
  }));
1290
1354
  }
1291
1355
  case "local": {
1292
1356
  const local = opts.local;
1293
- if (!local?.router?.baseUrl || !local.router.key || !local.router.model) throw new Error("resolveSandboxClient: backend 'local' requires local.router.baseUrl, local.router.key and local.router.model");
1357
+ if (!local?.router?.baseUrl || !local.router.key) throw new Error("resolveSandboxClient: backend 'local' requires local.router.baseUrl and local.router.key");
1294
1358
  return localSandboxClient(local);
1295
1359
  }
1296
1360
  }
@@ -1314,7 +1378,7 @@ function resolveSandboxClient(opts) {
1314
1378
  *
1315
1379
  * - LEVEL 0 (declarative): `cases` / `prompt` / `score` / `axis`.
1316
1380
  * - LEVEL 1 (seams): `backends`, `flags`, `parseOutput`, `onCellEvents`,
1317
- * `resolveModel`, `setup`/`teardown`, `export`, `modelBackend`, `matrix`
1381
+ * `resolveModel`, `setup`/`teardown`, `export`, `matrix`
1318
1382
  * passthrough.
1319
1383
  * - LEVEL 2 (replacement): `dispatch` and `judges` swap out the whole
1320
1384
  * loop wiring or scoring; `runProfileMatrix` itself stays public as the
@@ -1373,10 +1437,6 @@ function splitList(v) {
1373
1437
  function withSnapshot(model, snapshot) {
1374
1438
  return model.includes("@") ? model : `${model}@${snapshot}`;
1375
1439
  }
1376
- /** The bare model id the backend actually serves (identity snapshot stripped). */
1377
- function bareModel(model) {
1378
- return model.split("@")[0] ?? model;
1379
- }
1380
1440
  function gitSha() {
1381
1441
  try {
1382
1442
  return execFileSync("git", ["rev-parse", "HEAD"], { encoding: "utf8" }).trim();
@@ -1458,13 +1518,11 @@ function defineLeaderboard(spec) {
1458
1518
  if (!rawModels || rawModels.length === 0) throw new Error(`defineLeaderboard(${spec.name}): no models — pass --models, set spec.axis.models, or give spec.baseProfile a model.default`);
1459
1519
  const models = rawModels.map((m) => withSnapshot(m, snapshot));
1460
1520
  const profiles = expandProfileAxes({
1461
- base: spec.baseProfile ?? {
1462
- name: spec.name,
1463
- model: { default: bareModel(models[0] ?? "") }
1464
- },
1521
+ base: spec.baseProfile,
1465
1522
  harnesses,
1466
1523
  models
1467
1524
  });
1525
+ for (const profile of profiles) assertExecutableAgentProfile(profile, `defineLeaderboard(${spec.name})`);
1468
1526
  const ctx = {
1469
1527
  name: spec.name,
1470
1528
  backend: backendName,
@@ -1489,7 +1547,6 @@ function defineLeaderboard(spec) {
1489
1547
  bridge: {
1490
1548
  url: process.env.CLI_BRIDGE_URL,
1491
1549
  bearer,
1492
- model: bareModel(models[0] ?? ""),
1493
1550
  timeoutMs: 9e5
1494
1551
  }
1495
1552
  });
@@ -1519,21 +1576,11 @@ function defineLeaderboard(spec) {
1519
1576
  return [...served][0] ?? cellProfile.model?.default;
1520
1577
  },
1521
1578
  toLoopOptions: (cellScenario, cellProfile) => {
1522
- const axis = harnessAxisOf(cellProfile);
1523
- const modelId = bareModel(axis?.model ?? models[0] ?? "");
1524
- const backendModel = {
1525
- ...spec.modelBackend,
1526
- ...!isHarnessNativeModel(modelId) || backendName === "cli-bridge" ? { model: modelId } : {}
1527
- };
1528
1579
  return {
1529
1580
  driver: naiveRetryDriver(shots),
1530
1581
  agentRun: {
1531
1582
  profile: cellProfile,
1532
- taskToPrompt: (s) => `${promptOf(s)}\n\n<!-- independent-attempt:${shotNonce++} -->`,
1533
- ...axis ? { sandboxOverrides: { backend: {
1534
- type: axis.harness,
1535
- ...Object.keys(backendModel).length > 0 ? { model: backendModel } : {}
1536
- } } } : {}
1583
+ taskToPrompt: (s) => `${promptOf(s)}\n\n<!-- independent-attempt:${shotNonce++} -->`
1537
1584
  },
1538
1585
  output: { parse: (events) => spec.parseOutput ? spec.parseOutput(events, cellScenario.case) : collectAgentResponseText(events) ?? "" },
1539
1586
  validator: { validate: async (output) => {
@@ -1654,11 +1701,10 @@ async function harvestCorpus(opts) {
1654
1701
  if (opts.signal?.aborted) return;
1655
1702
  try {
1656
1703
  const obs = await observe(input, {
1657
- chat: opts.chat,
1658
- ...opts.model ? { model: opts.model } : {},
1704
+ profile: opts.profile,
1705
+ executor: opts.executor,
1659
1706
  corpus: opts.corpus,
1660
1707
  tags: opts.tags ?? [],
1661
- ...opts.analystInstruction ? { analystInstruction: opts.analystInstruction } : {},
1662
1708
  ...opts.signal ? { signal: opts.signal } : {}
1663
1709
  });
1664
1710
  report.runsObserved += 1;
@@ -1942,7 +1988,7 @@ function observedBestScore(settledSoFar) {
1942
1988
  * CONCRETE blocker (never an eager over-fan, never a silent drop), and a `blocked` outcome always
1943
1989
  * names at least one blocker (a shape that cannot finish MUST say why — `blocked([])` throws).
1944
1990
  *
1945
- * @experimental
1991
+ * @stable
1946
1992
  */
1947
1993
  /**
1948
1994
  * The single content-free valid-only winner selector. Among the gated-VALID children only
@@ -1977,6 +2023,8 @@ function selectValidWinner(opts) {
1977
2023
  * pool would not admit, or a stage whose `collect` chose to block) short-circuits — its blockers
1978
2024
  * ARE the pipeline's blockers, never coerced past a failed stage. The terminal stage's `done`
1979
2025
  * deliverable is the pipeline's deliverable.
2026
+ *
2027
+ * @stable
1980
2028
  */
1981
2029
  function pipeline(stages) {
1982
2030
  if (stages.length === 0) throw new ValidationError("pipeline: at least one stage is required");
@@ -2013,6 +2061,8 @@ function pipeline(stages) {
2013
2061
  * `opts.width` swaps the single round for `rollingDispatch`: at most `width` items live at once,
2014
2062
  * refilled the instant one settles. Selection, blockers, and the conserved pool are unchanged —
2015
2063
  * the refill behavior lives in the existing combinator rather than in a rival primitive.
2064
+ *
2065
+ * @stable
2016
2066
  */
2017
2067
  function fanout(items, opts) {
2018
2068
  if (opts.synthesize && opts.selectWinner) throw new ValidationError("fanout: pass at most one of `synthesize` or `selectWinner`");
@@ -2096,6 +2146,8 @@ function fanout(items, opts) {
2096
2146
  * `until` on the resulting trace-derived findings (the analyst spawns into THIS scope, so its
2097
2147
  * compute is conserved-pooled — equal-k holds by construction). Absent an analyst the findings
2098
2148
  * argument is the empty array — never a fabricated finding (fail-loud honesty over a silent default).
2149
+ *
2150
+ * @stable
2099
2151
  */
2100
2152
  function loopUntil(seed, spec) {
2101
2153
  return (ctx) => ({
@@ -2144,6 +2196,8 @@ function loopUntil(seed, spec) {
2144
2196
  * reaches another judge's task; the merge never spawns or re-ranks). A `down` judge carries no
2145
2197
  * verdict and is excluded from the merge denominator. A panel that admitted no judge is a
2146
2198
  * concrete blocker before `merge` is consulted.
2199
+ *
2200
+ * @stable
2147
2201
  */
2148
2202
  function panel(spec) {
2149
2203
  if (spec.judges.length === 0) throw new ValidationError("panel: at least one judge is required");
@@ -2193,6 +2247,8 @@ function panel(spec) {
2193
2247
  * it; only a `valid` verifier verdict ships. Any other outcome (implement down, verifier down,
2194
2248
  * verifier verdict absent or not `valid`) is a concrete blocker carrying the failure verbatim —
2195
2249
  * never a coerced "done". The implement child does not grade itself.
2250
+ *
2251
+ * @stable
2196
2252
  */
2197
2253
  function verify(spec) {
2198
2254
  return (ctx) => ({
@@ -2237,6 +2293,8 @@ function verify(spec) {
2237
2293
  * the widen loop sees it. The shipped default (`flatWidenGate`) never widens, so no widen child is
2238
2294
  * ever live when the analyst runs and the wire is exact; a non-flat gate must drive the analyst on
2239
2295
  * a scope whose siblings are quiesced, or read findings without the shared-cursor drain.
2296
+ *
2297
+ * @stable
2240
2298
  */
2241
2299
  function widen(spec) {
2242
2300
  return (ctx) => ({
@@ -2686,19 +2744,22 @@ function registerShape(name, factory) {
2686
2744
  * receive a ctx with the persona seams merged in — so a persona never has to pre-close its
2687
2745
  * factories by hand. A persona may instead supply a fully-built `registry` and skip the wrap.
2688
2746
  *
2689
- * @experimental
2747
+ * @stable
2690
2748
  */
2691
2749
  /**
2692
2750
  * Build a frozen `Persona`. Fails loud on the executors-supplied invariant: a persona with
2693
2751
  * neither a pre-built registry nor a seam bag cannot resolve its built-in runtimes, so it is
2694
2752
  * unrunnable — refuse it at definition time, not at the first spawn. Pure; no I/O.
2753
+ *
2754
+ * @stable
2695
2755
  */
2696
2756
  function definePersona(input) {
2697
2757
  if (!input.executors.registry && !input.executors.seams) throw new ValidationError(`definePersona("${input.name}"): executors must supply a registry or a seams bag (built-in runtimes read their seams off ExecutorContext; neither was provided)`);
2698
2758
  if (!input.root || typeof input.root !== "object" || !("harness" in input.root)) throw new ValidationError(`definePersona("${input.name}"): root must be an AgentSpec`);
2759
+ const root = executableAgentSpecSnapshot(input.root, `definePersona("${input.name}")`);
2699
2760
  return Object.freeze({
2700
2761
  name: input.name,
2701
- root: input.root,
2762
+ root,
2702
2763
  directive: input.directive,
2703
2764
  context: input.context,
2704
2765
  executors: input.executors,
@@ -2741,6 +2802,8 @@ function createShapeContext(persona, budget, analyst) {
2741
2802
  * `ShapeContext`, and runs the resulting root `Agent` to a typed `SupervisedResult<Outcome>`.
2742
2803
  * Fail loud on an unknown shape name or an unresolvable persona registry — never a silent
2743
2804
  * default-shape fallback.
2805
+ *
2806
+ * @stable
2744
2807
  */
2745
2808
  async function runPersonified(options) {
2746
2809
  const { persona } = options;
@@ -3291,22 +3354,34 @@ async function pool(items, limit, fn) {
3291
3354
  async function preflightModels(cfg) {
3292
3355
  if (cfg.modelPreflight === false) return;
3293
3356
  if (cfg.worker.complete && !cfg.modelPreflight) return;
3294
- const models = [.../* @__PURE__ */ new Set([cfg.worker.model, cfg.worker.analystModel ?? cfg.worker.model])];
3357
+ const profiles = [cfg.worker.workerProfile, cfg.worker.analystProfile ?? cfg.worker.workerProfile];
3358
+ const profilesByModel = /* @__PURE__ */ new Map();
3359
+ for (const [index, profile] of profiles.entries()) {
3360
+ const model = concreteModelId(profile.model?.default);
3361
+ if (!model) throw new Error(`Benchmark ${index === 0 ? "worker" : "analyst"} AgentProfile.model.default must name an exact model`);
3362
+ if (!profilesByModel.has(model)) profilesByModel.set(model, profile);
3363
+ }
3364
+ const models = [...profilesByModel.keys()];
3295
3365
  const timeoutMs = cfg.modelPreflightTimeoutMs ?? 3e4;
3296
3366
  if (!Number.isFinite(timeoutMs) || timeoutMs <= 0) throw new Error("modelPreflightTimeoutMs must be a positive finite number");
3297
3367
  const check = cfg.modelPreflight ?? (async (model, worker, signal) => {
3298
- await routerChatWithUsage({
3299
- routerBaseUrl: worker.routerBaseUrl,
3300
- routerKey: worker.routerKey,
3301
- model
3302
- }, [{
3303
- role: "user",
3304
- content: "Reply OK."
3305
- }], {
3306
- maxTokens: 1,
3307
- reasoningEffort: "none",
3308
- signal
3309
- });
3368
+ const profile = profilesByModel.get(model);
3369
+ if (!profile) throw new Error(`Benchmark preflight has no AgentProfile for model ${model}`);
3370
+ await profileChatClient({
3371
+ profile,
3372
+ context: `runBenchmark model preflight (${model})`,
3373
+ executor: {
3374
+ backend: "router",
3375
+ routerBaseUrl: worker.routerBaseUrl,
3376
+ routerKey: worker.routerKey
3377
+ }
3378
+ }).chat({
3379
+ model,
3380
+ messages: [{
3381
+ role: "user",
3382
+ content: "Reply OK."
3383
+ }]
3384
+ }, { signal });
3310
3385
  });
3311
3386
  const failures = (await Promise.allSettled(models.map(async (model) => {
3312
3387
  const controller = new AbortController();
@@ -3362,8 +3437,10 @@ async function runBenchmark(cfg) {
3362
3437
  resolved: r.resolved,
3363
3438
  progression: r.progression,
3364
3439
  usd: r.usd,
3440
+ usdKnown: r.usdKnown,
3365
3441
  ms: r.ms,
3366
- tokens: r.tokens
3442
+ tokens: r.tokens,
3443
+ tokensKnown: r.tokensKnown
3367
3444
  };
3368
3445
  } catch (e) {
3369
3446
  errors[s.name] = e instanceof Error ? e.message.slice(0, 300) : String(e);
@@ -3372,11 +3449,13 @@ async function runBenchmark(cfg) {
3372
3449
  resolved: false,
3373
3450
  progression: [],
3374
3451
  usd: 0,
3452
+ usdKnown: true,
3375
3453
  ms: 0,
3376
3454
  tokens: {
3377
3455
  input: 0,
3378
3456
  output: 0
3379
- }
3457
+ },
3458
+ tokensKnown: true
3380
3459
  };
3381
3460
  }
3382
3461
  row = {
@@ -3403,6 +3482,7 @@ async function runBenchmark(cfg) {
3403
3482
  score: mean(cells.map((c) => c.score)),
3404
3483
  resolved: mean(cells.map((c) => c.resolved ? 1 : 0)),
3405
3484
  usd: mean(cells.map((c) => c.usd)),
3485
+ usdKnownRate: mean(cells.map((c) => c.usdKnown ? 1 : 0)),
3406
3486
  ms: mean(cells.map((c) => c.ms))
3407
3487
  };
3408
3488
  }
@@ -3794,6 +3874,8 @@ export default defineStrategy('your-strategy-name', async ({ surface, task, budg
3794
3874
  // your composition (listTools comes from the destructured context — it is NOT a global)
3795
3875
  })
3796
3876
  `;
3877
+ /** Standing behavior callers put in the strategy-author AgentProfile. */
3878
+ const strategyAuthorSystemPrompt = "You are a senior researcher authoring optimization strategies for agent loops: you read per-task losses like experimental data, form a mechanism-level hypothesis, and author the one composition that tests it. Output exactly one fenced ```ts code block and nothing else.";
3797
3879
  /** Static CONTRACT lint over an authored strategy module — the module-boundary
3798
3880
  * enforcement of the harness's two measurement invariants:
3799
3881
  * - author blindness: the only import allowed is the kernel surface. A body that could
@@ -3820,21 +3902,17 @@ function assertStrategyContract(code) {
3820
3902
  }
3821
3903
  /** One authoring attempt: chat with the given model, extract the fenced module. Throws
3822
3904
  * when the reply carries no code block. */
3823
- async function requestAuthoredCode(opts, model) {
3824
- const res = await opts.chat.chat({
3825
- ...model ? { model } : {},
3826
- ...opts.temperature !== void 0 ? { temperature: opts.temperature } : {},
3827
- ...opts.maxTokens !== void 0 ? { maxTokens: opts.maxTokens } : {},
3828
- messages: [{
3829
- role: "system",
3830
- content: "You are a senior researcher authoring optimization strategies for agent loops: you read per-task losses like experimental data, form a mechanism-level hypothesis, and author the one composition that tests it. Output exactly one fenced ```ts code block and nothing else."
3831
- }, {
3832
- role: "user",
3833
- content: `${opts.contract ?? strategyAuthorContract}\n\nBASELINE RESULTS on the "${opts.environmentName}" environment (budget=${opts.budget}) — the per-task losses are your gradient:\n${opts.lossesJson}\n\nAuthor ONE new strategy that you expect to beat the baselines on THIS environment at the same budget.\n${strategyAuthorMethod}\n\nOutput only the module code block.`
3834
- }]
3835
- }, { ...opts.signal ? { signal: opts.signal } : {} });
3905
+ async function requestAuthoredCode(opts, profile) {
3906
+ const res = await profileChatClient({
3907
+ profile,
3908
+ executor: opts.executor,
3909
+ context: "strategy author"
3910
+ }).chat({ messages: [{
3911
+ role: "user",
3912
+ content: `${opts.contract ?? strategyAuthorContract}\n\nBASELINE RESULTS on the "${opts.environmentName}" environment (budget=${opts.budget}) the per-task losses are your gradient:\n${opts.lossesJson}\n\nAuthor ONE new strategy that you expect to beat the baselines on THIS environment at the same budget.\n${strategyAuthorMethod}\n\nOutput only the module code block.`
3913
+ }] }, { ...opts.signal ? { signal: opts.signal } : {} });
3836
3914
  const match = res.content.match(/```(?:ts|typescript)?\s*\n([\s\S]*?)```/);
3837
- if (!match?.[1]) throw new Error(`authorStrategy: no code block in the author's reply (model=${model ?? "default"}): ${res.content.slice(0, 300)}`);
3915
+ if (!match?.[1]) throw new Error(`authorStrategy: no code block in the author's reply: ${res.content.slice(0, 300)}`);
3838
3916
  return match[1];
3839
3917
  }
3840
3918
  /** Author + load a strategy from losses. Throws when the author emits no loadable module;
@@ -3842,10 +3920,10 @@ async function requestAuthoredCode(opts, model) {
3842
3920
  async function authorStrategy(opts) {
3843
3921
  let code;
3844
3922
  try {
3845
- code = await requestAuthoredCode(opts, opts.model);
3923
+ code = await requestAuthoredCode(opts, opts.profile);
3846
3924
  } catch (primaryError) {
3847
- if (!opts.fallbackModel) throw primaryError;
3848
- code = await requestAuthoredCode(opts, opts.fallbackModel);
3925
+ if (!opts.fallbackProfile) throw primaryError;
3926
+ code = await requestAuthoredCode(opts, opts.fallbackProfile);
3849
3927
  }
3850
3928
  assertStrategyContract(code);
3851
3929
  mkdirSync(opts.outDir, { recursive: true });
@@ -3883,6 +3961,8 @@ async function authorStrategy(opts) {
3883
3961
  * Lineage fields (`parent`, `generation`) are recorded on every archive node so a
3884
3962
  * descendant-productivity parent-selection policy can be added without changing the
3885
3963
  * report schema; the v1 search authors from the latest tournament's losses.
3964
+ *
3965
+ * @experimental
3886
3966
  */
3887
3967
  /** Strategy means recomputed over the DISCRIMINATING tasks only — tasks where the field
3888
3968
  * strategies did not all score identically. Zero-spread tasks (everyone 1.0, everyone
@@ -4084,11 +4164,9 @@ async function runStrategyEvolution(cfg) {
4084
4164
  const contract = `${strategyAuthorContract}${cfg.objective === "cost" ? `\n\nYOUR OBJECTIVE: match or exceed the incumbent's SCORE while spending LESS (the losses include usd per task). Promotion requires proven score non-inferiority PLUS significant cost savings — a strategy that ties the score at half the cost WINS; a cheaper strategy that loses score by more than ${((cfg.scoreTolerance ?? .05) * 100).toFixed(0)}pp LOSES.` : ""}\n\nEXAMPLE TOOLS FROM ONE TASK (tool sets VARY per task on this domain — a strategy MUST select tool names from await listTools(handle) at runtime; hardcoding these example names will zero your score on most tasks):\n${toolCatalog}\n\nSTRATEGIES ALREADY IN THE TOURNAMENT (author something MEANINGFULLY different — a new composition, not a rename):\n${fieldSummary(archive)}\n\nYou are authoring candidate ${i + 1} of ${populationSize} this generation; explore a distinct region of the strategy space from your siblings.`;
4085
4165
  try {
4086
4166
  const authored = await authorStrategy({
4087
- chat: cfg.author.chat,
4088
- ...cfg.author.model ? { model: cfg.author.model } : {},
4089
- ...cfg.author.fallbackModel ? { fallbackModel: cfg.author.fallbackModel } : {},
4090
- ...cfg.author.temperature !== void 0 ? { temperature: cfg.author.temperature } : {},
4091
- ...cfg.author.maxTokens !== void 0 ? { maxTokens: cfg.author.maxTokens } : {},
4167
+ profile: cfg.author.profile,
4168
+ executor: cfg.author.executor,
4169
+ ...cfg.author.fallbackProfile ? { fallbackProfile: cfg.author.fallbackProfile } : {},
4092
4170
  contract,
4093
4171
  environmentName: cfg.environment.name,
4094
4172
  lossesJson,
@@ -4225,24 +4303,24 @@ async function runStrategyEvolution(cfg) {
4225
4303
  const tolerance = cfg.reproducerCheck.tolerance ?? .05;
4226
4304
  const championHoldoutScore = holdout.perStrategy[incumbent.name]?.score ?? 0;
4227
4305
  try {
4228
- const summary = (await cfg.author.chat.chat({
4229
- ...cfg.author.model ? { model: cfg.author.model } : {},
4230
- temperature: .2,
4231
- maxTokens: 512,
4232
- messages: [{
4233
- role: "system",
4234
- content: `Summarize the optimization strategy implemented by this code in at most ${words} words. Describe the COMPOSITION (shots, critique, artifact handling, restarts, stopping) — not the code. Output only the summary.`
4235
- }, {
4236
- role: "user",
4237
- content: championCode
4238
- }]
4239
- })).content.trim();
4306
+ const summary = (await profileChatClient({
4307
+ profile: {
4308
+ ...cfg.author.profile,
4309
+ prompt: {
4310
+ ...cfg.author.profile.prompt,
4311
+ systemPrompt: `Summarize the optimization strategy implemented by this code in at most ${words} words. Describe the COMPOSITION (shots, critique, artifact handling, restarts, stopping) — not the code. Output only the summary.`
4312
+ }
4313
+ },
4314
+ executor: cfg.author.executor,
4315
+ context: "strategy reproducer summary"
4316
+ }).chat({ messages: [{
4317
+ role: "user",
4318
+ content: championCode
4319
+ }] })).content.trim();
4240
4320
  const reproduced = await authorStrategy({
4241
- chat: cfg.author.chat,
4242
- ...cfg.author.model ? { model: cfg.author.model } : {},
4243
- ...cfg.author.fallbackModel ? { fallbackModel: cfg.author.fallbackModel } : {},
4244
- ...cfg.author.maxTokens !== void 0 ? { maxTokens: cfg.author.maxTokens } : {},
4245
- temperature: .2,
4321
+ profile: cfg.author.profile,
4322
+ executor: cfg.author.executor,
4323
+ ...cfg.author.fallbackProfile ? { fallbackProfile: cfg.author.fallbackProfile } : {},
4246
4324
  contract: `${strategyAuthorContract}\n\nIMPLEMENT EXACTLY THIS STRATEGY (a colleague's description — do not invent a different approach):\n${summary}`,
4247
4325
  environmentName: cfg.environment.name,
4248
4326
  lossesJson: "[]",
@@ -4289,631 +4367,121 @@ async function runStrategyEvolution(cfg) {
4289
4367
  };
4290
4368
  }
4291
4369
  //#endregion
4292
- //#region src/runtime/stream-agent-turn.ts
4293
- /**
4294
- * `streamAgentTurn` — the ONE run-a-turn event-stream contract over every
4295
- * execution substrate: a sandbox box (`SandboxInstance.streamPrompt`), a
4296
- * one-shot `Executor` (cli-bridge / router / BYO, via `ExecutorFactory`), and
4297
- * an in-process `AgentExecutionBackend` (the `resolveAgentBackend` output).
4298
- *
4299
- * One function, one vocabulary: every backend kind yields the existing
4300
- * `RuntimeStreamEvent` union incrementally and ALWAYS terminates with a
4301
- * `final` event whose `text` is the turn's final text and whose
4302
- * `metadata.tokenUsage` / `metadata.costUsd` / `metadata.model` carry the
4303
- * turn's metered usage. `collectAgentTurn` drains a stream into that terminal
4304
- * summary plus the full event list.
4305
- *
4306
- * This is a UNIFICATION seam, not a new stream parser — each kind is a thin
4307
- * adapter over code that already exists and is already hardened:
4308
- * - `box` — `mapSandboxEvent` + `extractLlmCallEvent` (sandbox-events.ts)
4309
- * project the sandbox event stream; nothing is re-mapped here.
4310
- * - `executor` — `inlineSandboxClient` (the ONE executor→box adapter) turns
4311
- * the factory into a box, then the box path drives it. The
4312
- * executor's settle/teardown lifecycle stays in that adapter.
4313
- * - `chat` — the backend's own `stream()` surface, normalized by
4314
- * `normalizeBackendStreamEvent` (the same projection
4315
- * `runAgentTaskStream` applies).
4316
- *
4317
- * Distinct from `openSandboxRun` (box-only, session resume over one persistent
4318
- * artifact, raw `SandboxEvent` deliverables) and from `runAgentTaskStream`
4319
- * (full task lifecycle: knowledge preflight, session store, resume). This is
4320
- * the minimal turn primitive underneath both worlds: prompt in, one normalized
4321
- * event stream out, terminal result+usage guaranteed on every non-thrown path.
4322
- *
4323
- * Stream envelope: `backend_start` → incremental events → (`backend_error` on
4324
- * failure) → `final`. A caller-initiated abort terminates with
4325
- * `final.status: 'aborted'`; an expired `timeoutMs` deadline with
4326
- * `final.status: 'failed'` — so cancellation stays distinguishable from a
4327
- * blown deadline.
4328
- *
4329
- * Mid-stream lifecycle work needs NO extra API: the generator is pull-based,
4330
- * so the producer is suspended between yields and resumes only when the caller
4331
- * pulls again. A consumer can therefore run arbitrary async work between
4332
- * events — sync state on each `tool_result`, decide a no-op retry after
4333
- * draining, run a pre-`done` flush when it receives `final` and BEFORE it
4334
- * forwards its own terminal event downstream. The interleaving is guaranteed
4335
- * (and locked by test): nothing is produced past the event the caller is
4336
- * holding.
4337
- *
4338
- * @experimental
4339
- */
4370
+ //#region src/runtime/supervise/chat-transport-executor.ts
4340
4371
  /**
4341
- * Run ONE agent turn on any backend kind and stream its events. Yields the
4342
- * `RuntimeStreamEvent` vocabulary incrementally and always ends with a `final`
4343
- * event carrying the turn's text and usage (`metadata.tokenUsage`,
4344
- * `metadata.costUsd?`, `metadata.model?`) — on success, failure, abort, and
4345
- * timeout alike. The generator never throws; failures surface in-band as
4346
- * `backend_error` + `final` with a typed `error` detail.
4372
+ * A session-owning composition over Runtime's canonical Router tool-loop executor.
4347
4373
  *
4348
- * @experimental
4349
- */
4350
- async function* streamAgentTurn(backend, prompt, opts = {}) {
4351
- const label = backend.kind === "chat" ? backend.backend.kind : backend.kind;
4352
- const task = {
4353
- id: `turn-${crypto.randomUUID()}`,
4354
- intent: prompt
4355
- };
4356
- const acc = {
4357
- deltaText: "",
4358
- input: 0,
4359
- output: 0,
4360
- costUsd: 0
4361
- };
4362
- const deadline = deriveTurnSignal(opts.signal, opts.timeoutMs ?? 0);
4363
- let session;
4364
- try {
4365
- session = await startTurnSession(backend, task, prompt, deadline.signal, label);
4366
- yield {
4367
- type: "backend_start",
4368
- task,
4369
- session,
4370
- backend: label,
4371
- timestamp: nowIso()
4372
- };
4373
- const inner = backend.kind === "chat" ? driveChatTurn(backend.backend, task, session, prompt, deadline.signal, acc) : driveBoxTurn(backend.kind === "executor" ? await inlineSandboxClient(backend.factory).create() : backend.box, prompt, deadline.signal, backend.agentRunName ?? "agent", acc, {
4374
- ...backend.kind !== "executor" && backend.options ? { options: backend.options } : {},
4375
- preserveToolParts: opts.preserveToolParts === true,
4376
- ...opts.onRawEvent ? { onRawEvent: opts.onRawEvent } : {}
4377
- });
4378
- for await (const event of inner) {
4379
- yield event;
4380
- throwIfAborted(deadline.signal);
4381
- }
4382
- yield buildFinalEvent(task, session, acc, {
4383
- status: "completed",
4384
- reason: "turn completed"
4385
- });
4386
- } catch (err) {
4387
- const callerAborted = opts.signal?.aborted === true;
4388
- const status = callerAborted ? "aborted" : "failed";
4389
- const message = err instanceof Error ? err.message : String(err);
4390
- const error = err instanceof BackendTransportError ? {
4391
- kind: "transport",
4392
- message,
4393
- status: err.status,
4394
- body: err.body
4395
- } : {
4396
- kind: "backend",
4397
- message
4398
- };
4399
- yield {
4400
- type: "backend_error",
4401
- task,
4402
- ...session ? { session } : {},
4403
- backend: label,
4404
- message,
4405
- recoverable: !callerAborted,
4406
- error,
4407
- timestamp: nowIso()
4408
- };
4409
- yield buildFinalEvent(task, session, acc, {
4410
- status,
4411
- reason: message,
4412
- error
4413
- });
4414
- } finally {
4415
- deadline.dispose();
4416
- }
4417
- }
4418
- /**
4419
- * Drain a `streamAgentTurn` stream (or any `RuntimeStreamEvent` stream that
4420
- * honors its terminal contract) into the turn summary plus the full event
4421
- * list. Fail-loud: throws when the stream ends without a terminal `final`
4422
- * event — a stream that violates the contract must not read as an empty turn.
4374
+ * This module adds conversation persistence for graph-edge `resume` continuity. It does not own
4375
+ * model selection, prompts, generation controls, retries, tool policy, or provider accounting:
4376
+ * those are lowered from one exact `AgentProfile` by `createExecutor({ backend: 'router-tools' })`.
4423
4377
  *
4424
4378
  * @experimental
4425
4379
  */
4426
- async function collectAgentTurn(stream) {
4427
- const events = [];
4428
- for await (const event of stream) events.push(event);
4429
- const final = events.at(-1);
4430
- if (final?.type !== "final") throw new Error(`collectAgentTurn: stream ended without a terminal 'final' event (last: ${final ? final.type : "none"})`);
4431
- const metadata = final.metadata ?? {};
4432
- const tokenUsage = metadata.tokenUsage && typeof metadata.tokenUsage === "object" ? metadata.tokenUsage : {};
4433
- const usage = {
4434
- input: finiteNumber(tokenUsage.input) ?? 0,
4435
- output: finiteNumber(tokenUsage.output) ?? 0
4436
- };
4437
- const costUsd = finiteNumber(metadata.costUsd);
4438
- if (costUsd !== void 0) usage.costUsd = costUsd;
4439
- if (typeof metadata.model === "string" && metadata.model.length > 0) usage.model = metadata.model;
4440
- return {
4441
- finalText: final.text ?? "",
4442
- usage,
4443
- events,
4444
- status: final.status,
4445
- ...final.error ? { error: final.error } : {}
4446
- };
4447
- }
4448
- /** Start the backend's session when it owns one (`chat` kind); mint a local
4449
- * correlation session otherwise. Box/executor turns carry no server session
4450
- * here — resume lives in `openSandboxRun`/`SandboxLineage`, not this primitive. */
4451
- async function startTurnSession(backend, task, prompt, signal, label) {
4452
- if (backend.kind === "chat" && backend.backend.start) return backend.backend.start({
4453
- task,
4454
- message: prompt
4455
- }, {
4456
- task,
4457
- knowledge: emptyReadiness(task),
4458
- signal
4459
- });
4460
- return newRuntimeSession(label);
4461
- }
4462
- /**
4463
- * One turn over a box: `box.streamPrompt` projected through the existing
4464
- * `mapSandboxEvent` (text/reasoning deltas +
4465
- * cost-bearing `llm_call`s), plus the opt-in `mapSandboxToolEvent` tool-part
4466
- * projection. Usage accumulates off the mapped `llm_call` events — the same
4467
- * fold `sumSandboxUsage` applies. Final text prefers the terminal
4468
- * `result`/`done`/`final` payload over concatenated deltas, because the
4469
- * sandbox `message.part.updated` fallback may carry running accumulations.
4470
- */
4471
- async function* driveBoxTurn(box, prompt, signal, agentRunName, acc, cfg) {
4472
- const callOptions = {
4473
- ...cfg.options ?? {},
4474
- signal
4475
- };
4476
- const stream = box.streamPrompt(prompt, callOptions);
4477
- const toolParts = cfg.preserveToolParts ? createSandboxToolPartState() : void 0;
4478
- for await (const event of stream) {
4479
- if (cfg.onRawEvent) await cfg.onRawEvent(event);
4480
- const terminalText = terminalTextFromSandboxEvent(event);
4481
- if (terminalText !== void 0) acc.terminalText = terminalText;
4482
- if (toolParts) for (const toolEvent of mapSandboxToolEvent(event, toolParts)) yield toolEvent;
4483
- const mapped = mapSandboxEvent(event, { agentRunName });
4484
- if (!mapped) continue;
4485
- foldEvent(mapped, acc, agentRunName);
4486
- yield mapped;
4487
- }
4488
- }
4489
- /** One turn over an in-process backend: its own `stream()` surface, projected
4490
- * through the same `normalizeBackendStreamEvent` the task lifecycle applies. */
4491
- async function* driveChatTurn(backend, task, session, prompt, signal, acc) {
4492
- const input = {
4493
- task,
4494
- message: prompt
4495
- };
4496
- const context = {
4497
- task,
4498
- knowledge: emptyReadiness(task),
4499
- session,
4500
- signal
4501
- };
4502
- for await (const raw of backend.stream(input, context)) {
4503
- const event = normalizeBackendStreamEvent(raw, task, session);
4504
- foldEvent(event, acc);
4505
- yield event;
4506
- }
4507
- }
4508
- /** Fold one normalized event into the turn accumulator (text + usage).
4509
- * `fallbackModelLabel` — a mapper-stamped run label to exclude from
4510
- * `usage.model` (it is not a backend-reported model). */
4511
- function foldEvent(event, acc, fallbackModelLabel) {
4512
- if (event.type === "text_delta") {
4513
- acc.deltaText += event.text;
4514
- return;
4515
- }
4516
- if (event.type === "llm_call") {
4517
- acc.input += event.tokensIn ?? 0;
4518
- acc.output += event.tokensOut ?? 0;
4519
- acc.costUsd += event.costUsd ?? 0;
4520
- if (event.model && event.model !== fallbackModelLabel) acc.model = event.model;
4521
- }
4522
- }
4523
- /** Read the final text off a terminal sandbox event, when present. */
4524
- function terminalTextFromSandboxEvent(event) {
4525
- if (!event || typeof event !== "object") return void 0;
4526
- const type = String(event.type ?? "");
4527
- if (type !== "result" && type !== "done" && type !== "final") return void 0;
4528
- const data = event.data && typeof event.data === "object" ? event.data : {};
4529
- for (const key of [
4530
- "finalText",
4531
- "text",
4532
- "response",
4533
- "content"
4534
- ]) {
4535
- const value = data[key];
4536
- if (typeof value === "string") return value;
4537
- }
4538
- }
4539
- function buildFinalEvent(task, session, acc, outcome) {
4540
- const finalText = acc.terminalText ?? acc.deltaText;
4380
+ /** In-memory, process-local conversation store with detached reads and writes. */
4381
+ function createChatSessionStore() {
4382
+ const sessions = /* @__PURE__ */ new Map();
4541
4383
  return {
4542
- type: "final",
4543
- task,
4544
- ...session ? { session } : {},
4545
- status: outcome.status,
4546
- reason: outcome.reason,
4547
- ...finalText ? { text: finalText } : {},
4548
- metadata: {
4549
- tokenUsage: {
4550
- input: acc.input,
4551
- output: acc.output
4552
- },
4553
- ...acc.costUsd > 0 ? { costUsd: acc.costUsd } : {},
4554
- ...acc.model ? { model: acc.model } : {}
4384
+ load(workerId) {
4385
+ const messages = sessions.get(workerId);
4386
+ return messages === void 0 ? void 0 : structuredClone(messages);
4555
4387
  },
4556
- ...outcome.error ? { error: outcome.error } : {},
4557
- timestamp: nowIso()
4558
- };
4559
- }
4560
- /** Minimal ready-by-construction readiness report for a requirement-free turn. */
4561
- function emptyReadiness(task) {
4562
- return scoreKnowledgeReadiness({
4563
- taskId: task.id,
4564
- requirements: []
4565
- });
4566
- }
4567
- function finiteNumber(value) {
4568
- return typeof value === "number" && Number.isFinite(value) ? value : void 0;
4569
- }
4570
- function throwIfAborted(signal) {
4571
- if (!signal.aborted) return;
4572
- throw signal.reason instanceof Error ? signal.reason : new Error(String(signal.reason));
4573
- }
4574
- /**
4575
- * Derive the turn's effective abort signal: fires when EITHER the caller's
4576
- * signal aborts OR the `timeoutMs` deadline elapses. `dispose()` clears the
4577
- * timer so a finished turn never leaks a pending timeout. `timeoutMs <= 0`
4578
- * disables the deadline. Node-portable (no `AbortSignal.any`, which needs
4579
- * >=20.3 — the package floor is >=20).
4580
- */
4581
- function deriveTurnSignal(callerSignal, timeoutMs) {
4582
- const controller = new AbortController();
4583
- const timer = timeoutMs > 0 ? setTimeout(() => controller.abort(/* @__PURE__ */ new Error(`agent turn timed out after ${timeoutMs}ms`)), timeoutMs) : void 0;
4584
- if (timer && typeof timer.unref === "function") timer.unref();
4585
- const onCallerAbort = () => controller.abort(callerSignal?.reason ?? /* @__PURE__ */ new Error("agent turn aborted"));
4586
- if (callerSignal) if (callerSignal.aborted) onCallerAbort();
4587
- else callerSignal.addEventListener("abort", onCallerAbort, { once: true });
4588
- return {
4589
- signal: controller.signal,
4590
- dispose: () => {
4591
- if (timer) clearTimeout(timer);
4592
- callerSignal?.removeEventListener("abort", onCallerAbort);
4388
+ save(workerId, messages) {
4389
+ sessions.set(workerId, structuredClone(messages));
4593
4390
  }
4594
4391
  };
4595
4392
  }
4596
- //#endregion
4597
- //#region src/runtime/supervise/chat-transport-executor.ts
4598
- /**
4599
- * The chat-transport leaf executor: a worker whose runtime is a plain OpenAI-compatible
4600
- * `/v1/chat/completions` transport — the worker IS a model conversation, not a sandboxed process
4601
- * (#721). Tool calls are optional (none, or a caller-provided tool table executed on this host).
4602
- * A chat worker gets everything real workers get through the open `Executor` port: node pinning,
4603
- * conserved spend, settle/verdict, journal + edge ledger.
4604
- *
4605
- * Module home: a standalone leaf-executor module beside `worktree-cli-executor.ts` — a direct
4606
- * `(options) Executor` constructor, NOT a `createExecutor` backend variant. The reason is
4607
- * continuity: `workerFromBackend` (the backend-as-data path every `ExecutorConfig` rides) creates
4608
- * a fresh executor per spawn with no session re-attachment and deliberately FAILS LOUD on a
4609
- * `continuity: 'resume'` spawn; the documented resume consumer is a session-owning
4610
- * `makeWorkerAgent` seam. {@link chatWorkerSeam} is that seam, and this module ships both halves
4611
- * together so no caller re-derives the resume wiring.
4612
- *
4613
- * Transport shape: NON-streaming, one buffered POST per turn the simplest honest choice.
4614
- * A streaming executor cannot mark an unmetered turn today (`UsageEvent`'s `tokens` variant has
4615
- * no `tokensKnown: false` twin — see the documented limitation in `./types`), while the one-shot
4616
- * path returns a whole `Spend` that carries both markers. Honesty wins over liveness here.
4617
- *
4618
- * Metering: tokens come from the transport's `usage` fields; a turn without usage marks
4619
- * `tokensKnown: false`. Dollars come ONLY from the response's own cost fields (`usage.cost` /
4620
- * `usage.cost_usd`, the cli-bridge and OpenRouter conventions); a turn without one marks
4621
- * `usdKnown: false`. NEVER estimated from a local price table — this executor speaks to arbitrary
4622
- * OpenAI-compatible endpoints whose models a local table cannot price, and a silent estimate is a
4623
- * fabricated measurement.
4624
- *
4625
- * @experimental
4626
- */
4627
- /** The default transport: POST `${url}/chat/completions` with an optional bearer. Fail-loud on
4628
- * any non-2xx — the status and body head become the settle reason. */
4629
- function chatCompletionsTransport(opts) {
4630
- if (typeof opts.url !== "string" || opts.url.length === 0) throw new ValidationError("chatCompletionsTransport: url required");
4631
- const endpoint = `${opts.url.replace(/\/$/, "")}/chat/completions`;
4632
- return async (body, signal) => {
4633
- const res = await fetch(endpoint, {
4634
- method: "POST",
4635
- headers: {
4636
- "content-type": "application/json",
4637
- ...opts.bearer ? { authorization: `Bearer ${opts.bearer}` } : {}
4393
+ function exactProfile(profile, context) {
4394
+ const parsed = agentProfileSchema.safeParse(profile);
4395
+ if (!parsed.success) throw new ValidationError(`${context}: invalid AgentProfile: ${parsed.error.message}`);
4396
+ assertExecutableAgentProfile(parsed.data, context);
4397
+ return parsed.data;
4398
+ }
4399
+ function initialMessages(opts) {
4400
+ if (!opts.resume) return void 0;
4401
+ if (!opts.sessions) throw new ValidationError("chat transport: a 'resume' spawn needs the session store holding the prior conversation");
4402
+ const prior = opts.sessions.load(opts.resume.ofWorker);
4403
+ if (prior === void 0) throw new ValidationError(`chat transport: no recorded conversation for worker '${opts.resume.ofWorker}'`);
4404
+ return prior;
4405
+ }
4406
+ function executorConfig(opts) {
4407
+ const profile = exactProfile(opts.profile, "chat transport");
4408
+ if (!opts.complete && !opts.url) throw new ValidationError("chat transport: url required unless complete is injected");
4409
+ const tools = opts.tools ?? [];
4410
+ for (const tool of tools) if (!tool.spec.function.name || typeof tool.execute !== "function") throw new ValidationError("chat transport: every tool needs spec.function.name and execute");
4411
+ const resumed = initialMessages(opts);
4412
+ return {
4413
+ profile,
4414
+ config: {
4415
+ backend: "router-tools",
4416
+ routerBaseUrl: opts.url ?? "http://injected.invalid",
4417
+ routerKey: opts.bearer ?? (opts.complete ? "injected-transport" : ""),
4418
+ tools: tools.map((tool) => tool.spec),
4419
+ executeToolCall: async (name, args, task) => {
4420
+ const tool = tools.find((candidate) => candidate.spec.function.name === name);
4421
+ if (!tool) throw new ValidationError(`chat transport: unknown tool ${JSON.stringify(name)}`);
4422
+ return tool.execute(args, task);
4638
4423
  },
4639
- body: JSON.stringify(body),
4640
- ...signal ? { signal } : {}
4641
- });
4642
- if (!res.ok) throw new ValidationError(`chat transport ${res.status}: ${(await res.text()).slice(0, 200)}`);
4643
- return res.json();
4424
+ ...opts.complete ? { complete: opts.complete } : {},
4425
+ ...resumed ? { initialMessages: resumed } : {},
4426
+ ...opts.sessions && opts.sessionKey ? { onMessages: (messages) => {
4427
+ opts.sessions?.save(opts.sessionKey, messages);
4428
+ } } : {}
4429
+ }
4644
4430
  };
4645
4431
  }
4646
- /** In-memory `ChatSessionStore`. Entries are detached copies — a caller mutating a saved array
4647
- * cannot corrupt a recorded session. */
4648
- function createChatSessionStore() {
4649
- const sessions = /* @__PURE__ */ new Map();
4650
- return {
4651
- load: (workerId) => sessions.get(workerId),
4652
- save: (workerId, messages) => {
4653
- sessions.set(workerId, structuredClone(messages));
4654
- }
4432
+ function buildChatTransportExecutor(opts, context) {
4433
+ const { config, profile } = executorConfig(opts);
4434
+ const spec = {
4435
+ profile,
4436
+ harness: null
4655
4437
  };
4438
+ return mapExecutorResult(createExecutor(config)(spec, context), (result) => {
4439
+ const raw = result.out;
4440
+ const content = typeof raw?.content === "string" ? raw.content : "";
4441
+ return {
4442
+ outRef: contentAddress({
4443
+ kind: "chat-transport",
4444
+ profile,
4445
+ content
4446
+ }),
4447
+ out: content,
4448
+ ...result.verdict ? { verdict: result.verdict } : {}
4449
+ };
4450
+ });
4656
4451
  }
4657
- const CHAT_TRANSPORT_RUNTIME = "chat-transport";
4658
4452
  /**
4659
- * Build the chat-transport `Executor`: one `execute` = one conversation SHOT — seed (fresh system
4660
- * prompt, or the resumed session's recorded history) + the task as the next user message, then
4661
- * loop completion → host tool calls → tool messages until the model answers without a tool call
4662
- * (or the turn cap). Settles with the final assistant text as `out`.
4663
- *
4664
- * Fail-loud contract: transport failures (non-2xx, network faults, malformed completions) throw
4665
- * `ValidationError`, which the scope settles as an INFRA failure (`Settled.down.infra`) — never a
4666
- * fake success. The accumulated conversation is still recorded before the throw when a store is
4667
- * configured, because the inference HAPPENED and a resume may continue a failed session (the
4668
- * kernel deliberately allows resume-after-failure; the seam decides).
4453
+ * Build one exact profile-driven chat executor through `createExecutor`.
4454
+ * Prefer `chatWorkerSeam` for supervised work because it supplies trusted node identity.
4669
4455
  */
4670
4456
  function chatTransportExecutor(opts) {
4671
- const model = concreteModelId(opts.model);
4672
- if (!model) throw new ValidationError("chatTransportExecutor: model required");
4673
- if (!opts.complete && (typeof opts.url !== "string" || opts.url.length === 0)) throw new ValidationError("chatTransportExecutor: url required (or inject `complete`)");
4674
- for (const tool of opts.tools ?? []) if (typeof tool.spec?.function?.name !== "string" || typeof tool.execute !== "function") throw new ValidationError("chatTransportExecutor: every tools entry needs spec.function.name + execute");
4675
- const maxTurns = opts.maxTurnsPerShot ?? 200;
4676
- if (!Number.isInteger(maxTurns) || maxTurns < 1) throw new ValidationError("chatTransportExecutor: maxTurnsPerShot must be a positive integer");
4677
- if (opts.maxTokens !== void 0 && (!Number.isInteger(opts.maxTokens) || opts.maxTokens < 1)) throw new ValidationError("chatTransportExecutor: maxTokens must be a positive integer");
4678
- let seed;
4679
- if (opts.resume) {
4680
- if (!opts.sessions) throw new ValidationError("chatTransportExecutor: a 'resume' spawn needs `sessions` — the store holding the conversation this shot continues");
4681
- const prior = opts.sessions.load(opts.resume.ofWorker);
4682
- if (prior === void 0) throw new ValidationError(`chatTransportExecutor: no recorded conversation for worker '${opts.resume.ofWorker}' — the session store holds only conversations recorded by this process (the kernel’s process-local resume boundary)`);
4683
- seed = structuredClone(prior);
4684
- } else seed = opts.system !== void 0 && opts.system.length > 0 ? [{
4685
- role: "system",
4686
- content: opts.system
4687
- }] : [];
4688
- const transport = opts.complete ?? chatCompletionsTransport({
4689
- url: opts.url,
4690
- ...opts.bearer ? { bearer: opts.bearer } : {}
4691
- });
4692
- const toolSpecs = (opts.tools ?? []).map((tool) => tool.spec);
4693
- const toolByName = new Map((opts.tools ?? []).map((tool) => [tool.spec.function.name, tool]));
4694
- const controller = new AbortController();
4695
- let artifact;
4696
- let executed = false;
4697
- const executionId = opts.sessionKey ?? `chat-session-${randomUUID()}`;
4698
- const attemptId = opts.attemptId ?? newExecutionAttemptId(executionId);
4699
- const executor = {
4700
- runtime: CHAT_TRANSPORT_RUNTIME,
4701
- async execute(task, signal) {
4702
- if (executed) throw new ValidationError("chatTransportExecutor: execute() called twice on one instance");
4703
- executed = true;
4704
- const started = Date.now();
4705
- const messages = seed;
4706
- messages.push({
4707
- role: "user",
4708
- content: taskToPrompt(task)
4709
- });
4710
- const linked = mergeAbortSignals(signal, controller.signal);
4711
- const tokens = zeroTokenUsage();
4712
- let tokensKnown = true;
4713
- let usd = 0;
4714
- let usdKnown = true;
4715
- let turns = 0;
4716
- let lastText = "";
4717
- try {
4718
- for (let t = 0; t < maxTurns; t += 1) {
4719
- const body = {
4720
- model,
4721
- messages,
4722
- ...toolSpecs.length > 0 ? {
4723
- tools: toolSpecs,
4724
- tool_choice: "auto"
4725
- } : {},
4726
- ...opts.temperature !== void 0 ? { temperature: opts.temperature } : {},
4727
- ...opts.maxTokens !== void 0 ? { max_tokens: opts.maxTokens } : {}
4728
- };
4729
- let raw;
4730
- try {
4731
- raw = await transport(body, linked);
4732
- } catch (cause) {
4733
- if (cause instanceof Error && cause.name === "AbortError") throw cause;
4734
- if (cause instanceof ValidationError) throw cause;
4735
- throw new ValidationError(`chatTransportExecutor: transport failed: ${cause instanceof Error ? cause.message : String(cause)}`);
4736
- }
4737
- turns += 1;
4738
- const data = raw;
4739
- const usage = data?.usage;
4740
- if (usage && typeof usage.prompt_tokens === "number" && typeof usage.completion_tokens === "number") {
4741
- tokens.input += usage.prompt_tokens;
4742
- tokens.output += usage.completion_tokens;
4743
- } else tokensKnown = false;
4744
- const turnCost = typeof usage?.cost === "number" ? usage.cost : typeof usage?.cost_usd === "number" ? usage.cost_usd : void 0;
4745
- if (turnCost !== void 0) usd += turnCost;
4746
- else usdKnown = false;
4747
- const msg = data?.choices?.[0]?.message;
4748
- if (msg === void 0) throw new ValidationError("chatTransportExecutor: transport returned no choices[0].message");
4749
- if (typeof msg.content === "string" && msg.content.length > 0) lastText = msg.content;
4750
- const toolCalls = msg.tool_calls ?? [];
4751
- if (toolCalls.length === 0 || toolSpecs.length === 0) {
4752
- messages.push({
4753
- role: "assistant",
4754
- content: msg.content ?? ""
4755
- });
4756
- break;
4757
- }
4758
- messages.push({
4759
- role: "assistant",
4760
- content: msg.content ?? "",
4761
- tool_calls: toolCalls.map((tc, i) => ({
4762
- id: tc.id ?? `call_${i}`,
4763
- type: "function",
4764
- function: {
4765
- name: tc.function?.name ?? "",
4766
- arguments: tc.function?.arguments ?? "{}"
4767
- }
4768
- }))
4769
- });
4770
- for (let i = 0; i < toolCalls.length; i += 1) {
4771
- const tc = toolCalls[i];
4772
- const id = tc?.id ?? `call_${i}`;
4773
- const name = tc?.function?.name ?? "";
4774
- const tool = toolByName.get(name);
4775
- if (!tool) {
4776
- messages.push({
4777
- role: "tool",
4778
- tool_call_id: id,
4779
- content: `error: unknown tool '${name}'`
4780
- });
4781
- continue;
4782
- }
4783
- let args;
4784
- try {
4785
- args = JSON.parse(tc?.function?.arguments ?? "{}");
4786
- } catch {
4787
- messages.push({
4788
- role: "tool",
4789
- tool_call_id: id,
4790
- content: "error: tool arguments were not valid JSON"
4791
- });
4792
- continue;
4793
- }
4794
- let result;
4795
- try {
4796
- result = await tool.execute(args, task);
4797
- } catch (cause) {
4798
- result = `error: ${cause instanceof Error ? cause.message : String(cause)}`;
4799
- }
4800
- messages.push({
4801
- role: "tool",
4802
- tool_call_id: id,
4803
- content: result
4804
- });
4805
- }
4806
- }
4807
- } finally {
4808
- if (opts.sessions && opts.sessionKey !== void 0) opts.sessions.save(opts.sessionKey, messages);
4809
- }
4810
- const spent = {
4811
- iterations: turns,
4812
- tokens,
4813
- ...tokensKnown ? {} : { tokensKnown: false },
4814
- usd,
4815
- ...usdKnown ? {} : { usdKnown: false },
4816
- ms: Date.now() - started
4817
- };
4818
- artifact = {
4819
- outRef: contentAddress({
4820
- kind: "chat-transport",
4821
- model,
4822
- content: lastText,
4823
- turns
4824
- }),
4825
- out: lastText,
4826
- spent
4827
- };
4828
- return artifact;
4829
- },
4830
- teardown(_grace) {
4831
- controller.abort();
4832
- return Promise.resolve({ destroyed: true });
4833
- },
4834
- resultArtifact() {
4835
- if (!artifact) throw new ValidationError("chatTransportExecutor: resultArtifact() read before execute()");
4836
- return {
4837
- ...artifact,
4838
- spent: artifact.spent
4839
- };
4840
- }
4841
- };
4842
- if (opts.profile === void 0) return executor;
4843
- return attestRuntimeOwnedExecutor(executor, {
4844
- effectiveProfile: opts.profile,
4845
- backend: "chat-transport",
4846
- model: {
4847
- status: "known",
4848
- id: model
4849
- },
4850
- execution: {
4851
- kind: "session",
4852
- id: executionId
4853
- },
4854
- materializer: "chat-transport-conversation",
4855
- plan: {
4856
- kind: "openai-chat-conversation",
4857
- model,
4858
- maxTurnsPerShot: maxTurns,
4859
- tools: toolSpecs,
4860
- resumeOf: opts.resume?.ofWorker ?? null
4861
- }
4862
- }, {
4863
- attemptId,
4864
- binding: {
4865
- endpoint: opts.complete ? "injected-transport" : opts.url,
4866
- model,
4867
- sessionKey: opts.sessionKey ?? null
4868
- },
4869
- descriptor: {
4870
- kind: "chat-transport-session",
4871
- transport: opts.complete ? "injected" : "http",
4872
- backend: "chat-transport"
4873
- }
4457
+ return buildChatTransportExecutor(opts, {
4458
+ signal: new AbortController().signal,
4459
+ seams: {}
4874
4460
  });
4875
4461
  }
4876
- /**
4877
- * The `makeWorkerAgent` seam over {@link chatTransportExecutor} — the continuity consumer
4878
- * `workerFromBackend` refuses to be. Every spawn becomes one conversation shot: the spawned
4879
- * profile's system prompt + instructions (which is where a graph's delegates directive lands)
4880
- * seed a fresh session, and a `'resume'` spawn re-attaches by loading `resume.ofWorker`'s
4881
- * recorded message list from the seam's session store. Conversations are recorded under the
4882
- * kernel node id, which is exactly what a later `resume.ofWorker` names.
4883
- */
4462
+ /** Session-owning worker factory for graph continuity. */
4884
4463
  function chatWorkerSeam(opts) {
4885
- if (!opts.complete && (typeof opts.url !== "string" || opts.url.length === 0)) throw new ValidationError("chatWorkerSeam: url required (or inject `complete`)");
4464
+ if (!opts.complete && !opts.url) throw new ValidationError("chatWorkerSeam: url required unless complete is injected");
4886
4465
  const sessions = opts.sessions ?? createChatSessionStore();
4887
4466
  return (rawProfile, spawnContext) => {
4888
- const parsed = agentProfileSchema.safeParse(canonicalizeAuthoredProfile(rawProfile));
4889
- if (!parsed.success) throw new ValidationError(`chatWorkerSeam: invalid AgentProfile: ${parsed.error.message}`);
4890
- const profile = parsed.data;
4891
- const model = concreteProfileModel(profile) ?? concreteModelId(opts.model);
4892
- if (!model) throw new ValidationError("chatWorkerSeam: no model — set ChatWorkerSeamOptions.model or AgentProfile.model.default");
4893
- const system = [profile.prompt?.systemPrompt, ...profile.prompt?.instructions ?? []].filter((line) => typeof line === "string" && line.trim().length > 0).join("\n");
4467
+ const profile = exactProfile(rawProfile, "chatWorkerSeam");
4894
4468
  return {
4895
4469
  name: profile.name ?? "chat-worker",
4896
4470
  act: async () => void 0,
4897
4471
  executorSpec: {
4898
4472
  profile,
4899
4473
  harness: null,
4900
- executorFactory: (executorSpec, ctx) => {
4901
- const executor = chatTransportExecutor({
4902
- url: opts.url,
4903
- ...opts.bearer !== void 0 ? { bearer: opts.bearer } : {},
4904
- model,
4905
- ...system.length > 0 ? { system } : {},
4906
- ...opts.tools !== void 0 ? { tools: opts.tools } : {},
4907
- ...opts.temperature !== void 0 ? { temperature: opts.temperature } : {},
4908
- ...opts.maxTokens !== void 0 ? { maxTokens: opts.maxTokens } : {},
4909
- ...opts.maxTurnsPerShot !== void 0 ? { maxTurnsPerShot: opts.maxTurnsPerShot } : {},
4910
- ...opts.complete !== void 0 ? { complete: opts.complete } : {},
4911
- sessions,
4912
- ...ctx.node?.nodeId !== void 0 ? { sessionKey: ctx.node.nodeId } : {},
4913
- ...spawnContext?.resume !== void 0 ? { resume: spawnContext.resume } : {},
4474
+ executorFactory: (executorSpec, context) => {
4475
+ const executor = buildChatTransportExecutor({
4914
4476
  profile: executorSpec.profile,
4915
- ...ctx.node?.attemptId !== void 0 ? { attemptId: ctx.node.attemptId } : {}
4916
- });
4477
+ ...opts.url ? { url: opts.url } : {},
4478
+ ...opts.bearer ? { bearer: opts.bearer } : {},
4479
+ ...opts.tools ? { tools: opts.tools } : {},
4480
+ ...opts.complete ? { complete: opts.complete } : {},
4481
+ sessions,
4482
+ ...context.node?.nodeId ? { sessionKey: context.node.nodeId } : {},
4483
+ ...spawnContext?.resume ? { resume: spawnContext.resume } : {}
4484
+ }, context);
4917
4485
  return opts.deliverable ? gateOnDeliverable(executor, opts.deliverable) : executor;
4918
4486
  }
4919
4487
  }
@@ -4921,457 +4489,6 @@ function chatWorkerSeam(opts) {
4921
4489
  };
4922
4490
  }
4923
4491
  //#endregion
4924
- //#region src/runtime/supervise/graph.ts
4925
- /**
4926
- *
4927
- * `runGraph` — agent graphs: profiles as nodes, registry-backed prompt directives as edges.
4928
- *
4929
- * A topology is PLAIN DATA an agent can author in a few lines: nodes are canonical
4930
- * `AgentProfile`s (the ONLY way a node is described — no role-builder functions), edges are typed
4931
- * values carrying versioned {@link PromptHandle} directives, `deliverable` (termination) and
4932
- * `budget` (one conserved pool) are mandatory. Driver↔worker is the two-node cyclic instance;
4933
- * "agent 3 analyzes 1 and 2 and reports to 1" is ONE edge, not a framework.
4934
- *
4935
- * NOT A SECOND SCHEDULER. `runGraph` is an interpretation layer over what already runs:
4936
- * `supervise()` is the execution core — the same `supervisorAgent`/`driverAgent` machinery,
4937
- * `makeWorkerAgent` seam, conserved-pool budget, and deliverable-gated settlement every
4938
- * supervised run uses. (`runAgentRounds` is deliberately NOT the substrate here.) What the graph
4939
- * layer ADDS is exactly what a bespoke driver loop never
4940
- * had:
4941
- *
4942
- * 1. **Node pinning** — a spawn names a node (`profile.name` = node id) and the node's canonical
4943
- * profile is what runs; a driver cannot smuggle capabilities into a worker it did not define.
4944
- * 2. **Observable edges** — every delegates/analyzes traversal lands in an EDGE LEDGER
4945
- * (`delivered | stripped | empty | unpropagated`, with byte counts), in memory on
4946
- * the result AND as `edge` events in the run journal. The motivating incident: a filter
4947
- * silently replaced 1,700-char steering with 241 chars of boilerplate for three rounds and
4948
- * NO artifact said so — an unobservable edge cannot be trusted and its directive cannot be
4949
- * optimized.
4950
- * 3. **Directives as data** — edge text lives in the prompt registry (`<surface>/v<n>`), so every
4951
- * edge is a versioned optimization target, never prose hardcoded in a builder function.
4952
- * 4. **Per-edge traversal caps** — the cyclic-graph backstop. A delegates edge whose cap is
4953
- * exhausted REFUSES further traversals (fail loud), so a cycle cannot spin the pool dry.
4954
- * 5. **Continuity as data** — a delegates edge may declare `continuity: 'resume'`, so each spawn
4955
- * after the node's first re-attaches to its latest SETTLED session (the spawn context hands
4956
- * the executor seam `resume: { ofWorker, sequence }`; the kernel keeps identity, ordering,
4957
- * ledger truth, and the one conserved pool). Every ledger row states how its hop continued:
4958
- * `'fresh' | 'resume'` for spawns, `'steer'` for mid-run deliveries — fresh respawns, session
4959
- * resumes, and live steers are all plain data, each a ledgered fact.
4960
- *
4961
- * ORACLES ARE ENVIRONMENT, NEVER WORKERS. Graders/verifiers must not be spawnable in the graph —
4962
- * a delegates edge to them leaks the rubric. An `analyzes` edge names its analyst in one of two
4963
- * forms: a LENS id from the environment's registry (a pure function over trace evidence), or the
4964
- * id of a graph NODE — a tool-equipped analyst AGENT spawned on each matching settle with the
4965
- * node's pinned profile, whose settle output IS the findings. Either way the oracle doctrine
4966
- * holds: an analyst node can never be a delegates target (refused loudly), so no driver can hand
4967
- * it work, and an id living in both the registry and the nodes is refused as ambiguous.
4968
- *
4969
- * @experimental
4970
- */
4971
- /** Default per-edge traversal cap — the cyclic-graph backstop when an edge names none. */
4972
- const defaultEdgeTraversalCap = 32;
4973
- /** A delegates edge exhausted its traversal cap and the run produced no winner: the cap, not the
4974
- * task, ended it. Carries the full evidence so failing loud loses nothing. */
4975
- var GraphEdgeCapError = class extends Error {
4976
- exhaustedEdges;
4977
- ledger;
4978
- result;
4979
- constructor(exhaustedEdges, ledger, result) {
4980
- super(`runGraph: edge traversal cap exhausted on ${exhaustedEdges.join(", ")} and the run delivered no winner — the cap (the cyclic-graph backstop), not the task, ended this run. Raise maxTraversals on the edge or fix the cycle; the full edge ledger and the supervised result ride on this error.`);
4981
- this.name = "GraphEdgeCapError";
4982
- this.exhaustedEdges = exhaustedEdges;
4983
- this.ledger = ledger;
4984
- this.result = result;
4985
- }
4986
- };
4987
- function edgeId(edge) {
4988
- return edge.kind === "delegates" ? `delegates:${edge.from}->${edge.to}` : `analyzes:${edge.analyst}:${edge.over.join("+")}->${edge.to}`;
4989
- }
4990
- /** Validate the graph and resolve every directive BEFORE any compute is spent — an invalid
4991
- * topology or an unknown directive is a configuration fault, never a mid-run surprise. */
4992
- function validateGraph(graph, registry, analysts) {
4993
- if (!Array.isArray(graph.nodes) || graph.nodes.length === 0) throw new ValidationError("runGraph: graph.nodes must be a non-empty array");
4994
- if (!Array.isArray(graph.edges) || graph.edges.length === 0) throw new ValidationError("runGraph: graph.edges must be a non-empty array");
4995
- if (typeof graph.deliverable?.check !== "function") throw new ValidationError("runGraph: graph.deliverable is mandatory (termination oracle)");
4996
- if (typeof graph.budget !== "object" || graph.budget === null) throw new ValidationError("runGraph: graph.budget is mandatory (the conserved pool)");
4997
- const byId = /* @__PURE__ */ new Map();
4998
- for (const node of graph.nodes) {
4999
- if (typeof node.id !== "string" || node.id.length === 0) throw new ValidationError("runGraph: every node needs a non-empty string id");
5000
- if (byId.has(node.id)) throw new ValidationError(`runGraph: duplicate node id '${node.id}'`);
5001
- const parsed = agentProfileSchema.safeParse(node.profile);
5002
- if (!parsed.success) throw new ValidationError(`runGraph: node '${node.id}' has an invalid AgentProfile: ${parsed.error.message}`);
5003
- if (node.profile.name !== node.id) throw new ValidationError(`runGraph: node '${node.id}' has profile.name ${JSON.stringify(node.profile.name)} — profile.name IS the node identity (node pinning and analyst routing match on it) and must equal the node id`);
5004
- byId.set(node.id, node);
5005
- }
5006
- const requireNode = (id, where) => {
5007
- const node = byId.get(id);
5008
- if (!node) throw new ValidationError(`runGraph: ${where} references unknown node '${id}'`);
5009
- return node;
5010
- };
5011
- const delegates = graph.edges.filter((edge) => edge.kind === "delegates");
5012
- const analyzes = graph.edges.filter((edge) => edge.kind === "analyzes");
5013
- if (delegates.length === 0) throw new ValidationError("runGraph: at least one delegates edge is required (who spawns whom)");
5014
- for (const edge of graph.edges) registry.resolve(edge.directive);
5015
- for (const edge of delegates) {
5016
- requireNode(edge.from, edgeId(edge));
5017
- requireNode(edge.to, edgeId(edge));
5018
- if (edge.from === edge.to) throw new ValidationError(`runGraph: ${edgeId(edge)} delegates to itself — the driver↔worker cycle is the settle-return loop, not a self-edge`);
5019
- if (edge.continuity !== void 0 && edge.continuity !== "fresh" && edge.continuity !== "resume") throw new ValidationError(`runGraph: ${edgeId(edge)} has invalid continuity ${JSON.stringify(edge.continuity)} — a delegates edge's continuity is 'fresh' or 'resume'`);
5020
- }
5021
- const delegatedTo = new Set(delegates.map((edge) => edge.to));
5022
- const roots = [...new Set(delegates.map((edge) => edge.from))].filter((id) => !delegatedTo.has(id));
5023
- if (roots.length !== 1) throw new ValidationError(`runGraph: expected exactly ONE root (a node that delegates and is never delegated to), found ${roots.length === 0 ? "none — delegates edges form a cycle with no entry" : roots.join(", ")}. P0 executes driver↔worker(s); nested driver graphs are P3.`);
5024
- const root = requireNode(roots[0], "root resolution");
5025
- for (const edge of delegates) if (edge.from !== root.id) throw new ValidationError(`runGraph: ${edgeId(edge)} delegates from a non-root node — P0 executes one driver over its workers (the 2-node cyclic case, star-generalized); deeper delegation is P3`);
5026
- const analystIds = /* @__PURE__ */ new Set();
5027
- const analystNodes = /* @__PURE__ */ new Map();
5028
- for (const edge of analyzes) {
5029
- if (edge.continuity !== void 0) throw new ValidationError(`runGraph: ${edgeId(edge)} carries continuity — analysts are spawned by the analyst machinery (every analyst run is a fresh session over settled evidence), so continuity is a delegates-edge axis only`);
5030
- if (analystIds.has(edge.analyst)) throw new ValidationError(`runGraph: two analyzes edges share analyst '${edge.analyst}' — one analyzes edge per analyst lens (traversals are ledgered by analyst id; a second edge would silently absorb the first's). Register the lens under a second id for a second edge.`);
5031
- analystIds.add(edge.analyst);
5032
- const analystNode = byId.get(edge.analyst);
5033
- const inRegistry = analysts?.kinds.some((kind) => kind.id === edge.analyst) === true;
5034
- if (analystNode !== void 0 && inRegistry) throw new ValidationError(`runGraph: ${edgeId(edge)} analyst '${edge.analyst}' is BOTH a graph node and a lens in the analysts registry — the id alone distinguishes the two analyst forms, so this is ambiguous; rename the node or register the lens under another id`);
5035
- if (analystNode !== void 0) {
5036
- if (analystNode.id === root.id) throw new ValidationError(`runGraph: ${edgeId(edge)} names the ROOT as its analyst — the root is the driver; give the analyst its own node with no delegates edge pointing at it`);
5037
- if (delegatedTo.has(analystNode.id)) throw new ValidationError(`runGraph: ${edgeId(edge)} names node '${edge.analyst}' as its analyst, but that node is a delegates target — oracle doctrine: an analyst is never delegated to. An analyst NODE is legal only with NO delegates edge pointing at it; give the analyst its own delegates-free node or pass a lens id from RunGraphOptions.analysts.`);
5038
- analystNodes.set(analystNode.id, analystNode);
5039
- } else if (!analysts) throw new ValidationError(`runGraph: ${edgeId(edge)} analyst '${edge.analyst}' is not a graph node, and no RunGraphOptions.analysts registry was provided to resolve it as a lens`);
5040
- else if (!inRegistry) throw new ValidationError(`runGraph: ${edgeId(edge)} analyst '${edge.analyst}' is neither a graph node nor in the analysts registry (known lenses: ${analysts.kinds.map((kind) => kind.id).join(", ") || "none"})`);
5041
- if (edge.over.length === 0) throw new ValidationError(`runGraph: ${edgeId(edge)} must analyze at least one node`);
5042
- for (const over of edge.over) {
5043
- requireNode(over, edgeId(edge));
5044
- if (over === root.id) throw new ValidationError(`runGraph: ${edgeId(edge)} analyzes the ROOT — analysts observe settled workers, and the root never settles as one, so this edge would silently never fire; list delegates-target nodes only`);
5045
- }
5046
- requireNode(edge.to, edgeId(edge));
5047
- }
5048
- for (const edge of analyzes) for (const over of edge.over) if (analystNodes.has(over)) throw new ValidationError(`runGraph: ${edgeId(edge)} analyzes '${over}', which is an analyst node — an analyst run settles as a finding, never as a worker, so this edge would silently never fire; analyst nodes are not analyzable`);
5049
- const workers = /* @__PURE__ */ new Map();
5050
- const delegatesByWorker = /* @__PURE__ */ new Map();
5051
- for (const edge of delegates) {
5052
- if (delegatesByWorker.has(edge.to)) throw new ValidationError(`runGraph: node '${edge.to}' is the target of two delegates edges — one delegation directive per worker node (version the directive instead of forking the edge)`);
5053
- delegatesByWorker.set(edge.to, edge);
5054
- workers.set(edge.to, requireNode(edge.to, edgeId(edge)));
5055
- }
5056
- for (const node of graph.nodes) if (node.id !== root.id && !workers.has(node.id) && !analystNodes.has(node.id)) throw new ValidationError(`runGraph: node '${node.id}' has no delegates edge to it — an unreachable node never runs`);
5057
- return {
5058
- root,
5059
- workers,
5060
- delegatesByWorker,
5061
- analyzes,
5062
- analystNodes
5063
- };
5064
- }
5065
- const byteLength = (text) => Buffer.byteLength(text, "utf8");
5066
- function stringifyPayload(payload) {
5067
- if (typeof payload === "string") return payload;
5068
- try {
5069
- return JSON.stringify(payload) ?? String(payload);
5070
- } catch {
5071
- return String(payload);
5072
- }
5073
- }
5074
- /**
5075
- * Execute an {@link AgentGraph}. The root node becomes the supervisor (`supervise()` — the
5076
- * execution core), each worker node is spawnable BY NODE ID (`spawn_agent` with
5077
- * `profile: { name: '<node id>' }`; the node's canonical profile is pinned by the graph), each
5078
- * delegates directive is appended to the worker profile's `prompt.instructions` per traversal,
5079
- * and each analyzes edge becomes an analyst-on-settle route with a real DESTINATION. Every
5080
- * traversal is ledgered and journaled.
5081
- */
5082
- function runGraph(graph, opts) {
5083
- const registry = opts.registry ?? kernelPromptRegistry();
5084
- const { root, workers, delegatesByWorker, analyzes, analystNodes } = validateGraph(graph, registry, opts.analysts);
5085
- if (!opts.backend && !opts.makeWorkerAgent) throw new ValidationError("runGraph: provide opts.backend (where nodes run) or opts.makeWorkerAgent");
5086
- const journal = opts.journal ?? new InMemorySpawnJournal();
5087
- const blobs = opts.blobs ?? new InMemoryResultBlobStore();
5088
- const runId = opts.runId ?? `graph-${canonicalCandidateDigest(graph.nodes.map((n) => n.id)).slice(7, 19)}`;
5089
- const now = opts.now ?? Date.now;
5090
- const ledger = [];
5091
- const journaled = /* @__PURE__ */ new Set();
5092
- const traversalCounts = /* @__PURE__ */ new Map();
5093
- const exhausted = /* @__PURE__ */ new Set();
5094
- const exhaustedDelegates = /* @__PURE__ */ new Set();
5095
- const journalWrites = [];
5096
- let ledgerSeq = 0;
5097
- const appendJournal = (entry, nodeIdForEvent) => {
5098
- if (journaled.has(entry)) return Promise.resolve();
5099
- journaled.add(entry);
5100
- const write = journal.appendEvent(runId, {
5101
- kind: "edge",
5102
- id: nodeIdForEvent,
5103
- edge: {
5104
- kind: entry.kind,
5105
- from: entry.from,
5106
- to: entry.to,
5107
- directive: entry.directive
5108
- },
5109
- traversal: entry.traversal,
5110
- outcome: entry.outcome,
5111
- continuity: entry.continuity,
5112
- bytes: entry.bytes,
5113
- ...entry.reason !== void 0 ? { reason: entry.reason } : {},
5114
- seq: ledgerSeq++,
5115
- at: new Date(now()).toISOString()
5116
- });
5117
- journalWrites.push(write);
5118
- return write;
5119
- };
5120
- const record = (entry, journalNow) => {
5121
- const count = (traversalCounts.get(entry.edge) ?? 0) + 1;
5122
- traversalCounts.set(entry.edge, count);
5123
- const row = {
5124
- ...entry,
5125
- traversal: count
5126
- };
5127
- ledger.push(row);
5128
- if (journalNow) appendJournal(row, row.workerId ?? `graph:${row.to}`);
5129
- return row;
5130
- };
5131
- const makeLeaf = opts.makeWorkerAgent ?? workerFromBackend(opts.backend, graph.deliverable);
5132
- const nodeByWorkerId = /* @__PURE__ */ new Map();
5133
- const pendingByAssignment = /* @__PURE__ */ new Map();
5134
- const graphWorker = (authoredProfile, spawnContext) => {
5135
- const requested = typeof authoredProfile?.name === "string" ? authoredProfile.name : void 0;
5136
- if (spawnContext?.analyst !== void 0) {
5137
- const analystNode = analystNodes.get(spawnContext.analyst);
5138
- if (!analystNode || requested !== analystNode.id) throw new ValidationError(`runGraph: analyst run for ${JSON.stringify(spawnContext.analyst)} does not name an analyst node of this graph (analyst nodes: ${[...analystNodes.keys()].join(", ") || "none"})`);
5139
- return makeLeaf(analystNode.profile, spawnContext);
5140
- }
5141
- const node = requested !== void 0 ? workers.get(requested) : void 0;
5142
- if (!node) throw new ValidationError(`runGraph: spawn_agent named profile ${JSON.stringify(requested)} which is not a worker node of this graph (nodes: ${[...workers.keys()].join(", ")}). Spawn by node id: profile.name selects the node; the node profile itself is pinned by the graph.`);
5143
- const edge = delegatesByWorker.get(node.id);
5144
- const id = edgeId(edge);
5145
- const cap = edge.maxTraversals ?? 32;
5146
- const used = traversalCounts.get(id) ?? 0;
5147
- const spawnContinuity = spawnContext?.continuity ?? "fresh";
5148
- if (used >= cap) {
5149
- exhausted.add(id);
5150
- exhaustedDelegates.add(id);
5151
- record({
5152
- edge: id,
5153
- kind: "delegates",
5154
- from: edge.from,
5155
- to: edge.to,
5156
- directive: formatPromptHandle(edge.directive),
5157
- outcome: "unpropagated",
5158
- continuity: spawnContinuity,
5159
- bytes: 0,
5160
- reason: `traversal-cap-exhausted (max ${cap})`
5161
- }, true);
5162
- throw new ValidationError(`runGraph: delegates edge ${id} exhausted its traversal cap (${cap}) — the cyclic-graph backstop refused this spawn`);
5163
- }
5164
- const directiveText = registry.resolve(edge.directive).text;
5165
- const taskText = stringifyPayload(spawnContext?.task);
5166
- const bytes = byteLength(directiveText) + byteLength(taskText);
5167
- const row = record({
5168
- edge: id,
5169
- kind: "delegates",
5170
- from: edge.from,
5171
- to: edge.to,
5172
- directive: formatPromptHandle(edge.directive),
5173
- outcome: bytes === 0 ? "empty" : "delivered",
5174
- continuity: spawnContinuity,
5175
- bytes,
5176
- ...bytes === 0 ? { reason: "no directive text and no task payload" } : {}
5177
- }, false);
5178
- if (spawnContext?.assignmentId !== void 0) pendingByAssignment.set(spawnContext.assignmentId, row);
5179
- else appendJournal(row, `graph:${row.to}`);
5180
- const pinned = directiveText.length === 0 ? node.profile : {
5181
- ...node.profile,
5182
- prompt: {
5183
- ...node.profile.prompt ?? {},
5184
- instructions: [...node.profile.prompt?.instructions ?? [], directiveText]
5185
- }
5186
- };
5187
- return makeLeaf(pinned, spawnContext);
5188
- };
5189
- const routes = analyzes.map((edge) => {
5190
- const analystNode = analystNodes.get(edge.analyst);
5191
- if (analystNode) return {
5192
- kind: edge.analyst,
5193
- over: edge.over,
5194
- agent: analystNode.profile,
5195
- directive: registry.resolve(edge.directive).text,
5196
- ...edge.to === root.id ? {} : { to: edge.to }
5197
- };
5198
- return edge.to === root.id ? {
5199
- kind: edge.analyst,
5200
- over: edge.over
5201
- } : {
5202
- kind: edge.analyst,
5203
- over: edge.over,
5204
- to: edge.to,
5205
- directive: registry.resolve(edge.directive).text
5206
- };
5207
- });
5208
- const driverAnalyzesBriefs = analyzes.filter((edge) => edge.to === root.id).map((edge) => analystNodes.has(edge.analyst) ? `Findings from analyst '${edge.analyst}' (a tool-equipped analyst agent node, over: ${edge.over.join(", ")}) will arrive as finding events.` : `Findings from analyst '${edge.analyst}' (over: ${edge.over.join(", ")}) will arrive as finding events.\n${registry.resolve(edge.directive).text}`);
5209
- const continuityByProfile = {};
5210
- for (const [nodeId, edge] of delegatesByWorker) if (edge.continuity !== void 0) continuityByProfile[nodeId] = edge.continuity;
5211
- const graphBrief = [
5212
- "AGENT GRAPH: you are the driver node of a fixed topology. You may spawn ONLY these worker",
5213
- "nodes, by EXACT name (spawn_agent with profile: { name: '<node id>' }; the node's full",
5214
- "profile is pinned by the graph — any other profile fields you author are ignored):",
5215
- ...[...workers.values()].map((node) => {
5216
- const edge = delegatesByWorker.get(node.id);
5217
- const cap = edge.maxTraversals ?? 32;
5218
- const description = typeof node.profile.description === "string" && node.profile.description.length > 0 ? ` — ${node.profile.description}` : "";
5219
- const continuityNote = edge.continuity === "resume" ? "; continuity: resume — each spawn after the first re-attaches to this node's latest settled session (spawn again to continue it; steer while it is live)" : "";
5220
- return `- '${node.id}'${description} (delegation cap: ${cap} traversals${continuityNote})`;
5221
- }),
5222
- ...driverAnalyzesBriefs.length > 0 ? ["", ...driverAnalyzesBriefs] : []
5223
- ].join("\n");
5224
- const rootProfile = {
5225
- ...root.profile,
5226
- prompt: {
5227
- ...root.profile.prompt ?? {},
5228
- instructions: [...root.profile.prompt?.instructions ?? [], graphBrief]
5229
- }
5230
- };
5231
- const strippedByDigest = /* @__PURE__ */ new Map();
5232
- const authorizeMessage = opts.authorizeMessage ? (input) => {
5233
- const decision = opts.authorizeMessage(input);
5234
- if (decision.instruction !== input.instruction) strippedByDigest.set(canonicalCandidateDigest(decision.instruction), { composedBytes: byteLength(input.instruction) });
5235
- return decision;
5236
- } : void 0;
5237
- const routedAnalyzesByAnalyst = /* @__PURE__ */ new Map();
5238
- const driverAnalyzesByAnalyst = /* @__PURE__ */ new Map();
5239
- for (const edge of analyzes) (edge.to === root.id ? driverAnalyzesByAnalyst : routedAnalyzesByAnalyst).set(edge.analyst, edge);
5240
- const analyzesCapReached = (edge) => {
5241
- const cap = edge.maxTraversals ?? 32;
5242
- if ((traversalCounts.get(edgeId(edge)) ?? 0) < cap) return false;
5243
- exhausted.add(edgeId(edge));
5244
- return true;
5245
- };
5246
- const ledgerAnalyzes = (edge, outcome, bytes, reason, workerId) => {
5247
- const capped = analyzesCapReached(edge);
5248
- record({
5249
- edge: edgeId(edge),
5250
- kind: "analyzes",
5251
- from: edge.over.join("+"),
5252
- to: edge.to,
5253
- directive: formatPromptHandle(edge.directive),
5254
- outcome: capped ? "unpropagated" : outcome,
5255
- continuity: "steer",
5256
- bytes,
5257
- ...capped ? { reason: `traversal-cap-exhausted (max ${edge.maxTraversals ?? 32})` } : reason !== void 0 ? { reason } : {},
5258
- ...workerId !== void 0 ? { workerId } : {}
5259
- }, true);
5260
- };
5261
- const onCoordinationEvent = async (_context, _eventId, recordEnvelope) => {
5262
- const event = recordEnvelope.event;
5263
- if (event.type === "finding") {
5264
- const edge = driverAnalyzesByAnalyst.get(event.finding.analyst);
5265
- if (!edge) return;
5266
- const sourceNode = nodeByWorkerId.get(event.finding.fromWorker);
5267
- if (sourceNode === void 0 || !edge.over.includes(sourceNode)) return;
5268
- const findingsText = event.finding.findings === void 0 ? "" : stringifyPayload(event.finding.findings);
5269
- const directiveBytes = byteLength(registry.resolve(edge.directive).text);
5270
- const empty = findingsText.length === 0;
5271
- ledgerAnalyzes(edge, empty ? "empty" : "delivered", directiveBytes + byteLength(findingsText), empty ? "analyst returned no findings" : void 0, event.finding.fromWorker);
5272
- return;
5273
- }
5274
- if (event.type === "steer") {
5275
- const down = event.down;
5276
- if (event.analyst !== void 0) {
5277
- const edge = routedAnalyzesByAnalyst.get(event.analyst);
5278
- if (!edge) return;
5279
- ledgerAnalyzes(edge, down.delivered ? "delivered" : "unpropagated", byteLength(down.instruction), down.delivered ? void 0 : down.outcome, down.toWorker);
5280
- return;
5281
- }
5282
- const nodeId = nodeByWorkerId.get(down.toWorker);
5283
- if (nodeId === void 0) return;
5284
- const edge = delegatesByWorker.get(nodeId);
5285
- if (!edge) return;
5286
- const stripped = strippedByDigest.get(down.instructionDigest);
5287
- record({
5288
- edge: edgeId(edge),
5289
- kind: "delegates",
5290
- from: edge.from,
5291
- to: edge.to,
5292
- directive: formatPromptHandle(edge.directive),
5293
- outcome: !down.delivered ? "unpropagated" : stripped ? "stripped" : "delivered",
5294
- continuity: "steer",
5295
- bytes: byteLength(down.instruction),
5296
- ...!down.delivered ? { reason: down.outcome } : stripped ? { reason: `authorization narrowed ${stripped.composedBytes} composed bytes` } : {},
5297
- workerId: down.toWorker
5298
- }, true);
5299
- }
5300
- };
5301
- const hooks = composeRuntimeHooks({ onEvent: (event) => {
5302
- if (event.target !== "agent.spawn" || event.phase !== "after") return;
5303
- const payload = event.payload;
5304
- if (typeof payload?.childId !== "string" || typeof payload.assignmentId !== "string") return;
5305
- const pending = pendingByAssignment.get(payload.assignmentId);
5306
- if (!pending) return;
5307
- pendingByAssignment.delete(payload.assignmentId);
5308
- const bound = {
5309
- ...pending,
5310
- workerId: payload.childId
5311
- };
5312
- ledger[ledger.indexOf(pending)] = bound;
5313
- nodeByWorkerId.set(payload.childId, bound.to);
5314
- return appendJournal(bound, payload.childId);
5315
- } }, opts.hooks);
5316
- const start = async () => {
5317
- const result = await supervise(rootProfile, graphTask(graph, root), {
5318
- budget: graph.budget,
5319
- deliverable: graph.deliverable,
5320
- makeWorkerAgent: graphWorker,
5321
- journal,
5322
- blobs,
5323
- runId,
5324
- hooks,
5325
- onCoordinationEvent,
5326
- ...routes.length > 0 ? {
5327
- analyzeOnSettle: routes,
5328
- ...opts.analysts ? { analysts: opts.analysts } : {}
5329
- } : {},
5330
- ...Object.keys(continuityByProfile).length > 0 ? { continuityByProfile } : {},
5331
- ...opts.watchWorkers ? { watchWorkers: opts.watchWorkers } : {},
5332
- ...opts.router ? { router: opts.router } : {},
5333
- ...opts.brain ? { brain: opts.brain } : {},
5334
- ...authorizeMessage ? { authorizeMessage } : {},
5335
- ...opts.perWorker ? { perWorker: opts.perWorker } : {},
5336
- ...opts.maxTurns !== void 0 ? { maxTurns: opts.maxTurns } : {},
5337
- ...opts.maxLiveWorkers !== void 0 ? { maxLiveWorkers: opts.maxLiveWorkers } : {},
5338
- ...opts.signal ? { signal: opts.signal } : {},
5339
- ...opts.now ? { now: opts.now } : {},
5340
- ...opts.otel ? { otel: opts.otel } : {},
5341
- ...opts.stallAfterMs !== void 0 ? { stallAfterMs: opts.stallAfterMs } : {},
5342
- ...opts.allowedModels ? { allowedModels: opts.allowedModels } : {}
5343
- });
5344
- for (const pending of pendingByAssignment.values()) {
5345
- const refused = {
5346
- ...pending,
5347
- outcome: "unpropagated",
5348
- bytes: 0,
5349
- reason: `no-live-worker-bound (spawn refused after the factory, or a keyed re-spawn deduplicated to a completed result; ${pending.bytes} composed bytes never crossed)`
5350
- };
5351
- ledger[ledger.indexOf(pending)] = refused;
5352
- await appendJournal(refused, `graph:${refused.to}`);
5353
- }
5354
- pendingByAssignment.clear();
5355
- await Promise.all(journalWrites);
5356
- const exhaustedEdges = Object.freeze([...exhausted]);
5357
- const frozenLedger = Object.freeze(ledger.map((row) => Object.freeze({ ...row })));
5358
- const lifecycleEnded = result.kind === "no-winner" && (result.reason === "aborted" || result.reason === "budget-exhausted");
5359
- if (result.kind !== "winner" && !lifecycleEnded && exhaustedDelegates.size > 0) throw new GraphEdgeCapError(Object.freeze([...exhaustedDelegates]), frozenLedger, result);
5360
- return {
5361
- result,
5362
- ledger: frozenLedger,
5363
- exhaustedEdges,
5364
- runId
5365
- };
5366
- };
5367
- return start();
5368
- }
5369
- /** The root task: the graph's own framing. The deliverable (mandatory) is the termination; the
5370
- * task names what the topology exists to produce. */
5371
- function graphTask(graph, root) {
5372
- return graph.deliverable.describe ?? `Deliver the graph's deliverable by driving your worker nodes (root: '${root.id}').`;
5373
- }
5374
- //#endregion
5375
4492
  //#region src/runtime/supervise/patch-checks.ts
5376
4493
  const DEFAULT_MAX_DIFF_LINES = 400;
5377
4494
  /**
@@ -5810,7 +4927,6 @@ function worktreeFanout(options) {
5810
4927
  return gateOnDeliverable(createWorktreeCliExecutor({
5811
4928
  repoRoot: options.repoRoot,
5812
4929
  profile: item.profile,
5813
- harness: item.harness,
5814
4930
  taskPrompt: options.taskPrompt,
5815
4931
  executionAttemptId: ctx.node.attemptId,
5816
4932
  ...item.budgetExempt !== void 0 ? { budgetExempt: item.budgetExempt } : {},
@@ -5983,7 +5099,7 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
5983
5099
  const guidance = typeof brief === "string" ? brief.trim() : brief ? JSON.stringify(brief) : "";
5984
5100
  const attemptTask = guidance ? {
5985
5101
  ...task,
5986
- systemPrompt: `${task.systemPrompt ?? ""}\n\n— Supervisor guidance for THIS attempt (incorporate it; do not just repeat a prior approach) —\n${guidance}`
5102
+ userPrompt: `${task.userPrompt}\n\n— Supervisor guidance for THIS attempt (incorporate it; do not just repeat a prior approach) —\n${guidance}`
5987
5103
  } : task;
5988
5104
  const r = await runAgentic({
5989
5105
  surface: traced.surface,
@@ -5992,8 +5108,8 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
5992
5108
  budget: worker.budget ?? 1,
5993
5109
  routerBaseUrl: worker.routerBaseUrl,
5994
5110
  routerKey: worker.routerKey,
5995
- model: worker.model,
5996
- ...worker.maxTokens !== void 0 ? { maxTokens: worker.maxTokens } : {},
5111
+ workerProfile: worker.profile,
5112
+ ...worker.analystProfile ? { analystProfile: worker.analystProfile } : {},
5997
5113
  ...worker.innerTurns !== void 0 ? { innerTurns: worker.innerTurns } : {}
5998
5114
  });
5999
5115
  const out = {
@@ -6006,7 +5122,9 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
6006
5122
  const spent = {
6007
5123
  iterations: r.completions,
6008
5124
  tokens: r.tokens,
5125
+ ...r.tokensKnown ? {} : { tokensKnown: false },
6009
5126
  usd: r.usd,
5127
+ ...r.usdKnown ? {} : { usdKnown: false },
6010
5128
  ms: r.ms
6011
5129
  };
6012
5130
  artifact = {
@@ -6036,13 +5154,13 @@ async function superviseSurface(profile, task, opts) {
6036
5154
  const innerTurns = opts.worker.innerTurns ?? 6;
6037
5155
  const router = opts.router ?? {
6038
5156
  routerBaseUrl: opts.worker.routerBaseUrl,
6039
- routerKey: opts.worker.routerKey,
6040
- model: opts.worker.model
5157
+ routerKey: opts.worker.routerKey
6041
5158
  };
6042
5159
  const budget = opts.budget ?? {
6043
5160
  maxIterations: (innerTurns + 2) * 5 + 16,
6044
- maxTokens: (opts.worker.maxTokens ?? 4e3) * 8
5161
+ maxTokens: 1e9
6045
5162
  };
5163
+ const workerMaxTokens = profileMaxTokens(opts.worker.profile) ?? Math.max(1, Math.floor(budget.maxTokens / 8));
6046
5164
  const makeWorkerAgent = (rawProfile) => {
6047
5165
  const p = rawProfile ?? {};
6048
5166
  return {
@@ -6067,7 +5185,7 @@ async function superviseSurface(profile, task, opts) {
6067
5185
  maxLiveWorkers: opts.maxLiveWorkers ?? 1,
6068
5186
  perWorker: {
6069
5187
  maxIterations: innerTurns + 2,
6070
- maxTokens: opts.worker.maxTokens ?? 4e3
5188
+ maxTokens: workerMaxTokens
6071
5189
  },
6072
5190
  router,
6073
5191
  ...analysts ? {
@@ -6087,6 +5205,12 @@ async function superviseSurface(profile, task, opts) {
6087
5205
  completions: sp.iterations
6088
5206
  };
6089
5207
  }
5208
+ function profileMaxTokens(profile) {
5209
+ const value = profile.model?.metadata?.maxTokens;
5210
+ if (value === void 0) return void 0;
5211
+ if (!Number.isSafeInteger(value) || value < 1) throw new Error("superviseSurface: AgentProfile.model.metadata.maxTokens must be a positive safe integer");
5212
+ return value;
5213
+ }
6090
5214
  //#endregion
6091
5215
  //#region src/runtime/verifier-environment.ts
6092
5216
  const submitTool = {
@@ -6501,6 +5625,6 @@ function tail(s) {
6501
5625
  return s.slice(-400);
6502
5626
  }
6503
5627
  //#endregion
6504
- export { renderCorpusToInstructions as $, collectAgentTurn as A, auditIntent as At, openSandboxRun as B, resolveSecretEnv as Bt, GraphEdgeCapError as C, stopSentinel as Ct, chatTransportExecutor as D, renderLeaderboardMarkdown as Dt, chatCompletionsTransport as E, renderLeaderboardHtml as Et, selectChampion as F, createMcpEnvironment as Ft, trajectoryReport as G, runBenchmark as H, assertStrategyContract as I, sanitizeMcpToolSchema as It, builtinShapes as J, definePersona as K, authorStrategy as L, envKeyProvider as Lt, discriminatingMeans as M, McpSpawnFault as Mt, pickChampion as N, connectStdioMcp as Nt, chatWorkerSeam as O, renderLeaderboardSvg as Ot, runStrategyEvolution as P, materializeLocalMcp as Pt, InMemoryCorpus as Q, strategyAuthorContract as R, mcpSecretEnvMetadataKey as Rt, runCoderChecks as S, sentinelCompletion as St, runGraph as T, pairwiseSignificance as Tt, promotionGate as U, printBenchmarkReport as V, secretEnvOfMcpServer as Vt, equalKOnCost as W, registerShape as X, createShapeRegistry as Y, FileCorpus as Z, settledWorkerOut as _, inlineSandboxClient as _t, localShell as a, selectValidWinner as at, analyzeTrace as b, completionAuthorizes as bt, createVerifierEnvironment as c, assertTraceDerivedFindings as ct, worktreeFanout as d, registryScopeAnalyst as dt, fanout as et, EVIDENCE_MAX_CHARS as f, inProcessSandboxClient as ft, composeWorkerEvidence as g, localSandboxClient as gt, closingWorkerNote as h, resolveSandboxClient as ht, jjWorkspace as i, pipeline as it, streamAgentTurn as j, defaultAuditorInstruction as jt, createChatSessionStore as k, renderPairwiseMarkdown as kt, failuresAnalyst as l, buildSteerContext as lt, VERIFY_TAIL_CHARS as m, defineLeaderboard as mt, makeFinding$1 as n, loopUntil as nt, runInWorkspace as o, verify as ot, NOTE_MAX_CHARS as p, harvestCorpus as pt, runPersonified as q, gitWorkspace as r, panel as rt, createWaterfallCollector as s, widen as st, computeFindingId$1 as t, flatWidenGate as tt, superviseSurface as u, createScopeAnalyst as ut, copyUntrackedIntoClone as v, loopCampaignDispatch as vt, defaultEdgeTraversalCap as w, leaderboard as wt, patchDelivered as x, deterministicCompletion as xt, withUntrackedArtifacts as y, loopDispatch as yt, SandboxRunAbortError as z, resolveMcpServerLaunch as zt };
5628
+ export { pipeline as $, assertStrategyContract as A, createMcpEnvironment as At, trajectoryReport as B, chatTransportExecutor as C, renderLeaderboardSvg as Ct, pickChampion as D, McpSpawnFault as Dt, discriminatingMeans as E, defaultAuditorInstruction as Et, openSandboxRun as F, resolveSecretEnv as Ft, registerShape as G, runPersonified as H, printBenchmarkReport as I, secretEnvOfMcpServer as It, renderCorpusToInstructions as J, FileCorpus as K, runBenchmark as L, strategyAuthorContract as M, envKeyProvider as Mt, strategyAuthorSystemPrompt as N, mcpSecretEnvMetadataKey as Nt, runStrategyEvolution as O, connectStdioMcp as Ot, SandboxRunAbortError as P, resolveMcpServerLaunch as Pt, panel as Q, promotionGate as R, runCoderChecks as S, renderLeaderboardMarkdown as St, createChatSessionStore as T, auditIntent as Tt, builtinShapes as U, definePersona as V, createShapeRegistry as W, flatWidenGate as X, fanout as Y, loopUntil as Z, settledWorkerOut as _, sentinelCompletion as _t, localShell as a, createScopeAnalyst as at, analyzeTrace as b, pairwiseSignificance as bt, createVerifierEnvironment as c, harvestCorpus as ct, worktreeFanout as d, localSandboxClient as dt, selectValidWinner as et, EVIDENCE_MAX_CHARS as f, inlineSandboxClient as ft, composeWorkerEvidence as g, deterministicCompletion as gt, closingWorkerNote as h, completionAuthorizes as ht, jjWorkspace as i, buildSteerContext as it, authorStrategy as j, sanitizeMcpToolSchema as jt, selectChampion as k, materializeLocalMcp as kt, failuresAnalyst as l, defineLeaderboard as lt, VERIFY_TAIL_CHARS as m, loopDispatch as mt, makeFinding$1 as n, widen as nt, runInWorkspace as o, registryScopeAnalyst as ot, NOTE_MAX_CHARS as p, loopCampaignDispatch as pt, InMemoryCorpus as q, gitWorkspace as r, assertTraceDerivedFindings as rt, createWaterfallCollector as s, inProcessSandboxClient as st, computeFindingId$1 as t, verify as tt, superviseSurface as u, resolveSandboxClient as ut, copyUntrackedIntoClone as v, stopSentinel as vt, chatWorkerSeam as w, renderPairwiseMarkdown as wt, patchDelivered as x, renderLeaderboardHtml as xt, withUntrackedArtifacts as y, leaderboard as yt, equalKOnCost as z };
6505
5629
 
6506
- //# sourceMappingURL=runtime-5uDVVfER.js.map
5630
+ //# sourceMappingURL=runtime-hiAABiTk.js.map