@tangle-network/agent-runtime 0.126.0 → 0.131.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (146) hide show
  1. package/README.md +70 -20
  2. package/dist/{activation-DhWJ3p8N.js → activation-BCzMOTaV.js} +3 -3
  3. package/dist/{activation-DhWJ3p8N.js.map → activation-BCzMOTaV.js.map} +1 -1
  4. package/dist/agent.d.ts +2 -3
  5. package/dist/agent.js +4 -5
  6. package/dist/agent.js.map +1 -1
  7. package/dist/{analyst-loop-DvSciOfB.js → analyst-loop-BE8cDs5Q.js} +2 -2
  8. package/dist/{analyst-loop-DvSciOfB.js.map → analyst-loop-BE8cDs5Q.js.map} +1 -1
  9. package/dist/analyst-loop.d.ts +1 -1
  10. package/dist/analyst-loop.js +1 -1
  11. package/dist/authoring-CvHwo1oW.js +163 -0
  12. package/dist/authoring-CvHwo1oW.js.map +1 -0
  13. package/dist/candidate-execution/index.js +4 -4
  14. package/dist/{candidate-execution-BFpq-Xi6.js → candidate-execution-BNxKr-Bu.js} +5 -5
  15. package/dist/{candidate-execution-BFpq-Xi6.js.map → candidate-execution-BNxKr-Bu.js.map} +1 -1
  16. package/dist/{conversation-BpLQZGPH.js → conversation-DNtxaJ1Z.js} +113 -31
  17. package/dist/conversation-DNtxaJ1Z.js.map +1 -0
  18. package/dist/conversation.d.ts +2 -2
  19. package/dist/conversation.js +2 -2
  20. package/dist/environment-provider-CxvSd1W6.d.ts +86 -0
  21. package/dist/{environment-provider-DqFS6FSZ.js → environment-provider-Dyg8DtLK.js} +8 -4
  22. package/dist/environment-provider-Dyg8DtLK.js.map +1 -0
  23. package/dist/environment-provider.d.ts +1 -1
  24. package/dist/environment-provider.js +1 -1
  25. package/dist/graph-BJTxGOFB.js +471 -0
  26. package/dist/graph-BJTxGOFB.js.map +1 -0
  27. package/dist/{improvement-cycle-IJgCbKWQ.js → improvement-cycle-Csp38cWg.js} +133 -224
  28. package/dist/improvement-cycle-Csp38cWg.js.map +1 -0
  29. package/dist/{index-EdjCQBV9.d.ts → index-CoO7atyo.d.ts} +640 -1184
  30. package/dist/{index-DIV33AF5.d.ts → index-DwGtu9nc.d.ts} +7 -9
  31. package/dist/{index-Efjb3nrQ.d.ts → index-qYHpsmG2.d.ts} +36 -18
  32. package/dist/index.d.ts +353 -11
  33. package/dist/index.js +111 -354
  34. package/dist/index.js.map +1 -1
  35. package/dist/intelligence.d.ts +6 -6
  36. package/dist/intelligence.js +9 -8
  37. package/dist/intelligence.js.map +1 -1
  38. package/dist/kernel.d.ts +7 -5
  39. package/dist/kernel.js +13 -9
  40. package/dist/{knowledge-EnuEqm_Y.js → knowledge-ce0_uKCl.js} +19 -17
  41. package/dist/knowledge-ce0_uKCl.js.map +1 -0
  42. package/dist/knowledge.d.ts +1 -1
  43. package/dist/knowledge.js +1 -1
  44. package/dist/{loop-runner-bin-Bo29_fiD.d.ts → loop-runner-bin-BwgZ8m_7.d.ts} +6 -3
  45. package/dist/{loop-runner-bin-qwT_4F5I.js → loop-runner-bin-DSbuDDqM.js} +5 -27
  46. package/dist/loop-runner-bin-DSbuDDqM.js.map +1 -0
  47. package/dist/loop-runner-bin.d.ts +1 -1
  48. package/dist/loop-runner-bin.js +1 -1
  49. package/dist/materialization-COJ1UYQ-.js +272 -0
  50. package/dist/materialization-COJ1UYQ-.js.map +1 -0
  51. package/dist/mcp/bin.js +39 -47
  52. package/dist/mcp/bin.js.map +1 -1
  53. package/dist/mcp/index.d.ts +24 -30
  54. package/dist/mcp/index.js +66 -83
  55. package/dist/mcp/index.js.map +1 -1
  56. package/dist/mcp/memory-bin.js +1 -1
  57. package/dist/{memory-server-DL6cE2Ag.js → memory-server-5HEJH672.js} +2 -2
  58. package/dist/{memory-server-DL6cE2Ag.js.map → memory-server-5HEJH672.js.map} +1 -1
  59. package/dist/model-policy-CqziaqS1.js +232 -0
  60. package/dist/model-policy-CqziaqS1.js.map +1 -0
  61. package/dist/{openai-tools-B68JaOCx.d.ts → openai-tools-D3oMyVGl.d.ts} +2 -2
  62. package/dist/{openai-tools-Bp1KSkP6.js → openai-tools-ru75mLjq.js} +2 -2
  63. package/dist/openai-tools-ru75mLjq.js.map +1 -0
  64. package/dist/{prepare--8EvLqCr.js → prepare-DYWjVcPx.js} +169 -153
  65. package/dist/prepare-DYWjVcPx.js.map +1 -0
  66. package/dist/primeintellect/index.d.ts +7 -6
  67. package/dist/primeintellect/index.js +9 -11
  68. package/dist/primeintellect/index.js.map +1 -1
  69. package/dist/profiles.d.ts +21 -174
  70. package/dist/profiles.js +67 -276
  71. package/dist/profiles.js.map +1 -1
  72. package/dist/{protected-model-port-CXVfOUu_.js → protected-model-port-48ALLGxT.js} +2 -2
  73. package/dist/{protected-model-port-CXVfOUu_.js.map → protected-model-port-48ALLGxT.js.map} +1 -1
  74. package/dist/{redact-PQzmE1Jn.d.ts → redact-9_Gf-8m3.d.ts} +58 -69
  75. package/dist/{researcher-CoVqNhfI.js → researcher-Skz5-Uc8.js} +50 -26
  76. package/dist/researcher-Skz5-Uc8.js.map +1 -0
  77. package/dist/{run-layout-C2FGmZ3v.js → run-layout-CeJsEAom.js} +2 -2
  78. package/dist/{run-layout-C2FGmZ3v.js.map → run-layout-CeJsEAom.js.map} +1 -1
  79. package/dist/runtime-D-QfLbSd.d.ts +893 -0
  80. package/dist/{runtime-BzXz7OjS.js → runtime-hiAABiTk.js} +329 -854
  81. package/dist/runtime-hiAABiTk.js.map +1 -0
  82. package/dist/{sandbox-events-Yhd1GYWl.js → sandbox-events-CRDwc5WN.js} +42 -5
  83. package/dist/sandbox-events-CRDwc5WN.js.map +1 -0
  84. package/dist/snapshot-CXiiuHhL.js +21 -0
  85. package/dist/snapshot-CXiiuHhL.js.map +1 -0
  86. package/dist/{spawn-journal-DsZKDqeh.js → spawn-journal-saHQzqYi.js} +15 -21
  87. package/dist/spawn-journal-saHQzqYi.js.map +1 -0
  88. package/dist/stream-agent-turn-C852AgMT.d.ts +128 -0
  89. package/dist/stream-agent-turn-rYgaOLO0.js +910 -0
  90. package/dist/stream-agent-turn-rYgaOLO0.js.map +1 -0
  91. package/dist/{structural-rollout-zY0oqQzO.js → structural-rollout-3uxVGcg2.js} +459 -235
  92. package/dist/structural-rollout-3uxVGcg2.js.map +1 -0
  93. package/dist/{supervise-Ds8FtyI9.js → supervise-iPN27pO0.js} +932 -4771
  94. package/dist/supervise-iPN27pO0.js.map +1 -0
  95. package/dist/{supervisor-DpjO0Gmy.js → supervisor-CV6Jh28D.js} +7439 -2897
  96. package/dist/supervisor-CV6Jh28D.js.map +1 -0
  97. package/dist/testing.d.ts +3 -1
  98. package/dist/testing.js +271 -221
  99. package/dist/testing.js.map +1 -1
  100. package/dist/{top-app-5unxqovu.js → top-app-G4M18b6i.js} +3 -3
  101. package/dist/{top-app-5unxqovu.js.map → top-app-G4M18b6i.js.map} +1 -1
  102. package/dist/tui/bin.js +1 -1
  103. package/dist/tui/index.js +1 -1
  104. package/dist/{environment-provider-PM9PeW_J.d.ts → types-C6Q-J0Dt.d.ts} +57 -114
  105. package/dist/{types-DnNGJ5Gz.d.ts → types-ebIY0dMG.d.ts} +556 -30
  106. package/dist/{util-MVgdwuIS.js → util-Bw6srryQ.js} +3 -2
  107. package/dist/{util-MVgdwuIS.js.map → util-Bw6srryQ.js.map} +1 -1
  108. package/dist/{workspace-archive-BQxvkypI.js → workspace-archive-aOfJ47ms.js} +3 -3
  109. package/dist/{workspace-archive-BQxvkypI.js.map → workspace-archive-aOfJ47ms.js.map} +1 -1
  110. package/package.json +12 -15
  111. package/skills/agent-graphs/IMPROVE.md +58 -0
  112. package/skills/agent-graphs/SKILL.md +139 -0
  113. package/skills/agent-graphs/cases/artifact-mission-release-notes.json +10 -0
  114. package/skills/agent-graphs/cases/audited-single-writer.json +9 -0
  115. package/skills/agent-graphs/cases/cap-as-stop-mistake.json +8 -0
  116. package/skills/agent-graphs/cases/mission-in-deliverable.json +8 -0
  117. package/skills/agent-graphs/cases/review-pipeline.json +13 -0
  118. package/skills/agent-graphs/cases/runtime-discovered-fanout.json +8 -0
  119. package/skills/agent-graphs/cases/single-agent-suffices.json +7 -0
  120. package/skills/agent-graphs/cases/steer-heavy-drafting.json +9 -0
  121. package/skills/agent-graphs/cases/unmeasured-harness.json +7 -0
  122. package/skills/agent-graphs/generations/gen1-baseline.json +248 -0
  123. package/skills/agent-graphs/generations/gen2.json +375 -0
  124. package/skills/agent-graphs/generations/gen3.json +702 -0
  125. package/skills/build-with-agent-runtime/SKILL.md +1 -0
  126. package/dist/backends-CiOCyRHb.js +0 -743
  127. package/dist/backends-CiOCyRHb.js.map +0 -1
  128. package/dist/conversation-BpLQZGPH.js.map +0 -1
  129. package/dist/environment-provider-DqFS6FSZ.js.map +0 -1
  130. package/dist/improvement-cycle-IJgCbKWQ.js.map +0 -1
  131. package/dist/index-D_M4d1_B.d.ts +0 -545
  132. package/dist/knowledge-EnuEqm_Y.js.map +0 -1
  133. package/dist/local-harness-BIajef4A.d.ts +0 -465
  134. package/dist/loop-runner-bin-qwT_4F5I.js.map +0 -1
  135. package/dist/model-resolution-Btd9iIKV.js +0 -98
  136. package/dist/model-resolution-Btd9iIKV.js.map +0 -1
  137. package/dist/openai-tools-Bp1KSkP6.js.map +0 -1
  138. package/dist/prepare--8EvLqCr.js.map +0 -1
  139. package/dist/researcher-CoVqNhfI.js.map +0 -1
  140. package/dist/runtime-BzXz7OjS.js.map +0 -1
  141. package/dist/sandbox-events-Yhd1GYWl.js.map +0 -1
  142. package/dist/spawn-journal-DsZKDqeh.js.map +0 -1
  143. package/dist/structural-rollout-zY0oqQzO.js.map +0 -1
  144. package/dist/supervise-Ds8FtyI9.js.map +0 -1
  145. package/dist/supervisor-DpjO0Gmy.js.map +0 -1
  146. package/dist/types-C9j4qg6l.d.ts +0 -500
@@ -1,19 +1,24 @@
1
- import { n as AnalystError, r as BackendTransportError, s as PlannerError, u as ValidationError } from "./errors-DEAvWQPy.js";
2
- import { i as normalizeBackendStreamEvent, o as newRuntimeSession, s as nowIso } from "./backends-CiOCyRHb.js";
3
- import { i as InMemorySpawnJournal, r as InMemoryResultBlobStore, v as isTraceAnalysisStore } from "./spawn-journal-DsZKDqeh.js";
4
- import { a as randomSuffix, c as stringifySafe, f as zeroTokenUsage, r as isAbortError, s as sleep, t as addTokenUsage } from "./util-MVgdwuIS.js";
1
+ import { n as AnalystError, s as PlannerError, u as ValidationError } from "./errors-DEAvWQPy.js";
2
+ import "./stream-agent-turn-rYgaOLO0.js";
3
+ import { i as InMemorySpawnJournal, r as InMemoryResultBlobStore, v as isTraceAnalysisStore, x as contentAddress } from "./spawn-journal-saHQzqYi.js";
4
+ import { a as randomSuffix, c as stringifySafe, f as zeroTokenUsage, r as isAbortError, s as sleep, t as addTokenUsage } from "./util-Bw6srryQ.js";
5
5
  import { i as redactProtectedValue, r as redactProtectedReason } from "./protected-redaction--F3v1oo8.js";
6
- import { _t as routerChatWithUsage, bt as runBrainLoop, en as isHarnessNativeModel, et as rollingDispatch, ht as routerBrain, l as withDriverExecutor, m as settledToIteration, n as createSupervisor } from "./supervisor-DpjO0Gmy.js";
7
- import { C as observe, O as strategyAuthorMethod, b as sample, v as refine, x as sampleThenRefine, y as runAgentic } from "./structural-rollout-zY0oqQzO.js";
8
- import { i as notifyRuntimeHookEvent, t as composeRuntimeHooks } from "./runtime-hooks-C7iJOWm3.js";
9
- import { a as notifySandboxEventObserver, i as mapSandboxToolEvent, r as mapSandboxEvent, t as createSandboxToolPartState } from "./sandbox-events-Yhd1GYWl.js";
10
- import { Ct as createWorktreeCliExecutor, Et as createPushTraceSource, Ft as probeSandboxCapabilities, Mt as runAgentRounds, Pt as createSandboxLineage, Qt as kernelPromptRegistry, St as createExecutorRegistry, Zt as formatPromptHandle, jt as defaultSelectWinner, n as supervise, nn as gateOnDeliverable, r as workerFromBackend, xt as createExecutor } from "./supervise-Ds8FtyI9.js";
11
- import { CODING_HARNESSES, InMemoryTraceStore, OUTPUT_VALUE, benjaminiHochberg, buildTrajectory, computeFindingId as computeFindingId$1, confidenceInterval, expandProfileAxes, harnessAxisOf, makeFinding as makeFinding$1, pairedBootstrap, paretoFrontier, scoreKnowledgeReadiness, wilcoxonSignedRank, wilson } from "@tangle-network/agent-eval";
6
+ import { a as notifySandboxEventObserver } from "./sandbox-events-CRDwc5WN.js";
7
+ import { c as profileProviderModel, i as concreteModelId, s as profileModelExecutionSettings, t as assertExecutableAgentProfile } from "./model-policy-CqziaqS1.js";
8
+ import { v as executableAgentProfileSnapshot, y as executableAgentSpecSnapshot } from "./materialization-COJ1UYQ-.js";
9
+ import { $ as createSandboxLineage, I as createWorktreeCliExecutor, N as createExecutor, P as createExecutorRegistry, Q as runAgentRounds, U as createPushTraceSource, Z as defaultSelectWinner, et as probeSandboxCapabilities, l as withDriverExecutor, m as settledToIteration, n as createSupervisor, nt as routerBrain, rt as runBrainLoop, w as rollingDispatch } from "./supervisor-CV6Jh28D.js";
10
+ import "./environment-provider-Dyg8DtLK.js";
11
+ import { i as notifyRuntimeHookEvent } from "./runtime-hooks-C7iJOWm3.js";
12
+ import { C as observe, O as strategyAuthorMethod, T as profileChatClient, b as sample, v as refine, x as sampleThenRefine, y as runAgentic } from "./structural-rollout-3uxVGcg2.js";
13
+ import { It as gateOnDeliverable, Lt as mapExecutorResult, n as supervise } from "./supervise-iPN27pO0.js";
14
+ import "./authoring-CvHwo1oW.js";
15
+ import "./graph-BJTxGOFB.js";
16
+ import { CODING_HARNESSES, InMemoryTraceStore, OUTPUT_VALUE, benjaminiHochberg, buildTrajectory, computeFindingId as computeFindingId$1, confidenceInterval, expandProfileAxes, makeFinding as makeFinding$1, pairedBootstrap, paretoFrontier, wilcoxonSignedRank, wilson } from "@tangle-network/agent-eval";
12
17
  import { heldoutSignificance, runProfileMatrix } from "@tangle-network/agent-eval/campaign";
13
- import { agentProfileSchema, canonicalCandidateDigest, validateAgentProfileSecurity } from "@tangle-network/agent-interface";
18
+ import { agentProfileSchema, canonicalAgentProfileDigest, validateAgentProfileSecurity } from "@tangle-network/agent-interface";
19
+ import { dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
14
20
  import { mkdir, readFile, writeFile } from "node:fs/promises";
15
21
  import { appendFileSync, chmodSync, constants, copyFileSync, existsSync, lstatSync, mkdirSync, mkdtempSync, readFileSync, readlinkSync, rmSync, symlinkSync, writeFileSync } from "node:fs";
16
- import { dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
17
22
  import { execFileSync, spawn, spawnSync } from "node:child_process";
18
23
  import { tmpdir } from "node:os";
19
24
  import { createInterface } from "node:readline";
@@ -530,13 +535,13 @@ const auditSchema = {
530
535
  };
531
536
  /** The route-rigor analyst: compare declared vs revealed vs user intent over a trajectory and return aligned / drifting / diverged with evidence and one recommended intervention. */
532
537
  async function auditIntent(input, opts) {
533
- const res = await opts.chat.chat({
534
- ...opts.model ? { model: opts.model } : {},
538
+ const res = await profileChatClient({
539
+ profile: opts.profile,
540
+ executor: opts.executor,
541
+ context: "intent auditor"
542
+ }).chat({
535
543
  jsonSchema: auditSchema,
536
544
  messages: [{
537
- role: "system",
538
- content: opts.auditorInstruction ?? "You audit whether an AI agent is on the RIGHT ROUTE — not whether it works hard, but whether its actions serve the stated intents. Infer the REVEALED intent from the action pattern (what the trajectory is actually optimizing). Compare against the declared task intent, the user intent when given, and the meta-intent when given. Flawless execution down the wrong route is DIVERGED. Busy-work that neither advances nor harms is DRIFTING. Judge only from the trajectory — be specific about which actions ground your verdict. Recommend abort only when continuing cannot serve the intent."
539
- }, {
540
545
  role: "user",
541
546
  content: `DECLARED INTENT (the task):\n${input.declaredIntent}\n\n` + (input.userIntent ? `USER INTENT (the principal's actual goal):\n${input.userIntent}\n\n` : "") + (input.metaIntent ? `META-INTENT (what the whole run is for):\n${input.metaIntent}\n\n` : "") + `TRAJECTORY (in order):\n${summarize(input.trace, opts.maxTraceLines ?? 80)}\n\nAudit the route: revealed intent, verdict, evidence, one recommendation.`
542
547
  }]
@@ -1035,7 +1040,10 @@ function loopCostReceipt(result, model) {
1035
1040
  model,
1036
1041
  inputTokens: result.tokenUsage.input,
1037
1042
  outputTokens: result.tokenUsage.output,
1038
- ...result.costUsd > 0 ? { actualCostUsd: result.costUsd } : {}
1043
+ ...result.tokenUsage.tokensKnown === false ? { usageUnknown: true } : {},
1044
+ ...result.costUsdKnown !== false ? { actualCostUsd: result.costUsd } : {},
1045
+ ...result.costUsdKnown === false ? { costUnknown: true } : {},
1046
+ ...result.estimatedCostUsd !== void 0 ? { estimatedCostUsd: result.estimatedCostUsd } : {}
1039
1047
  };
1040
1048
  }
1041
1049
  function modelFromLoopOptions(options) {
@@ -1070,6 +1078,20 @@ function loopDispatch(opts) {
1070
1078
  }
1071
1079
  //#endregion
1072
1080
  //#region src/runtime/inline-sandbox-client.ts
1081
+ /**
1082
+ * The ONE pseudo-box adapter: present any one-shot `Executor` (router / bridge /
1083
+ * BYO) as a `SandboxClient` so the round-synchronous `runAgentRounds` can drive it
1084
+ * without each call site re-faking a box. This is the single shell that
1085
+ * `bench/src/router-executor.ts`, generate-eval's old `bridgeSandboxClient`, and
1086
+ * the search-bench bridge transport were each re-implementing.
1087
+ *
1088
+ * It is deliberately for NON-box executors only — a real sandbox harness already
1089
+ * IS a `SandboxClient` (boxes, sessions, fs, fork are real there). Here each
1090
+ * `streamPrompt` runs the executor once and emits the terminal
1091
+ * `{type:'result', data:{finalText, tokenUsage, costUsd}}` event that
1092
+ * `answerOutput`/the kernel's cost ledger already parse — no sessions, no fs,
1093
+ * no fork (those degrade gracefully via the optional `SandboxClient` methods).
1094
+ */
1073
1095
  function isAsyncIterable$1(v) {
1074
1096
  return typeof v === "object" && v !== null && Symbol.asyncIterator in v;
1075
1097
  }
@@ -1087,7 +1109,8 @@ async function settle(exec, task, signal) {
1087
1109
  * instantiated fresh per `streamPrompt` (mirrors the per-spawn executor lifecycle):
1088
1110
  * run once on the prompt, emit the terminal result event, tear down.
1089
1111
  */
1090
- function inlineSandboxClient(factory) {
1112
+ function inlineSandboxClient(factory, defaults = {}) {
1113
+ const capturedDefaultProfile = defaults.profile === void 0 ? void 0 : agentProfileSchema.parse(structuredClone(defaults.profile));
1091
1114
  let seq = 0;
1092
1115
  return { async create(options) {
1093
1116
  const id = `inline-${seq++}`;
@@ -1100,8 +1123,11 @@ function inlineSandboxClient(factory) {
1100
1123
  const onAbort = () => controller.abort(callerSignal?.reason ?? /* @__PURE__ */ new Error("prompt aborted"));
1101
1124
  if (callerSignal) if (callerSignal.aborted) onAbort();
1102
1125
  else callerSignal.addEventListener("abort", onAbort, { once: true });
1126
+ const requestedProfile = (options?.backend && typeof options.backend === "object" ? options.backend.profile : void 0) ?? capturedDefaultProfile;
1127
+ const parsedProfile = agentProfileSchema.safeParse(requestedProfile);
1128
+ if (!parsedProfile.success) throw new Error("inlineSandboxClient: an exact AgentProfile is required; pass defaults.profile or create({ backend: { profile } })");
1103
1129
  const exec = factory({
1104
- profile: { name: id },
1130
+ profile: parsedProfile.data,
1105
1131
  harness: null
1106
1132
  }, {
1107
1133
  signal: controller.signal,
@@ -1113,23 +1139,43 @@ function inlineSandboxClient(factory) {
1113
1139
  const tokensIn = artifact.spent.tokens.input;
1114
1140
  const tokensOut = artifact.spent.tokens.output;
1115
1141
  const costUsd = artifact.spent.usd;
1116
- if (tokensIn || tokensOut || costUsd) yield {
1142
+ const estimatedCostUsd = out?.estimatedCostUsd;
1143
+ if (artifact.spent.iterations > 0 || artifact.spent.tokensKnown === false || artifact.spent.usdKnown === false || tokensIn > 0 || tokensOut > 0 || costUsd > 0 || estimatedCostUsd !== void 0) yield {
1117
1144
  type: "llm_call",
1118
1145
  data: {
1119
- tokensIn,
1120
- tokensOut,
1121
- costUsd
1146
+ ...artifact.spent.tokensKnown === false ? {} : {
1147
+ tokensIn,
1148
+ tokensOut
1149
+ },
1150
+ ...artifact.spent.usdKnown !== false ? { costUsd } : {},
1151
+ ...artifact.spent.tokensKnown === false ? { tokensKnown: false } : {},
1152
+ ...artifact.spent.usdKnown === false ? { costKnown: false } : {},
1153
+ ...artifact.spent.usdKnown !== false ? {
1154
+ costKnown: true,
1155
+ costProvenance: "provider-receipt"
1156
+ } : {},
1157
+ ...estimatedCostUsd !== void 0 ? { estimatedCostUsd } : {},
1158
+ ...out?.promptCache ? { promptCache: out.promptCache } : {}
1122
1159
  }
1123
1160
  };
1124
1161
  yield {
1125
1162
  type: "result",
1126
1163
  data: {
1127
1164
  finalText: out?.content ?? "",
1128
- tokenUsage: {
1165
+ ...artifact.spent.tokensKnown === false ? { tokensKnown: false } : { tokenUsage: {
1129
1166
  inputTokens: tokensIn,
1130
1167
  outputTokens: tokensOut
1168
+ } },
1169
+ ...artifact.spent.usdKnown === false ? {
1170
+ costKnown: false,
1171
+ ...costUsd > 0 ? { costUsd } : {}
1172
+ } : {
1173
+ costUsd,
1174
+ costKnown: true,
1175
+ costProvenance: "provider-receipt"
1131
1176
  },
1132
- costUsd
1177
+ ...estimatedCostUsd !== void 0 ? { estimatedCostUsd } : {},
1178
+ ...out?.promptCache ? { promptCache: out.promptCache } : {}
1133
1179
  }
1134
1180
  };
1135
1181
  } finally {
@@ -1158,36 +1204,52 @@ function inlineSandboxClient(factory) {
1158
1204
  * local MCP process applies only when its full canonical bytes match the fixed
1159
1205
  * constructor profile; a different generated profile is refused.
1160
1206
  *
1161
- * Event protocol matches `inlineSandboxClient`: one `llm_call` metering event
1162
- * + one terminal `result` event with finalText/tokenUsage/costUsd.
1207
+ * Event protocol matches `inlineSandboxClient`: known token usage is emitted as one `llm_call`;
1208
+ * Router catalog cost remains a separately-labelled estimate, never billed spend.
1163
1209
  */
1164
1210
  /** A same-host `SandboxClient` adapter with no process isolation. Local MCP is
1165
1211
  * refused unless the caller explicitly supplies a policy that allows it. */
1166
1212
  function localSandboxClient(opts) {
1167
- if (opts.profileSecurityPolicy?.allowLocalMcp && opts.profile === void 0) throw new ValidationError("localSandboxClient: allowLocalMcp requires a fixed author-controlled profile; dynamic profiles need a real sandbox");
1168
- const trustedProfileDigest = opts.profileSecurityPolicy?.allowLocalMcp && opts.profile !== void 0 ? canonicalCandidateDigest(opts.profile) : void 0;
1169
- const maxTurns = opts.maxTurns ?? 8;
1213
+ const defaultProfile = opts.profile === void 0 ? void 0 : executableAgentProfileSnapshot(opts.profile, "localSandboxClient default profile");
1214
+ const router = Object.freeze({ ...opts.router });
1215
+ if (opts.profileSecurityPolicy?.allowLocalMcp && defaultProfile === void 0) throw new ValidationError("localSandboxClient: allowLocalMcp requires a fixed author-controlled profile; dynamic profiles need a real sandbox");
1216
+ const trustedProfileDigest = opts.profileSecurityPolicy?.allowLocalMcp && defaultProfile !== void 0 ? canonicalAgentProfileDigest(defaultProfile) : void 0;
1170
1217
  let seq = 0;
1171
1218
  return { async create(options) {
1172
- const profile = (options?.backend)?.profile ?? opts.profile ?? {};
1173
- const policyApplies = opts.profileSecurityPolicy !== void 0 && (!opts.profileSecurityPolicy.allowLocalMcp || trustedProfileDigest !== void 0 && canonicalCandidateDigest(profile) === trustedProfileDigest);
1219
+ const profile = executableAgentProfileSnapshot((options?.backend)?.profile ?? defaultProfile, "localSandboxClient");
1220
+ const model = profileProviderModel(profile);
1221
+ const settings = profileModelExecutionSettings(profile, "localSandboxClient");
1222
+ const policyApplies = opts.profileSecurityPolicy !== void 0 && (!opts.profileSecurityPolicy.allowLocalMcp || trustedProfileDigest !== void 0 && canonicalAgentProfileDigest(profile) === trustedProfileDigest);
1174
1223
  const mcp = await materializeLocalMcp(profile, {
1175
1224
  ...opts.keys ? { keys: opts.keys } : {},
1176
1225
  ...policyApplies ? { profileSecurityPolicy: opts.profileSecurityPolicy } : {}
1177
1226
  });
1178
1227
  const brain = routerBrain({
1179
- routerBaseUrl: opts.router.baseUrl,
1180
- routerKey: opts.router.key,
1181
- model: opts.router.model
1182
- }, opts.temperature !== void 0 ? { temperature: opts.temperature } : {});
1228
+ routerBaseUrl: router.baseUrl,
1229
+ routerKey: router.key,
1230
+ model,
1231
+ ...settings.retry !== void 0 ? { retry: settings.retry } : {},
1232
+ ...settings.maxTokens !== void 0 ? { maxTokens: settings.maxTokens } : {},
1233
+ ...settings.stream !== void 0 ? { stream: settings.stream } : {}
1234
+ }, {
1235
+ ...settings.temperature !== void 0 ? { temperature: settings.temperature } : {},
1236
+ ...settings.seed !== void 0 ? { seed: settings.seed } : {},
1237
+ ...settings.toolChoice !== void 0 ? { toolChoice: settings.toolChoice } : {},
1238
+ ...settings.extraBody !== void 0 ? { extraBody: settings.extraBody } : {},
1239
+ ...profile.model?.reasoningEffort ? { reasoningEffort: profile.model.reasoningEffort } : {}
1240
+ });
1183
1241
  const system = [profile.prompt?.systemPrompt, ...profile.prompt?.instructions ?? []].filter((s) => typeof s === "string" && s.trim().length > 0).join("\n\n");
1184
1242
  return {
1185
1243
  id: `local-${seq++}`,
1186
1244
  async *streamPrompt(message, popts) {
1187
- let costUsd = 0;
1245
+ let estimatedCostUsd = 0;
1246
+ let sawEstimatedCost = false;
1188
1247
  const chat = async (messages, tools) => {
1189
1248
  const r = await brain(messages, tools);
1190
- if (r.costUsd) costUsd += r.costUsd;
1249
+ if (r.costProvenance === "catalog-estimate" && r.costUsd !== void 0) {
1250
+ estimatedCostUsd += r.costUsd;
1251
+ sawEstimatedCost = true;
1252
+ }
1191
1253
  return r;
1192
1254
  };
1193
1255
  const r = await runBrainLoop({
@@ -1201,26 +1263,31 @@ function localSandboxClient(opts) {
1201
1263
  role: "user",
1202
1264
  content: message
1203
1265
  }],
1204
- maxTurns,
1266
+ maxTurns: settings.maxTurns ?? 0,
1205
1267
  hooks: { stopBefore: () => popts?.signal?.aborted === true }
1206
1268
  });
1207
- if (r.usage.input || r.usage.output || costUsd) yield {
1269
+ if (r.turns > 0) yield {
1208
1270
  type: "llm_call",
1209
1271
  data: {
1210
- tokensIn: r.usage.input,
1211
- tokensOut: r.usage.output,
1212
- costUsd
1272
+ model,
1273
+ ...r.tokensKnown === false ? { tokensKnown: false } : {
1274
+ tokensIn: r.usage.input,
1275
+ tokensOut: r.usage.output
1276
+ },
1277
+ costKnown: false,
1278
+ ...sawEstimatedCost ? { estimatedCostUsd } : {}
1213
1279
  }
1214
1280
  };
1215
1281
  yield {
1216
1282
  type: "result",
1217
1283
  data: {
1218
1284
  finalText: r.final,
1219
- tokenUsage: {
1285
+ ...r.tokensKnown === false ? { tokensKnown: false } : { tokenUsage: {
1220
1286
  inputTokens: r.usage.input,
1221
1287
  outputTokens: r.usage.output
1222
- },
1223
- costUsd
1288
+ } },
1289
+ costKnown: false,
1290
+ ...sawEstimatedCost ? { estimatedCostUsd } : {}
1224
1291
  }
1225
1292
  };
1226
1293
  },
@@ -1268,28 +1335,26 @@ function resolveSandboxClient(opts) {
1268
1335
  return opts.sandboxClient;
1269
1336
  case "bridge": {
1270
1337
  const bridge = opts.bridge;
1271
- if (!bridge?.bearer || !bridge.model) throw new Error("resolveSandboxClient: backend 'bridge' requires bridge.bearer and bridge.model");
1338
+ if (!bridge?.bearer) throw new Error("resolveSandboxClient: backend 'bridge' requires bridge.bearer");
1272
1339
  return inlineSandboxClient(createExecutor({
1273
1340
  backend: "bridge",
1274
1341
  bridgeUrl: bridge.url ?? "http://127.0.0.1:3355",
1275
1342
  bridgeBearer: bridge.bearer,
1276
- model: bridge.model,
1277
1343
  timeoutMs: bridge.timeoutMs
1278
1344
  }));
1279
1345
  }
1280
1346
  case "router": {
1281
1347
  const router = opts.router;
1282
- if (!router?.baseUrl || !router.key || !router.model) throw new Error("resolveSandboxClient: backend 'router' requires router.baseUrl, router.key and router.model");
1348
+ if (!router?.baseUrl || !router.key) throw new Error("resolveSandboxClient: backend 'router' requires router.baseUrl and router.key");
1283
1349
  return inlineSandboxClient(createExecutor({
1284
1350
  backend: "router",
1285
1351
  routerBaseUrl: router.baseUrl,
1286
- routerKey: router.key,
1287
- model: router.model
1352
+ routerKey: router.key
1288
1353
  }));
1289
1354
  }
1290
1355
  case "local": {
1291
1356
  const local = opts.local;
1292
- if (!local?.router?.baseUrl || !local.router.key || !local.router.model) throw new Error("resolveSandboxClient: backend 'local' requires local.router.baseUrl, local.router.key and local.router.model");
1357
+ if (!local?.router?.baseUrl || !local.router.key) throw new Error("resolveSandboxClient: backend 'local' requires local.router.baseUrl and local.router.key");
1293
1358
  return localSandboxClient(local);
1294
1359
  }
1295
1360
  }
@@ -1313,7 +1378,7 @@ function resolveSandboxClient(opts) {
1313
1378
  *
1314
1379
  * - LEVEL 0 (declarative): `cases` / `prompt` / `score` / `axis`.
1315
1380
  * - LEVEL 1 (seams): `backends`, `flags`, `parseOutput`, `onCellEvents`,
1316
- * `resolveModel`, `setup`/`teardown`, `export`, `modelBackend`, `matrix`
1381
+ * `resolveModel`, `setup`/`teardown`, `export`, `matrix`
1317
1382
  * passthrough.
1318
1383
  * - LEVEL 2 (replacement): `dispatch` and `judges` swap out the whole
1319
1384
  * loop wiring or scoring; `runProfileMatrix` itself stays public as the
@@ -1372,10 +1437,6 @@ function splitList(v) {
1372
1437
  function withSnapshot(model, snapshot) {
1373
1438
  return model.includes("@") ? model : `${model}@${snapshot}`;
1374
1439
  }
1375
- /** The bare model id the backend actually serves (identity snapshot stripped). */
1376
- function bareModel(model) {
1377
- return model.split("@")[0] ?? model;
1378
- }
1379
1440
  function gitSha() {
1380
1441
  try {
1381
1442
  return execFileSync("git", ["rev-parse", "HEAD"], { encoding: "utf8" }).trim();
@@ -1457,13 +1518,11 @@ function defineLeaderboard(spec) {
1457
1518
  if (!rawModels || rawModels.length === 0) throw new Error(`defineLeaderboard(${spec.name}): no models — pass --models, set spec.axis.models, or give spec.baseProfile a model.default`);
1458
1519
  const models = rawModels.map((m) => withSnapshot(m, snapshot));
1459
1520
  const profiles = expandProfileAxes({
1460
- base: spec.baseProfile ?? {
1461
- name: spec.name,
1462
- model: { default: bareModel(models[0] ?? "") }
1463
- },
1521
+ base: spec.baseProfile,
1464
1522
  harnesses,
1465
1523
  models
1466
1524
  });
1525
+ for (const profile of profiles) assertExecutableAgentProfile(profile, `defineLeaderboard(${spec.name})`);
1467
1526
  const ctx = {
1468
1527
  name: spec.name,
1469
1528
  backend: backendName,
@@ -1488,7 +1547,6 @@ function defineLeaderboard(spec) {
1488
1547
  bridge: {
1489
1548
  url: process.env.CLI_BRIDGE_URL,
1490
1549
  bearer,
1491
- model: bareModel(models[0] ?? ""),
1492
1550
  timeoutMs: 9e5
1493
1551
  }
1494
1552
  });
@@ -1518,21 +1576,11 @@ function defineLeaderboard(spec) {
1518
1576
  return [...served][0] ?? cellProfile.model?.default;
1519
1577
  },
1520
1578
  toLoopOptions: (cellScenario, cellProfile) => {
1521
- const axis = harnessAxisOf(cellProfile);
1522
- const modelId = bareModel(axis?.model ?? models[0] ?? "");
1523
- const backendModel = {
1524
- ...spec.modelBackend,
1525
- ...!isHarnessNativeModel(modelId) || backendName === "cli-bridge" ? { model: modelId } : {}
1526
- };
1527
1579
  return {
1528
1580
  driver: naiveRetryDriver(shots),
1529
1581
  agentRun: {
1530
1582
  profile: cellProfile,
1531
- taskToPrompt: (s) => `${promptOf(s)}\n\n<!-- independent-attempt:${shotNonce++} -->`,
1532
- ...axis ? { sandboxOverrides: { backend: {
1533
- type: axis.harness,
1534
- ...Object.keys(backendModel).length > 0 ? { model: backendModel } : {}
1535
- } } } : {}
1583
+ taskToPrompt: (s) => `${promptOf(s)}\n\n<!-- independent-attempt:${shotNonce++} -->`
1536
1584
  },
1537
1585
  output: { parse: (events) => spec.parseOutput ? spec.parseOutput(events, cellScenario.case) : collectAgentResponseText(events) ?? "" },
1538
1586
  validator: { validate: async (output) => {
@@ -1653,11 +1701,10 @@ async function harvestCorpus(opts) {
1653
1701
  if (opts.signal?.aborted) return;
1654
1702
  try {
1655
1703
  const obs = await observe(input, {
1656
- chat: opts.chat,
1657
- ...opts.model ? { model: opts.model } : {},
1704
+ profile: opts.profile,
1705
+ executor: opts.executor,
1658
1706
  corpus: opts.corpus,
1659
1707
  tags: opts.tags ?? [],
1660
- ...opts.analystInstruction ? { analystInstruction: opts.analystInstruction } : {},
1661
1708
  ...opts.signal ? { signal: opts.signal } : {}
1662
1709
  });
1663
1710
  report.runsObserved += 1;
@@ -1941,7 +1988,7 @@ function observedBestScore(settledSoFar) {
1941
1988
  * CONCRETE blocker (never an eager over-fan, never a silent drop), and a `blocked` outcome always
1942
1989
  * names at least one blocker (a shape that cannot finish MUST say why — `blocked([])` throws).
1943
1990
  *
1944
- * @experimental
1991
+ * @stable
1945
1992
  */
1946
1993
  /**
1947
1994
  * The single content-free valid-only winner selector. Among the gated-VALID children only
@@ -1976,6 +2023,8 @@ function selectValidWinner(opts) {
1976
2023
  * pool would not admit, or a stage whose `collect` chose to block) short-circuits — its blockers
1977
2024
  * ARE the pipeline's blockers, never coerced past a failed stage. The terminal stage's `done`
1978
2025
  * deliverable is the pipeline's deliverable.
2026
+ *
2027
+ * @stable
1979
2028
  */
1980
2029
  function pipeline(stages) {
1981
2030
  if (stages.length === 0) throw new ValidationError("pipeline: at least one stage is required");
@@ -2012,6 +2061,8 @@ function pipeline(stages) {
2012
2061
  * `opts.width` swaps the single round for `rollingDispatch`: at most `width` items live at once,
2013
2062
  * refilled the instant one settles. Selection, blockers, and the conserved pool are unchanged —
2014
2063
  * the refill behavior lives in the existing combinator rather than in a rival primitive.
2064
+ *
2065
+ * @stable
2015
2066
  */
2016
2067
  function fanout(items, opts) {
2017
2068
  if (opts.synthesize && opts.selectWinner) throw new ValidationError("fanout: pass at most one of `synthesize` or `selectWinner`");
@@ -2095,6 +2146,8 @@ function fanout(items, opts) {
2095
2146
  * `until` on the resulting trace-derived findings (the analyst spawns into THIS scope, so its
2096
2147
  * compute is conserved-pooled — equal-k holds by construction). Absent an analyst the findings
2097
2148
  * argument is the empty array — never a fabricated finding (fail-loud honesty over a silent default).
2149
+ *
2150
+ * @stable
2098
2151
  */
2099
2152
  function loopUntil(seed, spec) {
2100
2153
  return (ctx) => ({
@@ -2143,6 +2196,8 @@ function loopUntil(seed, spec) {
2143
2196
  * reaches another judge's task; the merge never spawns or re-ranks). A `down` judge carries no
2144
2197
  * verdict and is excluded from the merge denominator. A panel that admitted no judge is a
2145
2198
  * concrete blocker before `merge` is consulted.
2199
+ *
2200
+ * @stable
2146
2201
  */
2147
2202
  function panel(spec) {
2148
2203
  if (spec.judges.length === 0) throw new ValidationError("panel: at least one judge is required");
@@ -2192,6 +2247,8 @@ function panel(spec) {
2192
2247
  * it; only a `valid` verifier verdict ships. Any other outcome (implement down, verifier down,
2193
2248
  * verifier verdict absent or not `valid`) is a concrete blocker carrying the failure verbatim —
2194
2249
  * never a coerced "done". The implement child does not grade itself.
2250
+ *
2251
+ * @stable
2195
2252
  */
2196
2253
  function verify(spec) {
2197
2254
  return (ctx) => ({
@@ -2236,6 +2293,8 @@ function verify(spec) {
2236
2293
  * the widen loop sees it. The shipped default (`flatWidenGate`) never widens, so no widen child is
2237
2294
  * ever live when the analyst runs and the wire is exact; a non-flat gate must drive the analyst on
2238
2295
  * a scope whose siblings are quiesced, or read findings without the shared-cursor drain.
2296
+ *
2297
+ * @stable
2239
2298
  */
2240
2299
  function widen(spec) {
2241
2300
  return (ctx) => ({
@@ -2685,19 +2744,22 @@ function registerShape(name, factory) {
2685
2744
  * receive a ctx with the persona seams merged in — so a persona never has to pre-close its
2686
2745
  * factories by hand. A persona may instead supply a fully-built `registry` and skip the wrap.
2687
2746
  *
2688
- * @experimental
2747
+ * @stable
2689
2748
  */
2690
2749
  /**
2691
2750
  * Build a frozen `Persona`. Fails loud on the executors-supplied invariant: a persona with
2692
2751
  * neither a pre-built registry nor a seam bag cannot resolve its built-in runtimes, so it is
2693
2752
  * unrunnable — refuse it at definition time, not at the first spawn. Pure; no I/O.
2753
+ *
2754
+ * @stable
2694
2755
  */
2695
2756
  function definePersona(input) {
2696
2757
  if (!input.executors.registry && !input.executors.seams) throw new ValidationError(`definePersona("${input.name}"): executors must supply a registry or a seams bag (built-in runtimes read their seams off ExecutorContext; neither was provided)`);
2697
2758
  if (!input.root || typeof input.root !== "object" || !("harness" in input.root)) throw new ValidationError(`definePersona("${input.name}"): root must be an AgentSpec`);
2759
+ const root = executableAgentSpecSnapshot(input.root, `definePersona("${input.name}")`);
2698
2760
  return Object.freeze({
2699
2761
  name: input.name,
2700
- root: input.root,
2762
+ root,
2701
2763
  directive: input.directive,
2702
2764
  context: input.context,
2703
2765
  executors: input.executors,
@@ -2740,6 +2802,8 @@ function createShapeContext(persona, budget, analyst) {
2740
2802
  * `ShapeContext`, and runs the resulting root `Agent` to a typed `SupervisedResult<Outcome>`.
2741
2803
  * Fail loud on an unknown shape name or an unresolvable persona registry — never a silent
2742
2804
  * default-shape fallback.
2805
+ *
2806
+ * @stable
2743
2807
  */
2744
2808
  async function runPersonified(options) {
2745
2809
  const { persona } = options;
@@ -3290,22 +3354,34 @@ async function pool(items, limit, fn) {
3290
3354
  async function preflightModels(cfg) {
3291
3355
  if (cfg.modelPreflight === false) return;
3292
3356
  if (cfg.worker.complete && !cfg.modelPreflight) return;
3293
- const models = [.../* @__PURE__ */ new Set([cfg.worker.model, cfg.worker.analystModel ?? cfg.worker.model])];
3357
+ const profiles = [cfg.worker.workerProfile, cfg.worker.analystProfile ?? cfg.worker.workerProfile];
3358
+ const profilesByModel = /* @__PURE__ */ new Map();
3359
+ for (const [index, profile] of profiles.entries()) {
3360
+ const model = concreteModelId(profile.model?.default);
3361
+ if (!model) throw new Error(`Benchmark ${index === 0 ? "worker" : "analyst"} AgentProfile.model.default must name an exact model`);
3362
+ if (!profilesByModel.has(model)) profilesByModel.set(model, profile);
3363
+ }
3364
+ const models = [...profilesByModel.keys()];
3294
3365
  const timeoutMs = cfg.modelPreflightTimeoutMs ?? 3e4;
3295
3366
  if (!Number.isFinite(timeoutMs) || timeoutMs <= 0) throw new Error("modelPreflightTimeoutMs must be a positive finite number");
3296
3367
  const check = cfg.modelPreflight ?? (async (model, worker, signal) => {
3297
- await routerChatWithUsage({
3298
- routerBaseUrl: worker.routerBaseUrl,
3299
- routerKey: worker.routerKey,
3300
- model
3301
- }, [{
3302
- role: "user",
3303
- content: "Reply OK."
3304
- }], {
3305
- maxTokens: 1,
3306
- reasoningEffort: "none",
3307
- signal
3308
- });
3368
+ const profile = profilesByModel.get(model);
3369
+ if (!profile) throw new Error(`Benchmark preflight has no AgentProfile for model ${model}`);
3370
+ await profileChatClient({
3371
+ profile,
3372
+ context: `runBenchmark model preflight (${model})`,
3373
+ executor: {
3374
+ backend: "router",
3375
+ routerBaseUrl: worker.routerBaseUrl,
3376
+ routerKey: worker.routerKey
3377
+ }
3378
+ }).chat({
3379
+ model,
3380
+ messages: [{
3381
+ role: "user",
3382
+ content: "Reply OK."
3383
+ }]
3384
+ }, { signal });
3309
3385
  });
3310
3386
  const failures = (await Promise.allSettled(models.map(async (model) => {
3311
3387
  const controller = new AbortController();
@@ -3361,8 +3437,10 @@ async function runBenchmark(cfg) {
3361
3437
  resolved: r.resolved,
3362
3438
  progression: r.progression,
3363
3439
  usd: r.usd,
3440
+ usdKnown: r.usdKnown,
3364
3441
  ms: r.ms,
3365
- tokens: r.tokens
3442
+ tokens: r.tokens,
3443
+ tokensKnown: r.tokensKnown
3366
3444
  };
3367
3445
  } catch (e) {
3368
3446
  errors[s.name] = e instanceof Error ? e.message.slice(0, 300) : String(e);
@@ -3371,11 +3449,13 @@ async function runBenchmark(cfg) {
3371
3449
  resolved: false,
3372
3450
  progression: [],
3373
3451
  usd: 0,
3452
+ usdKnown: true,
3374
3453
  ms: 0,
3375
3454
  tokens: {
3376
3455
  input: 0,
3377
3456
  output: 0
3378
- }
3457
+ },
3458
+ tokensKnown: true
3379
3459
  };
3380
3460
  }
3381
3461
  row = {
@@ -3402,6 +3482,7 @@ async function runBenchmark(cfg) {
3402
3482
  score: mean(cells.map((c) => c.score)),
3403
3483
  resolved: mean(cells.map((c) => c.resolved ? 1 : 0)),
3404
3484
  usd: mean(cells.map((c) => c.usd)),
3485
+ usdKnownRate: mean(cells.map((c) => c.usdKnown ? 1 : 0)),
3405
3486
  ms: mean(cells.map((c) => c.ms))
3406
3487
  };
3407
3488
  }
@@ -3793,6 +3874,8 @@ export default defineStrategy('your-strategy-name', async ({ surface, task, budg
3793
3874
  // your composition (listTools comes from the destructured context — it is NOT a global)
3794
3875
  })
3795
3876
  `;
3877
+ /** Standing behavior callers put in the strategy-author AgentProfile. */
3878
+ const strategyAuthorSystemPrompt = "You are a senior researcher authoring optimization strategies for agent loops: you read per-task losses like experimental data, form a mechanism-level hypothesis, and author the one composition that tests it. Output exactly one fenced ```ts code block and nothing else.";
3796
3879
  /** Static CONTRACT lint over an authored strategy module — the module-boundary
3797
3880
  * enforcement of the harness's two measurement invariants:
3798
3881
  * - author blindness: the only import allowed is the kernel surface. A body that could
@@ -3819,21 +3902,17 @@ function assertStrategyContract(code) {
3819
3902
  }
3820
3903
  /** One authoring attempt: chat with the given model, extract the fenced module. Throws
3821
3904
  * when the reply carries no code block. */
3822
- async function requestAuthoredCode(opts, model) {
3823
- const res = await opts.chat.chat({
3824
- ...model ? { model } : {},
3825
- ...opts.temperature !== void 0 ? { temperature: opts.temperature } : {},
3826
- ...opts.maxTokens !== void 0 ? { maxTokens: opts.maxTokens } : {},
3827
- messages: [{
3828
- role: "system",
3829
- content: "You are a senior researcher authoring optimization strategies for agent loops: you read per-task losses like experimental data, form a mechanism-level hypothesis, and author the one composition that tests it. Output exactly one fenced ```ts code block and nothing else."
3830
- }, {
3831
- role: "user",
3832
- content: `${opts.contract ?? strategyAuthorContract}\n\nBASELINE RESULTS on the "${opts.environmentName}" environment (budget=${opts.budget}) — the per-task losses are your gradient:\n${opts.lossesJson}\n\nAuthor ONE new strategy that you expect to beat the baselines on THIS environment at the same budget.\n${strategyAuthorMethod}\n\nOutput only the module code block.`
3833
- }]
3834
- }, { ...opts.signal ? { signal: opts.signal } : {} });
3905
+ async function requestAuthoredCode(opts, profile) {
3906
+ const res = await profileChatClient({
3907
+ profile,
3908
+ executor: opts.executor,
3909
+ context: "strategy author"
3910
+ }).chat({ messages: [{
3911
+ role: "user",
3912
+ content: `${opts.contract ?? strategyAuthorContract}\n\nBASELINE RESULTS on the "${opts.environmentName}" environment (budget=${opts.budget}) the per-task losses are your gradient:\n${opts.lossesJson}\n\nAuthor ONE new strategy that you expect to beat the baselines on THIS environment at the same budget.\n${strategyAuthorMethod}\n\nOutput only the module code block.`
3913
+ }] }, { ...opts.signal ? { signal: opts.signal } : {} });
3835
3914
  const match = res.content.match(/```(?:ts|typescript)?\s*\n([\s\S]*?)```/);
3836
- if (!match?.[1]) throw new Error(`authorStrategy: no code block in the author's reply (model=${model ?? "default"}): ${res.content.slice(0, 300)}`);
3915
+ if (!match?.[1]) throw new Error(`authorStrategy: no code block in the author's reply: ${res.content.slice(0, 300)}`);
3837
3916
  return match[1];
3838
3917
  }
3839
3918
  /** Author + load a strategy from losses. Throws when the author emits no loadable module;
@@ -3841,10 +3920,10 @@ async function requestAuthoredCode(opts, model) {
3841
3920
  async function authorStrategy(opts) {
3842
3921
  let code;
3843
3922
  try {
3844
- code = await requestAuthoredCode(opts, opts.model);
3923
+ code = await requestAuthoredCode(opts, opts.profile);
3845
3924
  } catch (primaryError) {
3846
- if (!opts.fallbackModel) throw primaryError;
3847
- code = await requestAuthoredCode(opts, opts.fallbackModel);
3925
+ if (!opts.fallbackProfile) throw primaryError;
3926
+ code = await requestAuthoredCode(opts, opts.fallbackProfile);
3848
3927
  }
3849
3928
  assertStrategyContract(code);
3850
3929
  mkdirSync(opts.outDir, { recursive: true });
@@ -3882,6 +3961,8 @@ async function authorStrategy(opts) {
3882
3961
  * Lineage fields (`parent`, `generation`) are recorded on every archive node so a
3883
3962
  * descendant-productivity parent-selection policy can be added without changing the
3884
3963
  * report schema; the v1 search authors from the latest tournament's losses.
3964
+ *
3965
+ * @experimental
3885
3966
  */
3886
3967
  /** Strategy means recomputed over the DISCRIMINATING tasks only — tasks where the field
3887
3968
  * strategies did not all score identically. Zero-spread tasks (everyone 1.0, everyone
@@ -4083,11 +4164,9 @@ async function runStrategyEvolution(cfg) {
4083
4164
  const contract = `${strategyAuthorContract}${cfg.objective === "cost" ? `\n\nYOUR OBJECTIVE: match or exceed the incumbent's SCORE while spending LESS (the losses include usd per task). Promotion requires proven score non-inferiority PLUS significant cost savings — a strategy that ties the score at half the cost WINS; a cheaper strategy that loses score by more than ${((cfg.scoreTolerance ?? .05) * 100).toFixed(0)}pp LOSES.` : ""}\n\nEXAMPLE TOOLS FROM ONE TASK (tool sets VARY per task on this domain — a strategy MUST select tool names from await listTools(handle) at runtime; hardcoding these example names will zero your score on most tasks):\n${toolCatalog}\n\nSTRATEGIES ALREADY IN THE TOURNAMENT (author something MEANINGFULLY different — a new composition, not a rename):\n${fieldSummary(archive)}\n\nYou are authoring candidate ${i + 1} of ${populationSize} this generation; explore a distinct region of the strategy space from your siblings.`;
4084
4165
  try {
4085
4166
  const authored = await authorStrategy({
4086
- chat: cfg.author.chat,
4087
- ...cfg.author.model ? { model: cfg.author.model } : {},
4088
- ...cfg.author.fallbackModel ? { fallbackModel: cfg.author.fallbackModel } : {},
4089
- ...cfg.author.temperature !== void 0 ? { temperature: cfg.author.temperature } : {},
4090
- ...cfg.author.maxTokens !== void 0 ? { maxTokens: cfg.author.maxTokens } : {},
4167
+ profile: cfg.author.profile,
4168
+ executor: cfg.author.executor,
4169
+ ...cfg.author.fallbackProfile ? { fallbackProfile: cfg.author.fallbackProfile } : {},
4091
4170
  contract,
4092
4171
  environmentName: cfg.environment.name,
4093
4172
  lossesJson,
@@ -4224,24 +4303,24 @@ async function runStrategyEvolution(cfg) {
4224
4303
  const tolerance = cfg.reproducerCheck.tolerance ?? .05;
4225
4304
  const championHoldoutScore = holdout.perStrategy[incumbent.name]?.score ?? 0;
4226
4305
  try {
4227
- const summary = (await cfg.author.chat.chat({
4228
- ...cfg.author.model ? { model: cfg.author.model } : {},
4229
- temperature: .2,
4230
- maxTokens: 512,
4231
- messages: [{
4232
- role: "system",
4233
- content: `Summarize the optimization strategy implemented by this code in at most ${words} words. Describe the COMPOSITION (shots, critique, artifact handling, restarts, stopping) — not the code. Output only the summary.`
4234
- }, {
4235
- role: "user",
4236
- content: championCode
4237
- }]
4238
- })).content.trim();
4306
+ const summary = (await profileChatClient({
4307
+ profile: {
4308
+ ...cfg.author.profile,
4309
+ prompt: {
4310
+ ...cfg.author.profile.prompt,
4311
+ systemPrompt: `Summarize the optimization strategy implemented by this code in at most ${words} words. Describe the COMPOSITION (shots, critique, artifact handling, restarts, stopping) — not the code. Output only the summary.`
4312
+ }
4313
+ },
4314
+ executor: cfg.author.executor,
4315
+ context: "strategy reproducer summary"
4316
+ }).chat({ messages: [{
4317
+ role: "user",
4318
+ content: championCode
4319
+ }] })).content.trim();
4239
4320
  const reproduced = await authorStrategy({
4240
- chat: cfg.author.chat,
4241
- ...cfg.author.model ? { model: cfg.author.model } : {},
4242
- ...cfg.author.fallbackModel ? { fallbackModel: cfg.author.fallbackModel } : {},
4243
- ...cfg.author.maxTokens !== void 0 ? { maxTokens: cfg.author.maxTokens } : {},
4244
- temperature: .2,
4321
+ profile: cfg.author.profile,
4322
+ executor: cfg.author.executor,
4323
+ ...cfg.author.fallbackProfile ? { fallbackProfile: cfg.author.fallbackProfile } : {},
4245
4324
  contract: `${strategyAuthorContract}\n\nIMPLEMENT EXACTLY THIS STRATEGY (a colleague's description — do not invent a different approach):\n${summary}`,
4246
4325
  environmentName: cfg.environment.name,
4247
4326
  lossesJson: "[]",
@@ -4288,737 +4367,126 @@ async function runStrategyEvolution(cfg) {
4288
4367
  };
4289
4368
  }
4290
4369
  //#endregion
4291
- //#region src/runtime/stream-agent-turn.ts
4292
- /**
4293
- * `streamAgentTurn` — the ONE run-a-turn event-stream contract over every
4294
- * execution substrate: a sandbox box (`SandboxInstance.streamPrompt`), a
4295
- * one-shot `Executor` (cli-bridge / router / BYO, via `ExecutorFactory`), and
4296
- * an in-process `AgentExecutionBackend` (the `resolveAgentBackend` output).
4297
- *
4298
- * One function, one vocabulary: every backend kind yields the existing
4299
- * `RuntimeStreamEvent` union incrementally and ALWAYS terminates with a
4300
- * `final` event whose `text` is the turn's final text and whose
4301
- * `metadata.tokenUsage` / `metadata.costUsd` / `metadata.model` carry the
4302
- * turn's metered usage. `collectAgentTurn` drains a stream into that terminal
4303
- * summary plus the full event list.
4304
- *
4305
- * This is a UNIFICATION seam, not a new stream parser — each kind is a thin
4306
- * adapter over code that already exists and is already hardened:
4307
- * - `box` — `mapSandboxEvent` + `extractLlmCallEvent` (sandbox-events.ts)
4308
- * project the sandbox event stream; nothing is re-mapped here.
4309
- * - `executor` — `inlineSandboxClient` (the ONE executor→box adapter) turns
4310
- * the factory into a box, then the box path drives it. The
4311
- * executor's settle/teardown lifecycle stays in that adapter.
4312
- * - `chat` — the backend's own `stream()` surface, normalized by
4313
- * `normalizeBackendStreamEvent` (the same projection
4314
- * `runAgentTaskStream` applies).
4315
- *
4316
- * Distinct from `openSandboxRun` (box-only, session resume over one persistent
4317
- * artifact, raw `SandboxEvent` deliverables) and from `runAgentTaskStream`
4318
- * (full task lifecycle: knowledge preflight, session store, resume). This is
4319
- * the minimal turn primitive underneath both worlds: prompt in, one normalized
4320
- * event stream out, terminal result+usage guaranteed on every non-thrown path.
4321
- *
4322
- * Stream envelope: `backend_start` → incremental events → (`backend_error` on
4323
- * failure) → `final`. A caller-initiated abort terminates with
4324
- * `final.status: 'aborted'`; an expired `timeoutMs` deadline with
4325
- * `final.status: 'failed'` — so cancellation stays distinguishable from a
4326
- * blown deadline.
4327
- *
4328
- * Mid-stream lifecycle work needs NO extra API: the generator is pull-based,
4329
- * so the producer is suspended between yields and resumes only when the caller
4330
- * pulls again. A consumer can therefore run arbitrary async work between
4331
- * events — sync state on each `tool_result`, decide a no-op retry after
4332
- * draining, run a pre-`done` flush when it receives `final` and BEFORE it
4333
- * forwards its own terminal event downstream. The interleaving is guaranteed
4334
- * (and locked by test): nothing is produced past the event the caller is
4335
- * holding.
4336
- *
4337
- * @experimental
4338
- */
4370
+ //#region src/runtime/supervise/chat-transport-executor.ts
4339
4371
  /**
4340
- * Run ONE agent turn on any backend kind and stream its events. Yields the
4341
- * `RuntimeStreamEvent` vocabulary incrementally and always ends with a `final`
4342
- * event carrying the turn's text and usage (`metadata.tokenUsage`,
4343
- * `metadata.costUsd?`, `metadata.model?`) — on success, failure, abort, and
4344
- * timeout alike. The generator never throws; failures surface in-band as
4345
- * `backend_error` + `final` with a typed `error` detail.
4372
+ * A session-owning composition over Runtime's canonical Router tool-loop executor.
4346
4373
  *
4347
- * @experimental
4348
- */
4349
- async function* streamAgentTurn(backend, prompt, opts = {}) {
4350
- const label = backend.kind === "chat" ? backend.backend.kind : backend.kind;
4351
- const task = {
4352
- id: `turn-${crypto.randomUUID()}`,
4353
- intent: prompt
4354
- };
4355
- const acc = {
4356
- deltaText: "",
4357
- input: 0,
4358
- output: 0,
4359
- costUsd: 0
4360
- };
4361
- const deadline = deriveTurnSignal(opts.signal, opts.timeoutMs ?? 0);
4362
- let session;
4363
- try {
4364
- session = await startTurnSession(backend, task, prompt, deadline.signal, label);
4365
- yield {
4366
- type: "backend_start",
4367
- task,
4368
- session,
4369
- backend: label,
4370
- timestamp: nowIso()
4371
- };
4372
- const inner = backend.kind === "chat" ? driveChatTurn(backend.backend, task, session, prompt, deadline.signal, acc) : driveBoxTurn(backend.kind === "executor" ? await inlineSandboxClient(backend.factory).create() : backend.box, prompt, deadline.signal, backend.agentRunName ?? "agent", acc, {
4373
- ...backend.kind !== "executor" && backend.options ? { options: backend.options } : {},
4374
- preserveToolParts: opts.preserveToolParts === true,
4375
- ...opts.onRawEvent ? { onRawEvent: opts.onRawEvent } : {}
4376
- });
4377
- for await (const event of inner) {
4378
- yield event;
4379
- throwIfAborted(deadline.signal);
4380
- }
4381
- yield buildFinalEvent(task, session, acc, {
4382
- status: "completed",
4383
- reason: "turn completed"
4384
- });
4385
- } catch (err) {
4386
- const callerAborted = opts.signal?.aborted === true;
4387
- const status = callerAborted ? "aborted" : "failed";
4388
- const message = err instanceof Error ? err.message : String(err);
4389
- const error = err instanceof BackendTransportError ? {
4390
- kind: "transport",
4391
- message,
4392
- status: err.status,
4393
- body: err.body
4394
- } : {
4395
- kind: "backend",
4396
- message
4397
- };
4398
- yield {
4399
- type: "backend_error",
4400
- task,
4401
- ...session ? { session } : {},
4402
- backend: label,
4403
- message,
4404
- recoverable: !callerAborted,
4405
- error,
4406
- timestamp: nowIso()
4407
- };
4408
- yield buildFinalEvent(task, session, acc, {
4409
- status,
4410
- reason: message,
4411
- error
4412
- });
4413
- } finally {
4414
- deadline.dispose();
4415
- }
4416
- }
4417
- /**
4418
- * Drain a `streamAgentTurn` stream (or any `RuntimeStreamEvent` stream that
4419
- * honors its terminal contract) into the turn summary plus the full event
4420
- * list. Fail-loud: throws when the stream ends without a terminal `final`
4421
- * event — a stream that violates the contract must not read as an empty turn.
4374
+ * This module adds conversation persistence for graph-edge `resume` continuity. It does not own
4375
+ * model selection, prompts, generation controls, retries, tool policy, or provider accounting:
4376
+ * those are lowered from one exact `AgentProfile` by `createExecutor({ backend: 'router-tools' })`.
4422
4377
  *
4423
4378
  * @experimental
4424
4379
  */
4425
- async function collectAgentTurn(stream) {
4426
- const events = [];
4427
- for await (const event of stream) events.push(event);
4428
- const final = events.at(-1);
4429
- if (final?.type !== "final") throw new Error(`collectAgentTurn: stream ended without a terminal 'final' event (last: ${final ? final.type : "none"})`);
4430
- const metadata = final.metadata ?? {};
4431
- const tokenUsage = metadata.tokenUsage && typeof metadata.tokenUsage === "object" ? metadata.tokenUsage : {};
4432
- const usage = {
4433
- input: finiteNumber(tokenUsage.input) ?? 0,
4434
- output: finiteNumber(tokenUsage.output) ?? 0
4435
- };
4436
- const costUsd = finiteNumber(metadata.costUsd);
4437
- if (costUsd !== void 0) usage.costUsd = costUsd;
4438
- if (typeof metadata.model === "string" && metadata.model.length > 0) usage.model = metadata.model;
4380
+ /** In-memory, process-local conversation store with detached reads and writes. */
4381
+ function createChatSessionStore() {
4382
+ const sessions = /* @__PURE__ */ new Map();
4439
4383
  return {
4440
- finalText: final.text ?? "",
4441
- usage,
4442
- events,
4443
- status: final.status,
4444
- ...final.error ? { error: final.error } : {}
4445
- };
4446
- }
4447
- /** Start the backend's session when it owns one (`chat` kind); mint a local
4448
- * correlation session otherwise. Box/executor turns carry no server session
4449
- * here — resume lives in `openSandboxRun`/`SandboxLineage`, not this primitive. */
4450
- async function startTurnSession(backend, task, prompt, signal, label) {
4451
- if (backend.kind === "chat" && backend.backend.start) return backend.backend.start({
4452
- task,
4453
- message: prompt
4454
- }, {
4455
- task,
4456
- knowledge: emptyReadiness(task),
4457
- signal
4458
- });
4459
- return newRuntimeSession(label);
4460
- }
4461
- /**
4462
- * One turn over a box: `box.streamPrompt` projected through the existing
4463
- * `mapSandboxEvent` (text/reasoning deltas +
4464
- * cost-bearing `llm_call`s), plus the opt-in `mapSandboxToolEvent` tool-part
4465
- * projection. Usage accumulates off the mapped `llm_call` events — the same
4466
- * fold `sumSandboxUsage` applies. Final text prefers the terminal
4467
- * `result`/`done`/`final` payload over concatenated deltas, because the
4468
- * sandbox `message.part.updated` fallback may carry running accumulations.
4469
- */
4470
- async function* driveBoxTurn(box, prompt, signal, agentRunName, acc, cfg) {
4471
- const callOptions = {
4472
- ...cfg.options ?? {},
4473
- signal
4474
- };
4475
- const stream = box.streamPrompt(prompt, callOptions);
4476
- const toolParts = cfg.preserveToolParts ? createSandboxToolPartState() : void 0;
4477
- for await (const event of stream) {
4478
- if (cfg.onRawEvent) await cfg.onRawEvent(event);
4479
- const terminalText = terminalTextFromSandboxEvent(event);
4480
- if (terminalText !== void 0) acc.terminalText = terminalText;
4481
- if (toolParts) for (const toolEvent of mapSandboxToolEvent(event, toolParts)) yield toolEvent;
4482
- const mapped = mapSandboxEvent(event, { agentRunName });
4483
- if (!mapped) continue;
4484
- foldEvent(mapped, acc, agentRunName);
4485
- yield mapped;
4486
- }
4487
- }
4488
- /** One turn over an in-process backend: its own `stream()` surface, projected
4489
- * through the same `normalizeBackendStreamEvent` the task lifecycle applies. */
4490
- async function* driveChatTurn(backend, task, session, prompt, signal, acc) {
4491
- const input = {
4492
- task,
4493
- message: prompt
4494
- };
4495
- const context = {
4496
- task,
4497
- knowledge: emptyReadiness(task),
4498
- session,
4499
- signal
4500
- };
4501
- for await (const raw of backend.stream(input, context)) {
4502
- const event = normalizeBackendStreamEvent(raw, task, session);
4503
- foldEvent(event, acc);
4504
- yield event;
4505
- }
4506
- }
4507
- /** Fold one normalized event into the turn accumulator (text + usage).
4508
- * `fallbackModelLabel` — a mapper-stamped run label to exclude from
4509
- * `usage.model` (it is not a backend-reported model). */
4510
- function foldEvent(event, acc, fallbackModelLabel) {
4511
- if (event.type === "text_delta") {
4512
- acc.deltaText += event.text;
4513
- return;
4514
- }
4515
- if (event.type === "llm_call") {
4516
- acc.input += event.tokensIn ?? 0;
4517
- acc.output += event.tokensOut ?? 0;
4518
- acc.costUsd += event.costUsd ?? 0;
4519
- if (event.model && event.model !== fallbackModelLabel) acc.model = event.model;
4520
- }
4521
- }
4522
- /** Read the final text off a terminal sandbox event, when present. */
4523
- function terminalTextFromSandboxEvent(event) {
4524
- if (!event || typeof event !== "object") return void 0;
4525
- const type = String(event.type ?? "");
4526
- if (type !== "result" && type !== "done" && type !== "final") return void 0;
4527
- const data = event.data && typeof event.data === "object" ? event.data : {};
4528
- for (const key of [
4529
- "finalText",
4530
- "text",
4531
- "response",
4532
- "content"
4533
- ]) {
4534
- const value = data[key];
4535
- if (typeof value === "string") return value;
4536
- }
4537
- }
4538
- function buildFinalEvent(task, session, acc, outcome) {
4539
- const finalText = acc.terminalText ?? acc.deltaText;
4540
- return {
4541
- type: "final",
4542
- task,
4543
- ...session ? { session } : {},
4544
- status: outcome.status,
4545
- reason: outcome.reason,
4546
- ...finalText ? { text: finalText } : {},
4547
- metadata: {
4548
- tokenUsage: {
4549
- input: acc.input,
4550
- output: acc.output
4551
- },
4552
- ...acc.costUsd > 0 ? { costUsd: acc.costUsd } : {},
4553
- ...acc.model ? { model: acc.model } : {}
4384
+ load(workerId) {
4385
+ const messages = sessions.get(workerId);
4386
+ return messages === void 0 ? void 0 : structuredClone(messages);
4554
4387
  },
4555
- ...outcome.error ? { error: outcome.error } : {},
4556
- timestamp: nowIso()
4388
+ save(workerId, messages) {
4389
+ sessions.set(workerId, structuredClone(messages));
4390
+ }
4557
4391
  };
4558
4392
  }
4559
- /** Minimal ready-by-construction readiness report for a requirement-free turn. */
4560
- function emptyReadiness(task) {
4561
- return scoreKnowledgeReadiness({
4562
- taskId: task.id,
4563
- requirements: []
4564
- });
4565
- }
4566
- function finiteNumber(value) {
4567
- return typeof value === "number" && Number.isFinite(value) ? value : void 0;
4568
- }
4569
- function throwIfAborted(signal) {
4570
- if (!signal.aborted) return;
4571
- throw signal.reason instanceof Error ? signal.reason : new Error(String(signal.reason));
4572
- }
4573
- /**
4574
- * Derive the turn's effective abort signal: fires when EITHER the caller's
4575
- * signal aborts OR the `timeoutMs` deadline elapses. `dispose()` clears the
4576
- * timer so a finished turn never leaks a pending timeout. `timeoutMs <= 0`
4577
- * disables the deadline. Node-portable (no `AbortSignal.any`, which needs
4578
- * >=20.3 — the package floor is >=20).
4579
- */
4580
- function deriveTurnSignal(callerSignal, timeoutMs) {
4581
- const controller = new AbortController();
4582
- const timer = timeoutMs > 0 ? setTimeout(() => controller.abort(/* @__PURE__ */ new Error(`agent turn timed out after ${timeoutMs}ms`)), timeoutMs) : void 0;
4583
- if (timer && typeof timer.unref === "function") timer.unref();
4584
- const onCallerAbort = () => controller.abort(callerSignal?.reason ?? /* @__PURE__ */ new Error("agent turn aborted"));
4585
- if (callerSignal) if (callerSignal.aborted) onCallerAbort();
4586
- else callerSignal.addEventListener("abort", onCallerAbort, { once: true });
4393
+ function exactProfile(profile, context) {
4394
+ const parsed = agentProfileSchema.safeParse(profile);
4395
+ if (!parsed.success) throw new ValidationError(`${context}: invalid AgentProfile: ${parsed.error.message}`);
4396
+ assertExecutableAgentProfile(parsed.data, context);
4397
+ return parsed.data;
4398
+ }
4399
+ function initialMessages(opts) {
4400
+ if (!opts.resume) return void 0;
4401
+ if (!opts.sessions) throw new ValidationError("chat transport: a 'resume' spawn needs the session store holding the prior conversation");
4402
+ const prior = opts.sessions.load(opts.resume.ofWorker);
4403
+ if (prior === void 0) throw new ValidationError(`chat transport: no recorded conversation for worker '${opts.resume.ofWorker}'`);
4404
+ return prior;
4405
+ }
4406
+ function executorConfig(opts) {
4407
+ const profile = exactProfile(opts.profile, "chat transport");
4408
+ if (!opts.complete && !opts.url) throw new ValidationError("chat transport: url required unless complete is injected");
4409
+ const tools = opts.tools ?? [];
4410
+ for (const tool of tools) if (!tool.spec.function.name || typeof tool.execute !== "function") throw new ValidationError("chat transport: every tool needs spec.function.name and execute");
4411
+ const resumed = initialMessages(opts);
4587
4412
  return {
4588
- signal: controller.signal,
4589
- dispose: () => {
4590
- if (timer) clearTimeout(timer);
4591
- callerSignal?.removeEventListener("abort", onCallerAbort);
4413
+ profile,
4414
+ config: {
4415
+ backend: "router-tools",
4416
+ routerBaseUrl: opts.url ?? "http://injected.invalid",
4417
+ routerKey: opts.bearer ?? (opts.complete ? "injected-transport" : ""),
4418
+ tools: tools.map((tool) => tool.spec),
4419
+ executeToolCall: async (name, args, task) => {
4420
+ const tool = tools.find((candidate) => candidate.spec.function.name === name);
4421
+ if (!tool) throw new ValidationError(`chat transport: unknown tool ${JSON.stringify(name)}`);
4422
+ return tool.execute(args, task);
4423
+ },
4424
+ ...opts.complete ? { complete: opts.complete } : {},
4425
+ ...resumed ? { initialMessages: resumed } : {},
4426
+ ...opts.sessions && opts.sessionKey ? { onMessages: (messages) => {
4427
+ opts.sessions?.save(opts.sessionKey, messages);
4428
+ } } : {}
4592
4429
  }
4593
4430
  };
4594
4431
  }
4595
- //#endregion
4596
- //#region src/runtime/supervise/graph.ts
4597
- /**
4598
- *
4599
- * `runGraph` — agent graphs: profiles as nodes, registry-backed prompt directives as edges.
4600
- *
4601
- * A topology is PLAIN DATA an agent can author in a few lines: nodes are canonical
4602
- * `AgentProfile`s (the ONLY way a node is described — no role-builder functions), edges are typed
4603
- * values carrying versioned {@link PromptHandle} directives, `deliverable` (termination) and
4604
- * `budget` (one conserved pool) are mandatory. Driver↔worker is the two-node cyclic instance;
4605
- * "agent 3 analyzes 1 and 2 and reports to 1" is ONE edge, not a framework.
4606
- *
4607
- * NOT A SECOND SCHEDULER. `runGraph` is an interpretation layer over what already runs:
4608
- * `supervise()` is the execution core — the same `supervisorAgent`/`driverAgent` machinery,
4609
- * `makeWorkerAgent` seam, conserved-pool budget, and deliverable-gated settlement every
4610
- * supervised run uses. (`runLoop` is a deprecated alias of `runAgentRounds` and is deliberately
4611
- * NOT the substrate here.) What the graph layer ADDS is exactly what a bespoke driver loop never
4612
- * had:
4613
- *
4614
- * 1. **Node pinning** — a spawn names a node (`profile.name` = node id) and the node's canonical
4615
- * profile is what runs; a driver cannot smuggle capabilities into a worker it did not define.
4616
- * 2. **Observable edges** — every delegates/analyzes traversal lands in an EDGE LEDGER
4617
- * (`delivered | stripped | empty | unpropagated`, with byte counts), in memory on
4618
- * the result AND as `edge` events in the run journal. The motivating incident: a filter
4619
- * silently replaced 1,700-char steering with 241 chars of boilerplate for three rounds and
4620
- * NO artifact said so — an unobservable edge cannot be trusted and its directive cannot be
4621
- * optimized.
4622
- * 3. **Directives as data** — edge text lives in the prompt registry (`<surface>/v<n>`), so every
4623
- * edge is a versioned optimization target, never prose hardcoded in a builder function.
4624
- * 4. **Per-edge traversal caps** — the cyclic-graph backstop. A delegates edge whose cap is
4625
- * exhausted REFUSES further traversals (fail loud), so a cycle cannot spin the pool dry.
4626
- *
4627
- * ORACLES ARE ENVIRONMENT, NEVER WORKERS. Graders/verifiers must not be spawnable in the graph —
4628
- * a delegates edge to them leaks the rubric. An `analyzes` edge names its analyst in one of two
4629
- * forms: a LENS id from the environment's registry (a pure function over trace evidence), or the
4630
- * id of a graph NODE — a tool-equipped analyst AGENT spawned on each matching settle with the
4631
- * node's pinned profile, whose settle output IS the findings. Either way the oracle doctrine
4632
- * holds: an analyst node can never be a delegates target (refused loudly), so no driver can hand
4633
- * it work, and an id living in both the registry and the nodes is refused as ambiguous.
4634
- *
4635
- * @experimental
4636
- */
4637
- /** Default per-edge traversal cap — the cyclic-graph backstop when an edge names none. */
4638
- const defaultEdgeTraversalCap = 32;
4639
- /** A delegates edge exhausted its traversal cap and the run produced no winner: the cap, not the
4640
- * task, ended it. Carries the full evidence so failing loud loses nothing. */
4641
- var GraphEdgeCapError = class extends Error {
4642
- exhaustedEdges;
4643
- ledger;
4644
- result;
4645
- constructor(exhaustedEdges, ledger, result) {
4646
- super(`runGraph: edge traversal cap exhausted on ${exhaustedEdges.join(", ")} and the run delivered no winner — the cap (the cyclic-graph backstop), not the task, ended this run. Raise maxTraversals on the edge or fix the cycle; the full edge ledger and the supervised result ride on this error.`);
4647
- this.name = "GraphEdgeCapError";
4648
- this.exhaustedEdges = exhaustedEdges;
4649
- this.ledger = ledger;
4650
- this.result = result;
4651
- }
4652
- };
4653
- function edgeId(edge) {
4654
- return edge.kind === "delegates" ? `delegates:${edge.from}->${edge.to}` : `analyzes:${edge.analyst}:${edge.over.join("+")}->${edge.to}`;
4655
- }
4656
- /** Validate the graph and resolve every directive BEFORE any compute is spent — an invalid
4657
- * topology or an unknown directive is a configuration fault, never a mid-run surprise. */
4658
- function validateGraph(graph, registry, analysts) {
4659
- if (!Array.isArray(graph.nodes) || graph.nodes.length === 0) throw new ValidationError("runGraph: graph.nodes must be a non-empty array");
4660
- if (!Array.isArray(graph.edges) || graph.edges.length === 0) throw new ValidationError("runGraph: graph.edges must be a non-empty array");
4661
- if (typeof graph.deliverable?.check !== "function") throw new ValidationError("runGraph: graph.deliverable is mandatory (termination oracle)");
4662
- if (typeof graph.budget !== "object" || graph.budget === null) throw new ValidationError("runGraph: graph.budget is mandatory (the conserved pool)");
4663
- const byId = /* @__PURE__ */ new Map();
4664
- for (const node of graph.nodes) {
4665
- if (typeof node.id !== "string" || node.id.length === 0) throw new ValidationError("runGraph: every node needs a non-empty string id");
4666
- if (byId.has(node.id)) throw new ValidationError(`runGraph: duplicate node id '${node.id}'`);
4667
- const parsed = agentProfileSchema.safeParse(node.profile);
4668
- if (!parsed.success) throw new ValidationError(`runGraph: node '${node.id}' has an invalid AgentProfile: ${parsed.error.message}`);
4669
- if (node.profile.name !== node.id) throw new ValidationError(`runGraph: node '${node.id}' has profile.name ${JSON.stringify(node.profile.name)} — profile.name IS the node identity (node pinning and analyst routing match on it) and must equal the node id`);
4670
- byId.set(node.id, node);
4671
- }
4672
- const requireNode = (id, where) => {
4673
- const node = byId.get(id);
4674
- if (!node) throw new ValidationError(`runGraph: ${where} references unknown node '${id}'`);
4675
- return node;
4432
+ function buildChatTransportExecutor(opts, context) {
4433
+ const { config, profile } = executorConfig(opts);
4434
+ const spec = {
4435
+ profile,
4436
+ harness: null
4676
4437
  };
4677
- const delegates = graph.edges.filter((edge) => edge.kind === "delegates");
4678
- const analyzes = graph.edges.filter((edge) => edge.kind === "analyzes");
4679
- if (delegates.length === 0) throw new ValidationError("runGraph: at least one delegates edge is required (who spawns whom)");
4680
- for (const edge of graph.edges) registry.resolve(edge.directive);
4681
- for (const edge of delegates) {
4682
- requireNode(edge.from, edgeId(edge));
4683
- requireNode(edge.to, edgeId(edge));
4684
- if (edge.from === edge.to) throw new ValidationError(`runGraph: ${edgeId(edge)} delegates to itself — the driver↔worker cycle is the settle-return loop, not a self-edge`);
4685
- }
4686
- const delegatedTo = new Set(delegates.map((edge) => edge.to));
4687
- const roots = [...new Set(delegates.map((edge) => edge.from))].filter((id) => !delegatedTo.has(id));
4688
- if (roots.length !== 1) throw new ValidationError(`runGraph: expected exactly ONE root (a node that delegates and is never delegated to), found ${roots.length === 0 ? "none — delegates edges form a cycle with no entry" : roots.join(", ")}. P0 executes driver↔worker(s); nested driver graphs are P3.`);
4689
- const root = requireNode(roots[0], "root resolution");
4690
- for (const edge of delegates) if (edge.from !== root.id) throw new ValidationError(`runGraph: ${edgeId(edge)} delegates from a non-root node — P0 executes one driver over its workers (the 2-node cyclic case, star-generalized); deeper delegation is P3`);
4691
- const analystIds = /* @__PURE__ */ new Set();
4692
- const analystNodes = /* @__PURE__ */ new Map();
4693
- for (const edge of analyzes) {
4694
- if (analystIds.has(edge.analyst)) throw new ValidationError(`runGraph: two analyzes edges share analyst '${edge.analyst}' — one analyzes edge per analyst lens (traversals are ledgered by analyst id; a second edge would silently absorb the first's). Register the lens under a second id for a second edge.`);
4695
- analystIds.add(edge.analyst);
4696
- const analystNode = byId.get(edge.analyst);
4697
- const inRegistry = analysts?.kinds.some((kind) => kind.id === edge.analyst) === true;
4698
- if (analystNode !== void 0 && inRegistry) throw new ValidationError(`runGraph: ${edgeId(edge)} analyst '${edge.analyst}' is BOTH a graph node and a lens in the analysts registry — the id alone distinguishes the two analyst forms, so this is ambiguous; rename the node or register the lens under another id`);
4699
- if (analystNode !== void 0) {
4700
- if (analystNode.id === root.id) throw new ValidationError(`runGraph: ${edgeId(edge)} names the ROOT as its analyst — the root is the driver; give the analyst its own node with no delegates edge pointing at it`);
4701
- if (delegatedTo.has(analystNode.id)) throw new ValidationError(`runGraph: ${edgeId(edge)} names node '${edge.analyst}' as its analyst, but that node is a delegates target — oracle doctrine: an analyst is never delegated to. An analyst NODE is legal only with NO delegates edge pointing at it; give the analyst its own delegates-free node or pass a lens id from RunGraphOptions.analysts.`);
4702
- analystNodes.set(analystNode.id, analystNode);
4703
- } else if (!analysts) throw new ValidationError(`runGraph: ${edgeId(edge)} analyst '${edge.analyst}' is not a graph node, and no RunGraphOptions.analysts registry was provided to resolve it as a lens`);
4704
- else if (!inRegistry) throw new ValidationError(`runGraph: ${edgeId(edge)} analyst '${edge.analyst}' is neither a graph node nor in the analysts registry (known lenses: ${analysts.kinds.map((kind) => kind.id).join(", ") || "none"})`);
4705
- if (edge.over.length === 0) throw new ValidationError(`runGraph: ${edgeId(edge)} must analyze at least one node`);
4706
- for (const over of edge.over) requireNode(over, edgeId(edge));
4707
- requireNode(edge.to, edgeId(edge));
4708
- }
4709
- for (const edge of analyzes) for (const over of edge.over) if (analystNodes.has(over)) throw new ValidationError(`runGraph: ${edgeId(edge)} analyzes '${over}', which is an analyst node — an analyst run settles as a finding, never as a worker, so this edge would silently never fire; analyst nodes are not analyzable`);
4710
- const workers = /* @__PURE__ */ new Map();
4711
- const delegatesByWorker = /* @__PURE__ */ new Map();
4712
- for (const edge of delegates) {
4713
- if (delegatesByWorker.has(edge.to)) throw new ValidationError(`runGraph: node '${edge.to}' is the target of two delegates edges — one delegation directive per worker node (version the directive instead of forking the edge)`);
4714
- delegatesByWorker.set(edge.to, edge);
4715
- workers.set(edge.to, requireNode(edge.to, edgeId(edge)));
4716
- }
4717
- for (const node of graph.nodes) if (node.id !== root.id && !workers.has(node.id) && !analystNodes.has(node.id)) throw new ValidationError(`runGraph: node '${node.id}' has no delegates edge to it — an unreachable node never runs`);
4718
- return {
4719
- root,
4720
- workers,
4721
- delegatesByWorker,
4722
- analyzes,
4723
- analystNodes
4724
- };
4725
- }
4726
- const byteLength = (text) => Buffer.byteLength(text, "utf8");
4727
- function stringifyPayload(payload) {
4728
- if (typeof payload === "string") return payload;
4729
- try {
4730
- return JSON.stringify(payload) ?? String(payload);
4731
- } catch {
4732
- return String(payload);
4733
- }
4438
+ return mapExecutorResult(createExecutor(config)(spec, context), (result) => {
4439
+ const raw = result.out;
4440
+ const content = typeof raw?.content === "string" ? raw.content : "";
4441
+ return {
4442
+ outRef: contentAddress({
4443
+ kind: "chat-transport",
4444
+ profile,
4445
+ content
4446
+ }),
4447
+ out: content,
4448
+ ...result.verdict ? { verdict: result.verdict } : {}
4449
+ };
4450
+ });
4734
4451
  }
4735
4452
  /**
4736
- * Execute an {@link AgentGraph}. The root node becomes the supervisor (`supervise()` — the
4737
- * execution core), each worker node is spawnable BY NODE ID (`spawn_agent` with
4738
- * `profile: { name: '<node id>' }`; the node's canonical profile is pinned by the graph), each
4739
- * delegates directive is appended to the worker profile's `prompt.instructions` per traversal,
4740
- * and each analyzes edge becomes an analyst-on-settle route with a real DESTINATION. Every
4741
- * traversal is ledgered and journaled.
4453
+ * Build one exact profile-driven chat executor through `createExecutor`.
4454
+ * Prefer `chatWorkerSeam` for supervised work because it supplies trusted node identity.
4742
4455
  */
4743
- function runGraph(graph, opts) {
4744
- const registry = opts.registry ?? kernelPromptRegistry();
4745
- const { root, workers, delegatesByWorker, analyzes, analystNodes } = validateGraph(graph, registry, opts.analysts);
4746
- if (!opts.backend && !opts.makeWorkerAgent) throw new ValidationError("runGraph: provide opts.backend (where nodes run) or opts.makeWorkerAgent");
4747
- const journal = opts.journal ?? new InMemorySpawnJournal();
4748
- const blobs = opts.blobs ?? new InMemoryResultBlobStore();
4749
- const runId = opts.runId ?? `graph-${canonicalCandidateDigest(graph.nodes.map((n) => n.id)).slice(7, 19)}`;
4750
- const now = opts.now ?? Date.now;
4751
- const ledger = [];
4752
- const journaled = /* @__PURE__ */ new Set();
4753
- const traversalCounts = /* @__PURE__ */ new Map();
4754
- const exhausted = /* @__PURE__ */ new Set();
4755
- const exhaustedDelegates = /* @__PURE__ */ new Set();
4756
- const journalWrites = [];
4757
- let ledgerSeq = 0;
4758
- const appendJournal = (entry, nodeIdForEvent) => {
4759
- if (journaled.has(entry)) return Promise.resolve();
4760
- journaled.add(entry);
4761
- const write = journal.appendEvent(runId, {
4762
- kind: "edge",
4763
- id: nodeIdForEvent,
4764
- edge: {
4765
- kind: entry.kind,
4766
- from: entry.from,
4767
- to: entry.to,
4768
- directive: entry.directive
4769
- },
4770
- traversal: entry.traversal,
4771
- outcome: entry.outcome,
4772
- bytes: entry.bytes,
4773
- ...entry.reason !== void 0 ? { reason: entry.reason } : {},
4774
- seq: ledgerSeq++,
4775
- at: new Date(now()).toISOString()
4776
- });
4777
- journalWrites.push(write);
4778
- return write;
4779
- };
4780
- const record = (entry, journalNow) => {
4781
- const count = (traversalCounts.get(entry.edge) ?? 0) + 1;
4782
- traversalCounts.set(entry.edge, count);
4783
- const row = {
4784
- ...entry,
4785
- traversal: count
4786
- };
4787
- ledger.push(row);
4788
- if (journalNow) appendJournal(row, row.workerId ?? `graph:${row.to}`);
4789
- return row;
4790
- };
4791
- const makeLeaf = opts.makeWorkerAgent ?? workerFromBackend(opts.backend, graph.deliverable);
4792
- const nodeByWorkerId = /* @__PURE__ */ new Map();
4793
- const pendingByAssignment = /* @__PURE__ */ new Map();
4794
- const graphWorker = (authoredProfile, spawnContext) => {
4795
- const requested = typeof authoredProfile?.name === "string" ? authoredProfile.name : void 0;
4796
- if (spawnContext?.analyst !== void 0) {
4797
- const analystNode = analystNodes.get(spawnContext.analyst);
4798
- if (!analystNode || requested !== analystNode.id) throw new ValidationError(`runGraph: analyst run for ${JSON.stringify(spawnContext.analyst)} does not name an analyst node of this graph (analyst nodes: ${[...analystNodes.keys()].join(", ") || "none"})`);
4799
- return makeLeaf(analystNode.profile, spawnContext);
4800
- }
4801
- const node = requested !== void 0 ? workers.get(requested) : void 0;
4802
- if (!node) throw new ValidationError(`runGraph: spawn_agent named profile ${JSON.stringify(requested)} which is not a worker node of this graph (nodes: ${[...workers.keys()].join(", ")}). Spawn by node id: profile.name selects the node; the node profile itself is pinned by the graph.`);
4803
- const edge = delegatesByWorker.get(node.id);
4804
- const id = edgeId(edge);
4805
- const cap = edge.maxTraversals ?? 32;
4806
- if ((traversalCounts.get(id) ?? 0) >= cap) {
4807
- exhausted.add(id);
4808
- exhaustedDelegates.add(id);
4809
- record({
4810
- edge: id,
4811
- kind: "delegates",
4812
- from: edge.from,
4813
- to: edge.to,
4814
- directive: formatPromptHandle(edge.directive),
4815
- outcome: "unpropagated",
4816
- bytes: 0,
4817
- reason: `traversal-cap-exhausted (max ${cap})`
4818
- }, true);
4819
- throw new ValidationError(`runGraph: delegates edge ${id} exhausted its traversal cap (${cap}) — the cyclic-graph backstop refused this spawn`);
4820
- }
4821
- const directiveText = registry.resolve(edge.directive).text;
4822
- const taskText = stringifyPayload(spawnContext?.task);
4823
- const bytes = byteLength(directiveText) + byteLength(taskText);
4824
- const row = record({
4825
- edge: id,
4826
- kind: "delegates",
4827
- from: edge.from,
4828
- to: edge.to,
4829
- directive: formatPromptHandle(edge.directive),
4830
- outcome: bytes === 0 ? "empty" : "delivered",
4831
- bytes,
4832
- ...bytes === 0 ? { reason: "no directive text and no task payload" } : {}
4833
- }, false);
4834
- if (spawnContext?.assignmentId !== void 0) pendingByAssignment.set(spawnContext.assignmentId, row);
4835
- else appendJournal(row, `graph:${row.to}`);
4836
- const pinned = directiveText.length === 0 ? node.profile : {
4837
- ...node.profile,
4838
- prompt: {
4839
- ...node.profile.prompt ?? {},
4840
- instructions: [...node.profile.prompt?.instructions ?? [], directiveText]
4841
- }
4842
- };
4843
- return makeLeaf(pinned, spawnContext);
4844
- };
4845
- const routes = analyzes.map((edge) => {
4846
- const analystNode = analystNodes.get(edge.analyst);
4847
- if (analystNode) return {
4848
- kind: edge.analyst,
4849
- over: edge.over,
4850
- agent: analystNode.profile,
4851
- directive: registry.resolve(edge.directive).text,
4852
- ...edge.to === root.id ? {} : { to: edge.to }
4853
- };
4854
- return edge.to === root.id ? {
4855
- kind: edge.analyst,
4856
- over: edge.over
4857
- } : {
4858
- kind: edge.analyst,
4859
- over: edge.over,
4860
- to: edge.to,
4861
- directive: registry.resolve(edge.directive).text
4862
- };
4456
+ function chatTransportExecutor(opts) {
4457
+ return buildChatTransportExecutor(opts, {
4458
+ signal: new AbortController().signal,
4459
+ seams: {}
4863
4460
  });
4864
- const driverAnalyzesBriefs = analyzes.filter((edge) => edge.to === root.id).map((edge) => analystNodes.has(edge.analyst) ? `Findings from analyst '${edge.analyst}' (a tool-equipped analyst agent node, over: ${edge.over.join(", ")}) will arrive as finding events.` : `Findings from analyst '${edge.analyst}' (over: ${edge.over.join(", ")}) will arrive as finding events.\n${registry.resolve(edge.directive).text}`);
4865
- const graphBrief = [
4866
- "AGENT GRAPH: you are the driver node of a fixed topology. You may spawn ONLY these worker",
4867
- "nodes, by EXACT name (spawn_agent with profile: { name: '<node id>' }; the node's full",
4868
- "profile is pinned by the graph — any other profile fields you author are ignored):",
4869
- ...[...workers.values()].map((node) => {
4870
- const cap = delegatesByWorker.get(node.id).maxTraversals ?? 32;
4871
- const description = typeof node.profile.description === "string" && node.profile.description.length > 0 ? ` — ${node.profile.description}` : "";
4872
- return `- '${node.id}'${description} (delegation cap: ${cap} traversals)`;
4873
- }),
4874
- ...driverAnalyzesBriefs.length > 0 ? ["", ...driverAnalyzesBriefs] : []
4875
- ].join("\n");
4876
- const rootProfile = {
4877
- ...root.profile,
4878
- prompt: {
4879
- ...root.profile.prompt ?? {},
4880
- instructions: [...root.profile.prompt?.instructions ?? [], graphBrief]
4881
- }
4882
- };
4883
- const strippedByDigest = /* @__PURE__ */ new Map();
4884
- const authorizeMessage = opts.authorizeMessage ? (input) => {
4885
- const decision = opts.authorizeMessage(input);
4886
- if (decision.instruction !== input.instruction) strippedByDigest.set(canonicalCandidateDigest(decision.instruction), { composedBytes: byteLength(input.instruction) });
4887
- return decision;
4888
- } : void 0;
4889
- const routedAnalyzesByAnalyst = /* @__PURE__ */ new Map();
4890
- const driverAnalyzesByAnalyst = /* @__PURE__ */ new Map();
4891
- for (const edge of analyzes) (edge.to === root.id ? driverAnalyzesByAnalyst : routedAnalyzesByAnalyst).set(edge.analyst, edge);
4892
- const analyzesCapReached = (edge) => {
4893
- const cap = edge.maxTraversals ?? 32;
4894
- if ((traversalCounts.get(edgeId(edge)) ?? 0) < cap) return false;
4895
- exhausted.add(edgeId(edge));
4896
- return true;
4897
- };
4898
- const ledgerAnalyzes = (edge, outcome, bytes, reason, workerId) => {
4899
- const capped = analyzesCapReached(edge);
4900
- record({
4901
- edge: edgeId(edge),
4902
- kind: "analyzes",
4903
- from: edge.over.join("+"),
4904
- to: edge.to,
4905
- directive: formatPromptHandle(edge.directive),
4906
- outcome: capped ? "unpropagated" : outcome,
4907
- bytes,
4908
- ...capped ? { reason: `traversal-cap-exhausted (max ${edge.maxTraversals ?? 32})` } : reason !== void 0 ? { reason } : {},
4909
- ...workerId !== void 0 ? { workerId } : {}
4910
- }, true);
4911
- };
4912
- const onCoordinationEvent = async (_context, _eventId, recordEnvelope) => {
4913
- const event = recordEnvelope.event;
4914
- if (event.type === "finding") {
4915
- const edge = driverAnalyzesByAnalyst.get(event.finding.analyst);
4916
- if (!edge) return;
4917
- const sourceNode = nodeByWorkerId.get(event.finding.fromWorker);
4918
- if (sourceNode === void 0 || !edge.over.includes(sourceNode)) return;
4919
- const findingsText = event.finding.findings === void 0 ? "" : stringifyPayload(event.finding.findings);
4920
- const directiveBytes = byteLength(registry.resolve(edge.directive).text);
4921
- const empty = findingsText.length === 0;
4922
- ledgerAnalyzes(edge, empty ? "empty" : "delivered", directiveBytes + byteLength(findingsText), empty ? "analyst returned no findings" : void 0, event.finding.fromWorker);
4923
- return;
4924
- }
4925
- if (event.type === "steer") {
4926
- const down = event.down;
4927
- if (event.analyst !== void 0) {
4928
- const edge = routedAnalyzesByAnalyst.get(event.analyst);
4929
- if (!edge) return;
4930
- ledgerAnalyzes(edge, down.delivered ? "delivered" : "unpropagated", byteLength(down.instruction), down.delivered ? void 0 : down.outcome, down.toWorker);
4931
- return;
4932
- }
4933
- const nodeId = nodeByWorkerId.get(down.toWorker);
4934
- if (nodeId === void 0) return;
4935
- const edge = delegatesByWorker.get(nodeId);
4936
- if (!edge) return;
4937
- const stripped = strippedByDigest.get(down.instructionDigest);
4938
- record({
4939
- edge: edgeId(edge),
4940
- kind: "delegates",
4941
- from: edge.from,
4942
- to: edge.to,
4943
- directive: formatPromptHandle(edge.directive),
4944
- outcome: !down.delivered ? "unpropagated" : stripped ? "stripped" : "delivered",
4945
- bytes: byteLength(down.instruction),
4946
- ...!down.delivered ? { reason: down.outcome } : stripped ? { reason: `authorization narrowed ${stripped.composedBytes} composed bytes` } : {},
4947
- workerId: down.toWorker
4948
- }, true);
4949
- }
4950
- };
4951
- const hooks = composeRuntimeHooks({ onEvent: (event) => {
4952
- if (event.target !== "agent.spawn" || event.phase !== "after") return;
4953
- const payload = event.payload;
4954
- if (typeof payload?.childId !== "string" || typeof payload.assignmentId !== "string") return;
4955
- const pending = pendingByAssignment.get(payload.assignmentId);
4956
- if (!pending) return;
4957
- pendingByAssignment.delete(payload.assignmentId);
4958
- const bound = {
4959
- ...pending,
4960
- workerId: payload.childId
4961
- };
4962
- ledger[ledger.indexOf(pending)] = bound;
4963
- nodeByWorkerId.set(payload.childId, bound.to);
4964
- return appendJournal(bound, payload.childId);
4965
- } }, opts.hooks);
4966
- const start = async () => {
4967
- const result = await supervise(rootProfile, graphTask(graph, root), {
4968
- budget: graph.budget,
4969
- deliverable: graph.deliverable,
4970
- makeWorkerAgent: graphWorker,
4971
- journal,
4972
- blobs,
4973
- runId,
4974
- hooks,
4975
- onCoordinationEvent,
4976
- ...routes.length > 0 ? {
4977
- analyzeOnSettle: routes,
4978
- ...opts.analysts ? { analysts: opts.analysts } : {}
4979
- } : {},
4980
- ...opts.watchWorkers ? { watchWorkers: opts.watchWorkers } : {},
4981
- ...opts.router ? { router: opts.router } : {},
4982
- ...opts.brain ? { brain: opts.brain } : {},
4983
- ...authorizeMessage ? { authorizeMessage } : {},
4984
- ...opts.perWorker ? { perWorker: opts.perWorker } : {},
4985
- ...opts.maxTurns !== void 0 ? { maxTurns: opts.maxTurns } : {},
4986
- ...opts.maxLiveWorkers !== void 0 ? { maxLiveWorkers: opts.maxLiveWorkers } : {},
4987
- ...opts.signal ? { signal: opts.signal } : {},
4988
- ...opts.now ? { now: opts.now } : {},
4989
- ...opts.otel ? { otel: opts.otel } : {},
4990
- ...opts.stallAfterMs !== void 0 ? { stallAfterMs: opts.stallAfterMs } : {},
4991
- ...opts.allowedModels ? { allowedModels: opts.allowedModels } : {}
4992
- });
4993
- for (const pending of pendingByAssignment.values()) {
4994
- const refused = {
4995
- ...pending,
4996
- outcome: "unpropagated",
4997
- bytes: 0,
4998
- reason: `no-live-worker-bound (spawn refused after the factory, or a keyed re-spawn deduplicated to a completed result; ${pending.bytes} composed bytes never crossed)`
4999
- };
5000
- ledger[ledger.indexOf(pending)] = refused;
5001
- await appendJournal(refused, `graph:${refused.to}`);
5002
- }
5003
- pendingByAssignment.clear();
5004
- await Promise.all(journalWrites);
5005
- const exhaustedEdges = Object.freeze([...exhausted]);
5006
- const frozenLedger = Object.freeze(ledger.map((row) => Object.freeze({ ...row })));
5007
- const lifecycleEnded = result.kind === "no-winner" && (result.reason === "aborted" || result.reason === "budget-exhausted");
5008
- if (result.kind !== "winner" && !lifecycleEnded && exhaustedDelegates.size > 0) throw new GraphEdgeCapError(Object.freeze([...exhaustedDelegates]), frozenLedger, result);
4461
+ }
4462
+ /** Session-owning worker factory for graph continuity. */
4463
+ function chatWorkerSeam(opts) {
4464
+ if (!opts.complete && !opts.url) throw new ValidationError("chatWorkerSeam: url required unless complete is injected");
4465
+ const sessions = opts.sessions ?? createChatSessionStore();
4466
+ return (rawProfile, spawnContext) => {
4467
+ const profile = exactProfile(rawProfile, "chatWorkerSeam");
5009
4468
  return {
5010
- result,
5011
- ledger: frozenLedger,
5012
- exhaustedEdges,
5013
- runId
4469
+ name: profile.name ?? "chat-worker",
4470
+ act: async () => void 0,
4471
+ executorSpec: {
4472
+ profile,
4473
+ harness: null,
4474
+ executorFactory: (executorSpec, context) => {
4475
+ const executor = buildChatTransportExecutor({
4476
+ profile: executorSpec.profile,
4477
+ ...opts.url ? { url: opts.url } : {},
4478
+ ...opts.bearer ? { bearer: opts.bearer } : {},
4479
+ ...opts.tools ? { tools: opts.tools } : {},
4480
+ ...opts.complete ? { complete: opts.complete } : {},
4481
+ sessions,
4482
+ ...context.node?.nodeId ? { sessionKey: context.node.nodeId } : {},
4483
+ ...spawnContext?.resume ? { resume: spawnContext.resume } : {}
4484
+ }, context);
4485
+ return opts.deliverable ? gateOnDeliverable(executor, opts.deliverable) : executor;
4486
+ }
4487
+ }
5014
4488
  };
5015
4489
  };
5016
- return start();
5017
- }
5018
- /** The root task: the graph's own framing. The deliverable (mandatory) is the termination; the
5019
- * task names what the topology exists to produce. */
5020
- function graphTask(graph, root) {
5021
- return graph.deliverable.describe ?? `Deliver the graph's deliverable by driving your worker nodes (root: '${root.id}').`;
5022
4490
  }
5023
4491
  //#endregion
5024
4492
  //#region src/runtime/supervise/patch-checks.ts
@@ -5459,7 +4927,6 @@ function worktreeFanout(options) {
5459
4927
  return gateOnDeliverable(createWorktreeCliExecutor({
5460
4928
  repoRoot: options.repoRoot,
5461
4929
  profile: item.profile,
5462
- harness: item.harness,
5463
4930
  taskPrompt: options.taskPrompt,
5464
4931
  executionAttemptId: ctx.node.attemptId,
5465
4932
  ...item.budgetExempt !== void 0 ? { budgetExempt: item.budgetExempt } : {},
@@ -5632,7 +5099,7 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
5632
5099
  const guidance = typeof brief === "string" ? brief.trim() : brief ? JSON.stringify(brief) : "";
5633
5100
  const attemptTask = guidance ? {
5634
5101
  ...task,
5635
- systemPrompt: `${task.systemPrompt ?? ""}\n\n— Supervisor guidance for THIS attempt (incorporate it; do not just repeat a prior approach) —\n${guidance}`
5102
+ userPrompt: `${task.userPrompt}\n\n— Supervisor guidance for THIS attempt (incorporate it; do not just repeat a prior approach) —\n${guidance}`
5636
5103
  } : task;
5637
5104
  const r = await runAgentic({
5638
5105
  surface: traced.surface,
@@ -5641,8 +5108,8 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
5641
5108
  budget: worker.budget ?? 1,
5642
5109
  routerBaseUrl: worker.routerBaseUrl,
5643
5110
  routerKey: worker.routerKey,
5644
- model: worker.model,
5645
- ...worker.maxTokens !== void 0 ? { maxTokens: worker.maxTokens } : {},
5111
+ workerProfile: worker.profile,
5112
+ ...worker.analystProfile ? { analystProfile: worker.analystProfile } : {},
5646
5113
  ...worker.innerTurns !== void 0 ? { innerTurns: worker.innerTurns } : {}
5647
5114
  });
5648
5115
  const out = {
@@ -5655,7 +5122,9 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
5655
5122
  const spent = {
5656
5123
  iterations: r.completions,
5657
5124
  tokens: r.tokens,
5125
+ ...r.tokensKnown ? {} : { tokensKnown: false },
5658
5126
  usd: r.usd,
5127
+ ...r.usdKnown ? {} : { usdKnown: false },
5659
5128
  ms: r.ms
5660
5129
  };
5661
5130
  artifact = {
@@ -5685,13 +5154,13 @@ async function superviseSurface(profile, task, opts) {
5685
5154
  const innerTurns = opts.worker.innerTurns ?? 6;
5686
5155
  const router = opts.router ?? {
5687
5156
  routerBaseUrl: opts.worker.routerBaseUrl,
5688
- routerKey: opts.worker.routerKey,
5689
- model: opts.worker.model
5157
+ routerKey: opts.worker.routerKey
5690
5158
  };
5691
5159
  const budget = opts.budget ?? {
5692
5160
  maxIterations: (innerTurns + 2) * 5 + 16,
5693
- maxTokens: (opts.worker.maxTokens ?? 4e3) * 8
5161
+ maxTokens: 1e9
5694
5162
  };
5163
+ const workerMaxTokens = profileMaxTokens(opts.worker.profile) ?? Math.max(1, Math.floor(budget.maxTokens / 8));
5695
5164
  const makeWorkerAgent = (rawProfile) => {
5696
5165
  const p = rawProfile ?? {};
5697
5166
  return {
@@ -5716,7 +5185,7 @@ async function superviseSurface(profile, task, opts) {
5716
5185
  maxLiveWorkers: opts.maxLiveWorkers ?? 1,
5717
5186
  perWorker: {
5718
5187
  maxIterations: innerTurns + 2,
5719
- maxTokens: opts.worker.maxTokens ?? 4e3
5188
+ maxTokens: workerMaxTokens
5720
5189
  },
5721
5190
  router,
5722
5191
  ...analysts ? {
@@ -5736,6 +5205,12 @@ async function superviseSurface(profile, task, opts) {
5736
5205
  completions: sp.iterations
5737
5206
  };
5738
5207
  }
5208
+ function profileMaxTokens(profile) {
5209
+ const value = profile.model?.metadata?.maxTokens;
5210
+ if (value === void 0) return void 0;
5211
+ if (!Number.isSafeInteger(value) || value < 1) throw new Error("superviseSurface: AgentProfile.model.metadata.maxTokens must be a positive safe integer");
5212
+ return value;
5213
+ }
5739
5214
  //#endregion
5740
5215
  //#region src/runtime/verifier-environment.ts
5741
5216
  const submitTool = {
@@ -6150,6 +5625,6 @@ function tail(s) {
6150
5625
  return s.slice(-400);
6151
5626
  }
6152
5627
  //#endregion
6153
- export { panel as $, runStrategyEvolution as A, materializeLocalMcp as At, equalKOnCost as B, GraphEdgeCapError as C, renderLeaderboardMarkdown as Ct, streamAgentTurn as D, defaultAuditorInstruction as Dt, collectAgentTurn as E, auditIntent as Et, SandboxRunAbortError as F, resolveMcpServerLaunch as Ft, createShapeRegistry as G, definePersona as H, openSandboxRun as I, resolveSecretEnv as It, InMemoryCorpus as J, registerShape as K, printBenchmarkReport as L, secretEnvOfMcpServer as Lt, assertStrategyContract as M, sanitizeMcpToolSchema as Mt, authorStrategy as N, envKeyProvider as Nt, discriminatingMeans as O, McpSpawnFault as Ot, strategyAuthorContract as P, mcpSecretEnvMetadataKey as Pt, loopUntil as Q, runBenchmark as R, runCoderChecks as S, renderLeaderboardHtml as St, runGraph as T, renderPairwiseMarkdown as Tt, runPersonified as U, trajectoryReport as V, builtinShapes as W, fanout as X, renderCorpusToInstructions as Y, flatWidenGate as Z, settledWorkerOut as _, deterministicCompletion as _t, localShell as a, buildSteerContext as at, analyzeTrace as b, leaderboard as bt, createVerifierEnvironment as c, inProcessSandboxClient as ct, worktreeFanout as d, resolveSandboxClient as dt, pipeline as et, EVIDENCE_MAX_CHARS as f, localSandboxClient as ft, composeWorkerEvidence as g, completionAuthorizes as gt, closingWorkerNote as h, loopDispatch as ht, jjWorkspace as i, assertTraceDerivedFindings as it, selectChampion as j, createMcpEnvironment as jt, pickChampion as k, connectStdioMcp as kt, failuresAnalyst as l, harvestCorpus as lt, VERIFY_TAIL_CHARS as m, loopCampaignDispatch as mt, makeFinding$1 as n, verify as nt, runInWorkspace as o, createScopeAnalyst as ot, NOTE_MAX_CHARS as p, inlineSandboxClient as pt, FileCorpus as q, gitWorkspace as r, widen as rt, createWaterfallCollector as s, registryScopeAnalyst as st, computeFindingId$1 as t, selectValidWinner as tt, superviseSurface as u, defineLeaderboard as ut, copyUntrackedIntoClone as v, sentinelCompletion as vt, defaultEdgeTraversalCap as w, renderLeaderboardSvg as wt, patchDelivered as x, pairwiseSignificance as xt, withUntrackedArtifacts as y, stopSentinel as yt, promotionGate as z };
5628
+ export { pipeline as $, assertStrategyContract as A, createMcpEnvironment as At, trajectoryReport as B, chatTransportExecutor as C, renderLeaderboardSvg as Ct, pickChampion as D, McpSpawnFault as Dt, discriminatingMeans as E, defaultAuditorInstruction as Et, openSandboxRun as F, resolveSecretEnv as Ft, registerShape as G, runPersonified as H, printBenchmarkReport as I, secretEnvOfMcpServer as It, renderCorpusToInstructions as J, FileCorpus as K, runBenchmark as L, strategyAuthorContract as M, envKeyProvider as Mt, strategyAuthorSystemPrompt as N, mcpSecretEnvMetadataKey as Nt, runStrategyEvolution as O, connectStdioMcp as Ot, SandboxRunAbortError as P, resolveMcpServerLaunch as Pt, panel as Q, promotionGate as R, runCoderChecks as S, renderLeaderboardMarkdown as St, createChatSessionStore as T, auditIntent as Tt, builtinShapes as U, definePersona as V, createShapeRegistry as W, flatWidenGate as X, fanout as Y, loopUntil as Z, settledWorkerOut as _, sentinelCompletion as _t, localShell as a, createScopeAnalyst as at, analyzeTrace as b, pairwiseSignificance as bt, createVerifierEnvironment as c, harvestCorpus as ct, worktreeFanout as d, localSandboxClient as dt, selectValidWinner as et, EVIDENCE_MAX_CHARS as f, inlineSandboxClient as ft, composeWorkerEvidence as g, deterministicCompletion as gt, closingWorkerNote as h, completionAuthorizes as ht, jjWorkspace as i, buildSteerContext as it, authorStrategy as j, sanitizeMcpToolSchema as jt, selectChampion as k, materializeLocalMcp as kt, failuresAnalyst as l, defineLeaderboard as lt, VERIFY_TAIL_CHARS as m, loopDispatch as mt, makeFinding$1 as n, widen as nt, runInWorkspace as o, registryScopeAnalyst as ot, NOTE_MAX_CHARS as p, loopCampaignDispatch as pt, InMemoryCorpus as q, gitWorkspace as r, assertTraceDerivedFindings as rt, createWaterfallCollector as s, inProcessSandboxClient as st, computeFindingId$1 as t, verify as tt, superviseSurface as u, resolveSandboxClient as ut, copyUntrackedIntoClone as v, stopSentinel as vt, chatWorkerSeam as w, renderPairwiseMarkdown as wt, patchDelivered as x, renderLeaderboardHtml as xt, withUntrackedArtifacts as y, leaderboard as yt, equalKOnCost as z };
6154
5629
 
6155
- //# sourceMappingURL=runtime-BzXz7OjS.js.map
5630
+ //# sourceMappingURL=runtime-hiAABiTk.js.map